Re: [PATCH net v2 1/1] ip6_tunnel: snapshot encap in xmit

Kuniyuki Iwashima <[email protected]>
Newsgroups org.kernel.vger.netdev
Message-ID <[email protected]>
From: Ren Wei <[email protected]>
Date: Sat,  8 Aug 2026 16:40:49 +0800
> From: Zixuan Chai <[email protected]>
> 
> ip6_tnl_changelink() can update encapsulation parameters while the
> netdevice is transmitting packets. ip6_tnl_xmit() can calculate packet
> headroom with t->encap_hlen and later build an encapsulation header from
> the live t->encap. A concurrent update can change the encapsulation
> header between these accesses and make skb_push() underflow the skb head.
> 
> Take a local snapshot of t->encap before calculating the encapsulation
> header length.

This intorduce per-skb cost in the fast path for unlikely changelink.

Right approach is to convert it to RCU pointer (and remove
synchronize_net() there).

0ba269933f73 geneve: convert config to RCU-protected pointer
777434f53e77 geneve: pass geneve_config pointer to helper functions


> Use that same snapshot for headroom accounting, metadata
> validation, and build_header(). This keeps all encapsulation decisions
> for an skb consistent even if changelink updates the live configuration.
> 
> Fixes: b3a27b519b22 ("ip6_tunnel: Add support for fou/gue encapsulation")
> Cc: [email protected]
> Reported-by: Vega <[email protected]>
> Assisted-by: Codex:gpt-5.4
> Signed-off-by: Zixuan Chai <[email protected]>
> Signed-off-by: Ren Wei <[email protected]>
> ---
>  include/net/ip6_tunnel.h | 10 +++++-----
>  net/ipv6/ip6_tunnel.c    | 21 ++++++++++++++++-----
>  2 files changed, 21 insertions(+), 10 deletions(-)
> 
> diff --git a/include/net/ip6_tunnel.h b/include/net/ip6_tunnel.h
> index b99805ee2fd1..6e76e50a4406 100644
> --- a/include/net/ip6_tunnel.h
> +++ b/include/net/ip6_tunnel.h
> @@ -106,22 +106,22 @@ static inline int ip6_encap_hlen(struct ip_tunnel_encap *e)
>  	return hlen;
>  }
>  
> -static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip6_tnl *t,
> +static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip_tunnel_encap *e,
>  				u8 *protocol, struct flowi6 *fl6)
>  {
>  	const struct ip6_tnl_encap_ops *ops;
>  	int ret = -EINVAL;
>  
> -	if (t->encap.type == TUNNEL_ENCAP_NONE)
> +	if (e->type == TUNNEL_ENCAP_NONE)
>  		return 0;
>  
> -	if (t->encap.type >= MAX_IPTUN_ENCAP_OPS)
> +	if (e->type >= MAX_IPTUN_ENCAP_OPS)
>  		return -EINVAL;
>  
>  	rcu_read_lock();
> -	ops = rcu_dereference(ip6tun_encaps[t->encap.type]);
> +	ops = rcu_dereference(ip6tun_encaps[e->type]);
>  	if (likely(ops && ops->build_header))
> -		ret = ops->build_header(skb, &t->encap, protocol, fl6);
> +		ret = ops->build_header(skb, e, protocol, fl6);
>  	rcu_read_unlock();
>  
>  	return ret;
> diff --git a/net/ipv6/ip6_tunnel.c b/net/ipv6/ip6_tunnel.c
> index ebf83f090376..d47757e8a388 100644
> --- a/net/ipv6/ip6_tunnel.c
> +++ b/net/ipv6/ip6_tunnel.c
> @@ -1102,6 +1102,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  		 __u8 proto)
>  {
>  	struct ip6_tnl *t = netdev_priv(dev);
> +	struct ip_tunnel_encap ipencap;
>  	struct net *net = t->net;
>  	struct ipv6hdr *ipv6h;
>  	struct ipv6_tel_txoption opt;
> @@ -1109,10 +1110,11 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  	struct net_device *tdev;
>  	int err_count, mtu;
>  	unsigned int eth_hlen = t->dev->type == ARPHRD_ETHER ? ETH_HLEN : 0;
> -	unsigned int psh_hlen = sizeof(struct ipv6hdr) + t->encap_hlen;
> -	unsigned int max_headroom = psh_hlen;
> +	unsigned int max_headroom;
>  	__be16 payload_protocol;
>  	bool use_cache = false;
> +	unsigned int psh_hlen;
> +	int encap_hlen;
>  	u8 hop_limit;
>  	int err = -1;
>  
> @@ -1202,6 +1204,15 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  				     t->parms.name);
>  		goto tx_err_dst_release;
>  	}
> +
> +	/* Can tear, but hlen and build_header() use the same snapshot. */
> +	ipencap = data_race(t->encap);
> +	encap_hlen = ip6_encap_hlen(&ipencap);
> +	if (unlikely(encap_hlen < 0))
> +		goto tx_err_dst_release;
> +	psh_hlen = sizeof(struct ipv6hdr) + encap_hlen;
> +	max_headroom = psh_hlen;
> +
>  	mtu = dst6_mtu(dst) - eth_hlen - psh_hlen - t->tun_hlen;
>  	if (encap_limit >= 0) {
>  		max_headroom += 8;
> @@ -1251,7 +1262,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  	}
>  
>  	if (t->parms.collect_md) {
> -		if (t->encap.type != TUNNEL_ENCAP_NONE)
> +		if (ipencap.type != TUNNEL_ENCAP_NONE)
>  			goto tx_err_dst_release;
>  	} else {
>  		if (use_cache && ndst)
> @@ -1272,10 +1283,10 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield,
>  	 * needed_headroom if necessary.
>  	 */
>  	max_headroom = LL_RESERVED_SPACE(tdev) + sizeof(struct ipv6hdr)
> -			+ dst->header_len + t->hlen;
> +			+ dst->header_len + t->tun_hlen + encap_hlen;
>  	ip_tunnel_adj_headroom(dev, max_headroom);
>  
> -	err = ip6_tnl_encap(skb, t, &proto, fl6);
> +	err = ip6_tnl_encap(skb, &ipencap, &proto, fl6);
>  	if (err)
>  		return err;
>  
> -- 
> 2.34.1
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.