Re: [PATCH net v2 1/1] ip6_tunnel: snapshot encap in xmit
Kuniyuki Iwashima <[email protected]>
| Newsgroups | org.kernel.vger.netdev |
|---|---|
| Message-ID | <[email protected]> |
From: Ren Wei <[email protected]> Date: Sat, 8 Aug 2026 16:40:49 +0800 > From: Zixuan Chai <[email protected]> > > ip6_tnl_changelink() can update encapsulation parameters while the > netdevice is transmitting packets. ip6_tnl_xmit() can calculate packet > headroom with t->encap_hlen and later build an encapsulation header from > the live t->encap. A concurrent update can change the encapsulation > header between these accesses and make skb_push() underflow the skb head. > > Take a local snapshot of t->encap before calculating the encapsulation > header length. This intorduce per-skb cost in the fast path for unlikely changelink. Right approach is to convert it to RCU pointer (and remove synchronize_net() there). 0ba269933f73 geneve: convert config to RCU-protected pointer 777434f53e77 geneve: pass geneve_config pointer to helper functions > Use that same snapshot for headroom accounting, metadata > validation, and build_header(). This keeps all encapsulation decisions > for an skb consistent even if changelink updates the live configuration. > > Fixes: b3a27b519b22 ("ip6_tunnel: Add support for fou/gue encapsulation") > Cc: [email protected] > Reported-by: Vega <[email protected]> > Assisted-by: Codex:gpt-5.4 > Signed-off-by: Zixuan Chai <[email protected]> > Signed-off-by: Ren Wei <[email protected]> > --- > include/net/ip6_tunnel.h | 10 +++++----- > net/ipv6/ip6_tunnel.c | 21 ++++++++++++++++----- > 2 files changed, 21 insertions(+), 10 deletions(-) > > diff --git a/include/net/ip6_tunnel.h b/include/net/ip6_tunnel.h > index b99805ee2fd1..6e76e50a4406 100644 > --- a/include/net/ip6_tunnel.h > +++ b/include/net/ip6_tunnel.h > @@ -106,22 +106,22 @@ static inline int ip6_encap_hlen(struct ip_tunnel_encap *e) > return hlen; > } > > -static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip6_tnl *t, > +static inline int ip6_tnl_encap(struct sk_buff *skb, struct ip_tunnel_encap *e, > u8 *protocol, struct flowi6 *fl6) > { > const struct ip6_tnl_encap_ops *ops; > int ret = -EINVAL; > > - if (t->encap.type == TUNNEL_ENCAP_NONE) > + if (e->type == TUNNEL_ENCAP_NONE) > return 0; > > - if (t->encap.type >= MAX_IPTUN_ENCAP_OPS) > + if (e->type >= MAX_IPTUN_ENCAP_OPS) > return -EINVAL; > > rcu_read_lock(); > - ops = rcu_dereference(ip6tun_encaps[t->encap.type]); > + ops = rcu_dereference(ip6tun_encaps[e->type]); > if (likely(ops && ops->build_header)) > - ret = ops->build_header(skb, &t->encap, protocol, fl6); > + ret = ops->build_header(skb, e, protocol, fl6); > rcu_read_unlock(); > > return ret; > diff --git a/net/ipv6/ip6_tunnel.c b/net/ipv6/ip6_tunnel.c > index ebf83f090376..d47757e8a388 100644 > --- a/net/ipv6/ip6_tunnel.c > +++ b/net/ipv6/ip6_tunnel.c > @@ -1102,6 +1102,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield, > __u8 proto) > { > struct ip6_tnl *t = netdev_priv(dev); > + struct ip_tunnel_encap ipencap; > struct net *net = t->net; > struct ipv6hdr *ipv6h; > struct ipv6_tel_txoption opt; > @@ -1109,10 +1110,11 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield, > struct net_device *tdev; > int err_count, mtu; > unsigned int eth_hlen = t->dev->type == ARPHRD_ETHER ? ETH_HLEN : 0; > - unsigned int psh_hlen = sizeof(struct ipv6hdr) + t->encap_hlen; > - unsigned int max_headroom = psh_hlen; > + unsigned int max_headroom; > __be16 payload_protocol; > bool use_cache = false; > + unsigned int psh_hlen; > + int encap_hlen; > u8 hop_limit; > int err = -1; > > @@ -1202,6 +1204,15 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield, > t->parms.name); > goto tx_err_dst_release; > } > + > + /* Can tear, but hlen and build_header() use the same snapshot. */ > + ipencap = data_race(t->encap); > + encap_hlen = ip6_encap_hlen(&ipencap); > + if (unlikely(encap_hlen < 0)) > + goto tx_err_dst_release; > + psh_hlen = sizeof(struct ipv6hdr) + encap_hlen; > + max_headroom = psh_hlen; > + > mtu = dst6_mtu(dst) - eth_hlen - psh_hlen - t->tun_hlen; > if (encap_limit >= 0) { > max_headroom += 8; > @@ -1251,7 +1262,7 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield, > } > > if (t->parms.collect_md) { > - if (t->encap.type != TUNNEL_ENCAP_NONE) > + if (ipencap.type != TUNNEL_ENCAP_NONE) > goto tx_err_dst_release; > } else { > if (use_cache && ndst) > @@ -1272,10 +1283,10 @@ int ip6_tnl_xmit(struct sk_buff *skb, struct net_device *dev, __u8 dsfield, > * needed_headroom if necessary. > */ > max_headroom = LL_RESERVED_SPACE(tdev) + sizeof(struct ipv6hdr) > - + dst->header_len + t->hlen; > + + dst->header_len + t->tun_hlen + encap_hlen; > ip_tunnel_adj_headroom(dev, max_headroom); > > - err = ip6_tnl_encap(skb, t, &proto, fl6); > + err = ip6_tnl_encap(skb, &ipencap, &proto, fl6); > if (err) > return err; > > -- > 2.34.1