[PATCH mptcp-next v12 1/4] mptcp: support MSG_ERRQUEUE on the parent socket
David Carlier <[email protected]> Thu, 30 Jul 2026 08:12:41 +0100
| Newsgroups | dev.linux.lists.mptcp |
|---|---|
| Message-ID | <[email protected]> |
Splice pending err skbs from each subflow's error queue onto the parent msk's error queue at error-report time, so poll() and recvmsg(MSG_ERRQUEUE) on the parent socket observe TX timestamps through the standard inet ABI. The splice filters by SO_EE_ORIGIN: TIMESTAMPING events forward to the parent because they are tied to user-handed data, not to a specific path; subflow-level ICMP errors are dropped because the legacy RECVERR ABI cannot meaningfully convey their per-subflow peer identity to single-path-aware userspace. Such events will be carried by a future MPTCP_RECERR channel. Forwarded events all go through sock_queue_err_skb(), which re-homes skb->sk onto the parent and charges sk_rmem_alloc, so the parent's error queue stays bounded by sk_rcvbuf and is dropped under rmem pressure (sk_rmem_alloc + truesize >= sk_rcvbuf), matching tcp's sk_rcvbuf-gated tx-timestamp path and ip_icmp_error() / ipv6_icmp_error(). The MSG_ERRQUEUE branch of mptcp_recvmsg() forwards to inet_recv_error() directly, and poll() advertises EPOLLERR purely on the parent's sk_err / sk_error_queue, matching tcp_poll(). Signed-off-by: David Carlier <[email protected]> --- net/mptcp/protocol.c | 54 ++++++++++++++++++++++++++++++++++++-------- 1 file changed, 45 insertions(+), 9 deletions(-) diff --git a/net/mptcp/protocol.c b/net/mptcp/protocol.c index 90dc894cb976..ade017f9ce5b 100644 --- a/net/mptcp/protocol.c +++ b/net/mptcp/protocol.c @@ -11,6 +11,7 @@ #include <linux/netdevice.h> #include <linux/sched/signal.h> #include <linux/atomic.h> +#include <linux/errqueue.h> #include <net/aligned_data.h> #include <net/rps.h> #include <net/sock.h> @@ -901,21 +902,52 @@ static bool __mptcp_ofo_queue(struct mptcp_sock *msk) return moved; } +static bool mptcp_errqueue_skb_forwardable(const struct sk_buff *skb) +{ + /* Subflow-level ICMP errors are dropped: the legacy RECVERR ABI + * cannot convey their per-subflow peer identity. + */ + return SKB_EXT_ERR(skb)->ee.ee_origin == SO_EE_ORIGIN_TIMESTAMPING; +} + +static bool __mptcp_subflow_splice_errqueue(struct sock *sk, struct sock *ssk) +{ + struct sk_buff *skb; + bool moved = false; + + while ((skb = skb_dequeue(&ssk->sk_error_queue))) { + /* sock_queue_err_skb() re-homes skb->sk onto the parent and + * charges sk_rmem_alloc, bounding the queue by sk_rcvbuf. + */ + if (!mptcp_errqueue_skb_forwardable(skb) || + sock_queue_err_skb(sk, skb)) { + kfree_skb(skb); + continue; + } + moved = true; + } + + return moved; +} + static bool __mptcp_subflow_error_report(struct sock *sk, struct sock *ssk) { + bool propagated = false; int ssk_state; + bool report; int err; + report = __mptcp_subflow_splice_errqueue(sk, ssk); + /* only propagate errors on fallen-back sockets or * on MPC connect */ if (sk->sk_state != TCP_SYN_SENT && !__mptcp_check_fallback(mptcp_sk(sk))) - return false; + goto out; err = sock_error(ssk); if (!err) - return false; - + goto out; /* We need to propagate only transition to CLOSE state. * Orphaned socket will see such state change via * subflow_sched_work_if_closed() and that path will properly @@ -925,11 +957,15 @@ static bool __mptcp_subflow_error_report(struct sock *sk, struct sock *ssk) if (ssk_state == TCP_CLOSE && !sock_flag(sk, SOCK_DEAD)) mptcp_set_state(sk, ssk_state); WRITE_ONCE(sk->sk_err, -err); + report = propagated = true; - /* This barrier is coupled with smp_rmb() in mptcp_poll() */ - smp_wmb(); - sk_error_report(sk); - return true; +out: + if (report) { + /* This barrier is coupled with smp_rmb() in mptcp_poll() */ + smp_wmb(); + sk_error_report(sk); + } + return propagated; } void __mptcp_error_report(struct sock *sk) @@ -2363,7 +2399,6 @@ static int mptcp_recvmsg(struct sock *sk, struct msghdr *msg, size_t len, int target; long timeo; - /* MSG_ERRQUEUE is really a no-op till we support IP_RECVERR */ if (unlikely(flags & MSG_ERRQUEUE)) return inet_recv_error(sk, msg, len); @@ -4493,7 +4528,8 @@ static __poll_t mptcp_poll(struct file *file, struct socket *sock, /* This barrier is coupled with smp_wmb() in __mptcp_error_report() */ smp_rmb(); - if (READ_ONCE(sk->sk_err)) + if (READ_ONCE(sk->sk_err) || + !skb_queue_empty_lockless(&sk->sk_error_queue)) mask |= EPOLLERR; return mask; -- 2.53.0