Re: [PATCH 1/2] net/socket: Record preference for synchronous wakeups

Shrikanth Hegde <[email protected]> Tue, 21 Jul 2026 10:30:11 +0530
Newsgroups org.kernel.vger.linux-sctp,org.kernel.vger.linux-kernel,org.kernel.vger.netdev
Message-ID <[email protected]>
Hi Srikar.

On 7/14/26 7:09 AM, Srikar Dronamraju wrote:
> Scheduler differentiates between affine and non-affine wakeups by the
> way of sync flags. Scheduler prefers to pull the tasks towards the waker
> if the sync flag is set.
> 
> In some cases, socket APIs are blindly requesting sync wakeups. This may
> cause load-balance issues and non-optimal performance.
> 
> Record whether the most recent blocking socket operation could benefit
> from synchronous wakeups. Subsequent readiness notifications use this
> hint to determine whether WF_SYNC should be propagated.
> 

What you mean by recent? Was it info on past set of packets?
Could you please explain the flow a bit?

Shouldn't it be
- if this socket has been defined as nonblock it shouldn't use sync
   always?

> The flag is advisory and affects only wakeup placement decisions.
> 
> Signed-off-by: Srikar Dronamraju <[email protected]>
> ---
>   include/net/sock.h |  1 +
>   net/socket.c       | 56 +++++++++++++++++++++++++++++++++++++++-------
>   2 files changed, 49 insertions(+), 8 deletions(-)
> 
> diff --git a/include/net/sock.h b/include/net/sock.h
> index 51185222aac2..acc6b1976dc4 100644
> --- a/include/net/sock.h
> +++ b/include/net/sock.h
> @@ -1022,6 +1022,7 @@ enum sock_flags {
>   	SOCK_RCVMARK, /* Receive SO_MARK  ancillary data with packet */
>   	SOCK_RCVPRIORITY, /* Receive SO_PRIORITY ancillary data with packet */
>   	SOCK_TIMESTAMPING_ANY, /* Copy of sk_tsflags & TSFLAGS_ANY */
> +	SOCK_SYNC_WAKEUP, /* Prefer synchronous socket wakeups */
>   };
>   
>   #define SK_FLAGS_TIMESTAMP ((1UL << SOCK_TIMESTAMP) | (1UL << SOCK_TIMESTAMPING_RX_SOFTWARE))
> diff --git a/net/socket.c b/net/socket.c
> index 63c69a0fa74e..0bcb57ae490e 100644
> --- a/net/socket.c
> +++ b/net/socket.c
> @@ -1198,15 +1198,27 @@ static void sock_splice_eof(struct file *file)
>   		ops->splice_eof(sock);
>   }
>   
> +static inline void sock_update_sync_wakeup(struct sock *sk, bool nonblock)
> +{
> +	if (unlikely(!sk))
> +		return;
> +
> +	if (nonblock) {
> +		if (sock_flag(sk, SOCK_SYNC_WAKEUP))
> +			sock_reset_flag(sk, SOCK_SYNC_WAKEUP);
> +	} else {
> +		if (!sock_flag(sk, SOCK_SYNC_WAKEUP))
> +			sock_set_flag(sk, SOCK_SYNC_WAKEUP);
> +	}
> +}

nit: You can combine two if statements. A bit easier to read.


static inline void sock_update_sync_wakeup(struct sock *sk, bool nonblock)
{
	if (unlikely(!sk))
		return;

	if (nonblock && sock_flag(sk, SOCK_SYNC_WAKEUP))
		sock_reset_flag(sk, SOCK_SYNC_WAKEUP);
	else if (!nonblock && !sock_flag(sk, SOCK_SYNC_WAKEUP))
		sock_set_flag(sk, SOCK_SYNC_WAKEUP);
}

> +
>   static ssize_t sock_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   {
>   	struct file *file = iocb->ki_filp;
>   	struct socket *sock = file->private_data;
>   	struct msghdr msg = {.msg_iter = *to};
>   	ssize_t res;
> -
> -	if (file->f_flags & O_NONBLOCK || (iocb->ki_flags & IOCB_NOWAIT))
> -		msg.msg_flags = MSG_DONTWAIT;
> +	bool nonblock;
>   
>   	if (iocb->ki_pos != 0)
>   		return -ESPIPE;
> @@ -1214,6 +1226,11 @@ static ssize_t sock_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   	if (!iov_iter_count(to))	/* Match SYS5 behaviour */
>   		return 0;
>   
> +	nonblock = (file->f_flags & O_NONBLOCK) || (iocb->ki_flags & IOCB_NOWAIT);
> +	if (nonblock)
> +		msg.msg_flags = MSG_DONTWAIT;
> +
> +	sock_update_sync_wakeup(sock->sk, nonblock);
>   	res = sock_recvmsg(sock, &msg, msg.msg_flags);
>   	*to = msg.msg_iter;
>   	return res;
> @@ -1225,13 +1242,17 @@ static ssize_t sock_write_iter(struct kiocb *iocb, struct iov_iter *from)
>   	struct socket *sock = file->private_data;
>   	struct msghdr msg = {.msg_iter = *from};
>   	ssize_t res;
> +	bool nonblock;
>   
>   	if (iocb->ki_pos != 0)
>   		return -ESPIPE;
>   
> -	if (file->f_flags & O_NONBLOCK || (iocb->ki_flags & IOCB_NOWAIT))
> +	nonblock = (file->f_flags & O_NONBLOCK) || (iocb->ki_flags & IOCB_NOWAIT);
> +	if (nonblock)
>   		msg.msg_flags = MSG_DONTWAIT;
>   
> +	sock_update_sync_wakeup(sock->sk, nonblock);
> +
>   	if (sock->type == SOCK_SEQPACKET)
>   		msg.msg_flags |= MSG_EOR;
>   
> @@ -2221,6 +2242,7 @@ int __sys_sendto(int fd, void __user *buff, size_t len, unsigned int flags,
>   	struct sockaddr_storage address;
>   	int err;
>   	struct msghdr msg;
> +	bool nonblock;
>   
>   	err = import_ubuf(ITER_SOURCE, buff, len, &msg.msg_iter);
>   	if (unlikely(err))
> @@ -2246,8 +2268,11 @@ int __sys_sendto(int fd, void __user *buff, size_t len, unsigned int flags,
>   		msg.msg_namelen = addr_len;
>   	}
>   	flags &= ~MSG_INTERNAL_SENDMSG_FLAGS;
> -	if (sock->file->f_flags & O_NONBLOCK)
> +	nonblock = (sock->file->f_flags & O_NONBLOCK);
> +	if (nonblock)
>   		flags |= MSG_DONTWAIT;
> +
> +	sock_update_sync_wakeup(sock->sk, nonblock);
>   	msg.msg_flags = flags;
>   	return __sock_sendmsg(sock, &msg);
>   }
> @@ -2284,6 +2309,7 @@ int __sys_recvfrom(int fd, void __user *ubuf, size_t size, unsigned int flags,
>   	};
>   	struct socket *sock;
>   	int err, err2;
> +	bool nonblock;
>   
>   	err = import_ubuf(ITER_DEST, ubuf, size, &msg.msg_iter);
>   	if (unlikely(err))
> @@ -2297,8 +2323,11 @@ int __sys_recvfrom(int fd, void __user *ubuf, size_t size, unsigned int flags,
>   	if (unlikely(!sock))
>   		return -ENOTSOCK;
>   
> -	if (sock->file->f_flags & O_NONBLOCK)
> +	nonblock = (sock->file->f_flags & O_NONBLOCK);
> +	if (nonblock)
>   		flags |= MSG_DONTWAIT;
> +
> +	sock_update_sync_wakeup(sock->sk, nonblock);
>   	err = sock_recvmsg(sock, &msg, flags);
>   
>   	if (err >= 0 && addr != NULL) {
> @@ -2634,6 +2663,7 @@ static int ____sys_sendmsg(struct socket *sock, struct msghdr *msg_sys,
>   	unsigned char *ctl_buf = ctl;
>   	int ctl_len;
>   	ssize_t err;
> +	bool nonblock;
>   
>   	err = -ENOBUFS;
>   
> @@ -2666,8 +2696,12 @@ static int ____sys_sendmsg(struct socket *sock, struct msghdr *msg_sys,
>   	flags &= ~MSG_INTERNAL_SENDMSG_FLAGS;
>   	msg_sys->msg_flags = flags;
>   
> -	if (sock->file->f_flags & O_NONBLOCK)
> +	nonblock = (sock->file->f_flags & O_NONBLOCK);
> +	if (nonblock)
>   		msg_sys->msg_flags |= MSG_DONTWAIT;
> +
> +	sock_update_sync_wakeup(sock->sk, nonblock);
> +
>   	/*
>   	 * If this is sendmmsg() and current destination address is same as
>   	 * previously succeeded address, omit asking LSM's decision.
> @@ -2887,6 +2921,7 @@ static int ____sys_recvmsg(struct socket *sock, struct msghdr *msg_sys,
>   	unsigned long cmsg_ptr;
>   	int len;
>   	ssize_t err;
> +	bool nonblock;
>   
>   	msg_sys->msg_name = &addr;
>   	cmsg_ptr = (unsigned long)msg_sys->msg_control;
> @@ -2895,9 +2930,12 @@ static int ____sys_recvmsg(struct socket *sock, struct msghdr *msg_sys,
>   	/* We assume all kernel code knows the size of sockaddr_storage */
>   	msg_sys->msg_namelen = 0;
>   
> -	if (sock->file->f_flags & O_NONBLOCK)
> +	nonblock = (sock->file->f_flags & O_NONBLOCK);
> +	if (nonblock)
>   		flags |= MSG_DONTWAIT;
>   
> +	sock_update_sync_wakeup(sock->sk, nonblock);
> +
>   	if (unlikely(nosec))
>   		err = sock_recvmsg_nosec(sock, msg_sys, flags);
>   	else
> @@ -3056,6 +3094,8 @@ static int do_recvmmsg(int fd, struct mmsghdr __user *mmsg,
>   		if (flags & MSG_WAITFORONE)
>   			flags |= MSG_DONTWAIT;
>   
> +		sock_update_sync_wakeup(sock->sk, flags & MSG_WAITFORONE);
> +
>   		if (timeout) {
>   			ktime_get_ts64(&timeout64);
>   			*timeout = timespec64_sub(end_time, timeout64);