Re: [PATCH v3 1/4] IPC: Added two new system call mq_recvmmsg() and mq_sendmmsg()

Pavel Tikhomirov <[email protected]> Fri, 12 Jun 2026 13:49:52 +0200
Newsgroups dev.linux.lists.criu
Message-ID <[email protected]>
Hi, please see inline review comments below:

On 6/12/26 07:20, Mathura_Kumar wrote:
> Implement two new POSIX message queue system calls, mq_sendmmsg() and
> mq_recvmmsg(), analogous to the existing sendmmsg()/recvmmsg() socket
> system calls.These allow sending and receiving or peek multiple messages in a
> single syscall, reducing the overhead of repeated context switches per discrete
> message.
> 
> It contains the core implementation of both system calls as
> part of a larger patchset.
> 
>   int mq_sendmmsg(mqd_t mqdes, struct mq_mmsg_attrs  *attrs,
>                 unsigned int attrs_len, unsigned long start_idx,
> 		const struct __kernel_timespec *u_abs_timeout);
> 
>   int mq_recvmmsg(mqd_t mqdes, struct mq_mmsg_attrs *attrs,
>                  unsigned int attrs_len, unsigned int flags,
>                  unsigned long start_idx, const struct __kernel_timespec *u_abs_timeout);
> 
> Implementation complete details available under
> mq_recvmmsg.rst and mq_sendmmsg.rst
> 
> Signed-off-by: Mathura_Kumar <[email protected]>
> ---
>  include/linux/compat.h            |   9 +-
>  include/linux/syscalls.h          |   8 +
>  include/uapi/asm-generic/unistd.h |   9 +-
>  include/uapi/linux/mqueue.h       |  27 +-
>  ipc/mqueue.c                      | 506 ++++++++++++++++++++++++++++--
>  ipc/msg.c                         |   2 +-
>  ipc/msgutil.c                     |  51 ++-
>  ipc/util.h                        |   3 +-
>  kernel/sys_ni.c                   |   6 +
>  9 files changed, 558 insertions(+), 63 deletions(-)
> 
> diff --git a/include/linux/compat.h b/include/linux/compat.h
> index 8da0a15c95f4..200d6d9b80e5 100644
> --- a/include/linux/compat.h
> +++ b/include/linux/compat.h
> @@ -22,6 +22,7 @@
>  #include <asm/compat.h>
>  #include <asm/siginfo.h>
>  #include <asm/signal.h>
> +#include <linux/mqueue.h>
>  
>  #ifdef CONFIG_ARCH_HAS_SYSCALL_WRAPPER
>  /*
> @@ -805,8 +806,12 @@ asmlinkage long compat_sys_pwritev64v2(unsigned long fd,
>  		const struct iovec __user *vec,
>  		unsigned long vlen, loff_t pos, rwf_t flags);
>  #endif
> -
> -
> +asmlinkage long compat_sys_mq_sendmmsg(mqd_t mqdes, struct compat_mq_mmsg_attrs __user *attrs,
> +						unsigned int attrs_len, unsigned int flags, unsigned long start_index,

nit: Alignment is a bit off (not only this place).

> A very commonly used style is to align descendants to a function open parenthesis.

https://www.kernel.org/doc/html/v5.8/process/coding-style.html#breaking-long-lines-and-strings

So preferably it should be aligned just after parenthesis:

compat_sys_mq_sendmmsg(mqd_t mqdes, ...
		       unsigned int attrs_len,
                      ^^

note: Normally you use offset/8 tabs + offset%8 spaces for this.

> +						const struct __kernel_timespec __user *abs_timeout);
> +asmlinkage long compat_sys_mq_recvmmsg(mqd_t mqdes, struct compat_mq_mmsg_attrs __user *attrs,
> +						unsigned int attrs_len, unsigned int flags, unsigned long start_index,
> +						const struct __kernel_timespec __user *abs_timeout);
>  /*
>   * Deprecated system calls which are still defined in
>   * include/uapi/asm-generic/unistd.h and wanted by >= 1 arch
> diff --git a/include/linux/syscalls.h b/include/linux/syscalls.h
> index 4fb7291f54b6..b3297f9dd126 100644
> --- a/include/linux/syscalls.h
> +++ b/include/linux/syscalls.h
> @@ -79,6 +79,7 @@ struct mnt_id_req;
>  struct ns_id_req;
>  struct xattr_args;
>  struct file_attr;
> +struct mq_mmsg_attrs;
>  
>  #include <linux/types.h>
>  #include <linux/aio_abi.h>
> @@ -93,6 +94,7 @@ struct file_attr;
>  #include <linux/key.h>
>  #include <linux/personality.h>
>  #include <trace/syscall.h>
> +#include <linux/mqueue.h>
>  
>  #ifdef CONFIG_ARCH_HAS_SYSCALL_WRAPPER
>  /*
> @@ -739,7 +741,13 @@ asmlinkage long sys_sysinfo(struct sysinfo __user *info);
>  asmlinkage long sys_mq_open(const char __user *name, int oflag, umode_t mode, struct mq_attr __user *attr);
>  asmlinkage long sys_mq_unlink(const char __user *name);
>  asmlinkage long sys_mq_timedsend(mqd_t mqdes, const char __user *msg_ptr, size_t msg_len, unsigned int msg_prio, const struct __kernel_timespec __user *abs_timeout);
> +asmlinkage long sys_mq_sendmmsg(mqd_t mqdes, struct mq_mmsg_attrs __user *attrs,
> +					unsigned int attrs_len, unsigned int flags, unsigned long start_index,
> +					const struct __kernel_timespec __user *abs_timeout);
>  asmlinkage long sys_mq_timedreceive(mqd_t mqdes, char __user *msg_ptr, size_t msg_len, unsigned int __user *msg_prio, const struct __kernel_timespec __user *abs_timeout);
> +asmlinkage long sys_mq_recvmmsg(mqd_t mqdes, struct mq_mmsg_attrs __user *attrs,
> +					unsigned int attrs_len, unsigned int flags, unsigned long start_index,
> +					const struct __kernel_timespec __user *abs_timeout);
>  asmlinkage long sys_mq_notify(mqd_t mqdes, const struct sigevent __user *notification);
>  asmlinkage long sys_mq_getsetattr(mqd_t mqdes, const struct mq_attr __user *mqstat, struct mq_attr __user *omqstat);
>  asmlinkage long sys_mq_timedreceive_time32(mqd_t mqdes,
> diff --git a/include/uapi/asm-generic/unistd.h b/include/uapi/asm-generic/unistd.h
> index a627acc8fb5f..1d06486d3aa5 100644
> --- a/include/uapi/asm-generic/unistd.h
> +++ b/include/uapi/asm-generic/unistd.h
> @@ -863,9 +863,14 @@ __SYSCALL(__NR_listns, sys_listns)
>  #define __NR_rseq_slice_yield 471
>  __SYSCALL(__NR_rseq_slice_yield, sys_rseq_slice_yield)
>  
> -#undef __NR_syscalls
> -#define __NR_syscalls 472
> +#define __NR_mq_recvmmsg 472
> +__SC_COMP(__NR_mq_recvmmsg, sys_mq_recvmmsg, compat_sys_mq_recvmmsg)
> +
> +#define __NR_mq_sendmmsg 473
> +__SC_COMP(__NR_mq_sendmmsg, sys_mq_sendmmsg, compat_sys_mq_sendmmsg)
>  
> +#undef __NR_syscalls
> +#define __NR_syscalls 474
>  /*
>   * 32 bit systems traditionally used different
>   * syscalls for off_t and loff_t arguments, while
> diff --git a/include/uapi/linux/mqueue.h b/include/uapi/linux/mqueue.h
> index b516b66840ad..9b2e8539472c 100644
> --- a/include/uapi/linux/mqueue.h
> +++ b/include/uapi/linux/mqueue.h
> @@ -18,8 +18,9 @@
>  
>  #ifndef _LINUX_MQUEUE_H
>  #define _LINUX_MQUEUE_H
> -
> +#include <linux/uio.h>
>  #include <linux/types.h>
> +#include <asm/compat.h>
>  
>  #define MQ_PRIO_MAX 	32768
>  /* per-uid limit of kernel memory used by mqueue, in bytes */
> @@ -33,6 +34,30 @@ struct mq_attr {
>  	__kernel_long_t	__reserved[4];	/* ignored for input, zeroed for output */
>  };
>  
> +struct mq_msg_attrs {
> +	__kernel_size_t msg_len;
> +	unsigned int __user *msg_prio;
> +	void __user *msg_ptr;
> +};
> +
> +struct mq_mmsg_attrs {
> +	struct iovec __user *msg_attrs_vec;
> +	__kernel_size_t vlen;
> +	int __user  *ret;
> +};
> +
> +struct compat_msg_attrs {
> +	compat_size_t msg_len;
> +	compat_uptr_t msg_prio;
> +	compat_uptr_t msg_ptr;
> +};
> +
> +struct compat_mq_mmsg_attrs {
> +	struct compat_iovec __user *msg_attrs_vec;
> +	compat_size_t vlen;
> +	compat_uptr_t ret;
> +};
> +
>  /*
>   * SIGEV_THREAD implementation:
>   * SIGEV_THREAD must be implemented in user space. If SIGEV_THREAD is passed
> diff --git a/ipc/mqueue.c b/ipc/mqueue.c
> index 4798b375972b..e3e83eb9e7bc 100644
> --- a/ipc/mqueue.c
> +++ b/ipc/mqueue.c
> @@ -12,6 +12,8 @@
>   * Audit:                   George Wilson           ([email protected])
>   */
>  
> +#include "linux/compat.h"
> +#include "linux/types.h"

The "relative" include is likely unintentional here, please replace "" to <>.

>  #include <linux/capability.h>
>  #include <linux/init.h>
>  #include <linux/pagemap.h>
> @@ -38,6 +40,8 @@
>  #include <linux/sched/wake_q.h>
>  #include <linux/sched/signal.h>
>  #include <linux/sched/user.h>
> +#include <linux/uio.h>
> +#include <linux/uaccess.h>
>  
>  #include <net/sock.h>
>  #include "util.h"
> @@ -54,6 +58,10 @@ struct mqueue_fs_context {
>  #define SEND		0
>  #define RECV		1
>  
> +#define MQ_PEEK     0x02
> +#define MQ_RECV     0x04
> +#define MQ_VALID_FLAGS (MQ_PEEK | MQ_RECV)
> +
>  #define STATE_NONE	0
>  #define STATE_READY	1
>  
> @@ -1034,16 +1042,15 @@ static inline void pipelined_receive(struct wake_q_head *wake_q,
>  	__pipelined_op(wake_q, info, sender);
>  }
>  
> -static int do_mq_timedsend(mqd_t mqdes, const char __user *u_msg_ptr,
> -		size_t msg_len, unsigned int msg_prio,
> -		struct timespec64 *ts)
> +static int do_mq_sendmsg(mqd_t mqdes, const char __user *u_msg_ptr,
> +						size_t msg_len, unsigned int msg_prio,
> +						const struct timespec64 *ts, ktime_t *timeout)
>  {
>  	struct inode *inode;
>  	struct ext_wait_queue wait;
>  	struct ext_wait_queue *receiver;
>  	struct msg_msg *msg_ptr;
>  	struct mqueue_inode_info *info;
> -	ktime_t expires, *timeout = NULL;
>  	struct posix_msg_tree_node *new_leaf = NULL;
>  	int ret = 0;
>  	DEFINE_WAKE_Q(wake_q);
> @@ -1051,11 +1058,6 @@ static int do_mq_timedsend(mqd_t mqdes, const char __user *u_msg_ptr,
>  	if (unlikely(msg_prio >= (unsigned long) MQ_PRIO_MAX))
>  		return -EINVAL;
>  
> -	if (ts) {
> -		expires = timespec64_to_ktime(*ts);
> -		timeout = &expires;
> -	}
> -
>  	audit_mq_sendrecv(mqdes, msg_len, msg_prio, ts);
>  
>  	CLASS(fd, f)(mqdes);
> @@ -1139,23 +1141,31 @@ static int do_mq_timedsend(mqd_t mqdes, const char __user *u_msg_ptr,
>  	return ret;
>  }
>  
> -static int do_mq_timedreceive(mqd_t mqdes, char __user *u_msg_ptr,
> -		size_t msg_len, unsigned int __user *u_msg_prio,
> -		struct timespec64 *ts)
> +static int do_mq_timedsend(mqd_t mqdes, const char __user *u_msg_ptr,
> +						size_t msg_len, unsigned int msg_prio,
> +						struct timespec64 *ts)
>  {
> -	ssize_t ret;
> -	struct msg_msg *msg_ptr;
> -	struct inode *inode;
> -	struct mqueue_inode_info *info;
> -	struct ext_wait_queue wait;
>  	ktime_t expires, *timeout = NULL;
> -	struct posix_msg_tree_node *new_leaf = NULL;
>  
>  	if (ts) {
>  		expires = timespec64_to_ktime(*ts);
>  		timeout = &expires;
>  	}
>  
> +	return do_mq_sendmsg(mqdes, u_msg_ptr, msg_len, msg_prio, ts, timeout);

It feels that timeout argument is not really required here, we can always
derive it when we need it from ts.

> +}
> +
> +static int do_mq_recvmsg(mqd_t mqdes, char __user *u_msg_ptr, size_t msg_len,
> +				unsigned int __user *u_msg_prio, const struct timespec64 *ts,
> +				ktime_t *timeout)
> +{
> +	ssize_t ret;
> +	struct msg_msg *msg_ptr;
> +	struct inode *inode;
> +	struct mqueue_inode_info *info;
> +	struct ext_wait_queue wait;
> +	struct posix_msg_tree_node *new_leaf = NULL;
> +
>  	audit_mq_sendrecv(mqdes, msg_len, 0, ts);
>  
>  	CLASS(fd, f)(mqdes);
> @@ -1230,6 +1240,350 @@ static int do_mq_timedreceive(mqd_t mqdes, char __user *u_msg_ptr,
>  	return ret;
>  }
>  
> +static int do_mq_timedreceive(mqd_t mqdes, char __user *u_msg_ptr, size_t msg_len,
> +						unsigned int __user *u_msg_prio,
> +						struct timespec64 *ts)
> +{
> +	ktime_t expires, *timeout = NULL;
> +
> +	if (ts) {
> +		expires = timespec64_to_ktime(*ts);
> +		timeout = &expires;
> +	}
> +
> +	return do_mq_recvmsg(mqdes, u_msg_ptr, msg_len,
> +						u_msg_prio, ts, timeout);
> +}
> +
> +static struct msg_msg *mq_peek_index(struct mqueue_inode_info *info, unsigned long index)
> +{
> +	struct rb_node *node;
> +	struct posix_msg_tree_node *leaf;
> +	struct msg_msg *msg;
> +
> +	int count = 0;
> +
> +	/* Start from highest priority */
> +	node = rb_last(&info->msg_tree);
> +	while (node) {
> +		leaf = rb_entry(node, struct posix_msg_tree_node, rb_node);
> +		list_for_each_entry(msg, &leaf->msg_list, m_list) {
> +			if (count == index)
> +				return msg;
> +			count++;
> +		}
> +
> +		node = rb_prev(node);
> +	}
> +
> +	return NULL;
> +}
> +
> +static int do_mq_recvmsg2(mqd_t mqdes, struct mq_msg_attrs *args, unsigned int flags,
> +						unsigned long index, const struct timespec64 *ts,
> +						ktime_t *timeout)
> +{
> +	ssize_t ret;
> +	struct msg_msg *msg_ptr, *k_msg_buffer;
> +	long k_m_type;
> +	size_t k_m_ts;
> +	struct inode *inode;
> +	struct mqueue_inode_info *info;
> +
> +	if (flags & MQ_PEEK) {
> +		audit_mq_sendrecv(mqdes, args->msg_len, 0, ts);
> +		CLASS(fd, f)(mqdes);
> +		if (fd_empty(f))
> +			return -EBADF;
> +
> +		inode = file_inode(fd_file(f));
> +		if (unlikely(fd_file(f)->f_op != &mqueue_file_operations))
> +			return -EBADF;
> +		info = MQUEUE_I(inode);
> +		audit_file(fd_file(f));
> +
> +		if (unlikely(!(fd_file(f)->f_mode & FMODE_READ)))
> +			return -EBADF;
> +
> +		if (unlikely(args->msg_len < info->attr.mq_msgsize))
> +			return -EMSGSIZE;
> +		if (index >= (unsigned long)info->attr.mq_maxmsg)
> +			return -EINVAL;
> +
> +		spin_lock(&info->lock);
> +		if (info->attr.mq_curmsgs == 0) {
> +			spin_unlock(&info->lock);
> +			return -EAGAIN;
> +		}
> +		msg_ptr = mq_peek_index(info, index);
> +		if (!msg_ptr) {
> +			spin_unlock(&info->lock);
> +			return -ENODATA;
> +		}
> +		k_m_type = msg_ptr->m_type;
> +		k_m_ts = msg_ptr->m_ts;
> +		spin_unlock(&info->lock);
> +
> +		k_msg_buffer = alloc_msg(k_m_ts);
> +		if (!k_msg_buffer)
> +			return -ENOMEM;
> +		ret = security_msg_msg_alloc(k_msg_buffer);
> +		if (ret)
> +			return ret;
> +
> +	/*
> +	 * Two spin locks are necessary here. We are avoiding atomic memory
> +	 * allocation and premature allocation before confirming
> +	 * a message actually exists to peek and retrieving required buffer size
> +	 * when first lock was taken.
> +	 */

nit: bad alignment

> +		spin_lock(&info->lock);
> +		msg_ptr = mq_peek_index(info, index);
> +		if (!msg_ptr || msg_ptr->m_type != k_m_type ||
> +			msg_ptr->m_ts != k_m_ts) {
> +			spin_unlock(&info->lock);
> +			free_msg(k_msg_buffer);
> +			return -EAGAIN;
> +		}
> +		msg_ptr = copy_msg(msg_ptr, k_msg_buffer, k_m_ts);
> +		if (IS_ERR(msg_ptr)) {
> +			spin_unlock(&info->lock);
> +			free_msg(k_msg_buffer);
> +			return PTR_ERR(msg_ptr);
> +		}
> +		spin_unlock(&info->lock);
> +
> +		ret = k_msg_buffer->m_ts;
> +		if (args->msg_prio && put_user(k_m_type, args->msg_prio)) {
> +			free_msg(k_msg_buffer);
> +			return -EFAULT;
> +		}
> +		if (store_msg((char *)args->msg_ptr, k_msg_buffer, k_m_ts)) {
> +			free_msg(k_msg_buffer);
> +			return -EFAULT;
> +		}
> +		free_msg(k_msg_buffer);
> +			return ret;
> +	}
> +	if (flags & MQ_RECV) {
> +		return do_mq_recvmsg(mqdes, (char *)args->msg_ptr, args->msg_len,
> +			args->msg_prio, ts, timeout);
> +	}
> +
> +	return -EINVAL;
> +}
> +
> +static int mq_mmsg_copy_attrs_from_user(struct mq_mmsg_attrs *attrs,
> +			      const struct mq_mmsg_attrs __user *uattrs,
> +			      unsigned int attrs_len)
> +{
> +	if (unlikely(attrs_len < sizeof(*attrs)))
> +		return -EINVAL;
> +	if (unlikely(attrs_len > PAGE_SIZE))
> +		return -E2BIG;
> +	return copy_struct_from_user(attrs, sizeof(*attrs), uattrs, attrs_len);
> +}
> +
> +#ifdef CONFIG_COMPAT
> +
> +static int mq_mmsg_copy_compat_attrs(struct mq_mmsg_attrs *attrs,
> +				     const struct compat_mq_mmsg_attrs __user *uattrs,
> +				     unsigned int attrs_len)
> +{
> +	struct compat_mq_mmsg_attrs v = {};
> +	int err;
> +
> +	if (unlikely(attrs_len < sizeof(v)))
> +		return -EINVAL;
> +	if (unlikely(attrs_len > PAGE_SIZE))
> +		return -E2BIG;
> +	err = copy_struct_from_user(&v, sizeof(v), uattrs, attrs_len);
> +	if (err)
> +		return err;
> +
> +	memset(attrs, 0, sizeof(*attrs));

Do we need this memset? In all callers kattrs is explicitly zero
initialized (with = {}), right?

> +	attrs->msg_attrs_vec->iov_base = compat_ptr(v.msg_attrs_vec->iov_base);

The value of attrs->msg_attrs_vec is NULL, here kernel will crash with
nullpointer dereference.

Also v.msg_attrs_vec seems like a user pointer, you can't just dereference
it using _user primitives from kernel space.

Did you try to run it in compat?

> +	attrs->msg_attrs_vec->iov_len = v.msg_attrs_vec->iov_len;
> +	attrs->ret = (int *)compat_ptr(v.ret);
> +	attrs->vlen = v.vlen;
> +	return 0;
> +}
> +
> +static int mq_mmsg_copy_compat_msg_attr(struct mq_msg_attrs *attr,
> +				 const struct iovec *desc_iov)
> +{
> +		struct compat_msg_attrs v = {};
> +
> +		if (desc_iov->iov_len != sizeof(v))
> +			return -EINVAL;
> +		if (copy_from_user(&v, desc_iov->iov_base, sizeof(v)))
> +			return -EFAULT;
> +
> +		memset(attr, 0, sizeof(*attr));
> +		attr->msg_len = v.msg_len;
> +		attr->msg_prio = (unsigned int *)compat_ptr(v.msg_prio);
> +		attr->msg_ptr = compat_ptr(v.msg_ptr);
> +		return 0;
> +	}
> +
> +#endif
> +
> +static int mq_mmsg_copy_msg_attr(struct mq_msg_attrs *attr,
> +					const struct iovec *desc_iov, bool compat)
> +{
> +	if (compat) {
> +		return mq_mmsg_copy_compat_msg_attr(attr, desc_iov);
> +	}
> +
> +	if (desc_iov->iov_len != sizeof(*attr))
> +		return -EINVAL;
> +	if (copy_from_user(attr, desc_iov->iov_base, sizeof(*attr)))
> +		return -EFAULT;
> +
> +	return 0;
> +}
> +
> +static inline int mq_mmsg_done_or_error(unsigned int done, int ret)
> +{
> +	if (ret < 0 && done)
> +		return done;
> +	return ret;
> +}
> +
> +static ssize_t do_mq_recvmmsg(mqd_t mqdes, const struct mq_mmsg_attrs *attrs,
> +					unsigned int flags, unsigned long start_idx,
> +					struct timespec64 *ts, bool compat)
> +{
> +	struct iov_iter iter;
> +	struct iovec outer_iovstack[UIO_FASTIOV], *free_iov = outer_iovstack;
> +	const struct iovec *msg_iov;
> +	ktime_t batch_deadline, *timeout = NULL;
> +	ssize_t ret = 0;
> +	ssize_t batch_success = 0;
> +	unsigned long index;
> +	unsigned int i;
> +
> +	if (flags & (~MQ_VALID_FLAGS))
> +		return -EINVAL;
> +	if ((flags & MQ_RECV) && (flags & MQ_PEEK))
> +		return -EINVAL;
> +	if (start_idx > attrs->vlen)
> +		return -EINVAL;
> +	if (!attrs->vlen || attrs->vlen > UIO_MAXIOV)
> +		return -EINVAL;
> +
> +	ret = __import_iovec(ITER_DEST,
> +			     attrs->msg_attrs_vec, attrs->vlen,
> +			     ARRAY_SIZE(outer_iovstack), &free_iov, &iter,
> +			     compat);
> +	if (ret < 0)
> +		return ret;
> +	msg_iov = iter_iov(&iter);
> +
> +	/* One absolute deadline for the whole batch. */
> +	if (ts) {
> +		batch_deadline = timespec64_to_ktime(*ts);
> +		timeout = &batch_deadline;
> +	}
> +
> +	for (i = start_idx; i < attrs->vlen; i++) {
> +		struct mq_msg_attrs desc = {};
> +
> +		ret = mq_mmsg_copy_msg_attr(&desc, &msg_iov[i], compat);
> +		if (ret < 0)
> +			break;
> +
> +		index = (flags & MQ_PEEK) ? i : 0;
> +		ret = do_mq_recvmsg2(mqdes, &desc, flags, index, ts, timeout);

The search insidde do_mq_recvmsg2 -> mq_peek_index does not seem that optimized,
maybe we can preserve last rbtree node pointer somehow to make the search
more effective instead of walking full tree each time.

> +
> +		if (ret < 0) {
> +			if (attrs->ret && put_user(ret, attrs->ret)) {
> +				kfree(free_iov);
> +				return -EFAULT;
> +			}
> +			break;
> +		}
> +		batch_success++;
> +	}
> +
> +	if (!(batch_success != attrs->vlen)) {
> +		if (attrs->ret && put_user(0, attrs->ret)) {
> +			kfree(free_iov);
> +			return -EFAULT;
> +			}
> +	}
> +	kfree(free_iov);
> +	return mq_mmsg_done_or_error(batch_success, ret < 0 ? ret : batch_success);
> +}
> +
> +static ssize_t do_mq_sendmmsg(mqd_t mqdes, const struct mq_mmsg_attrs *attrs,
> +						unsigned long start_idx, struct timespec64 *ts,
> +						bool compat)
> +{
> +	struct iov_iter iter;
> +	struct iovec outer_iovstack[UIO_FASTIOV], *free_iov = outer_iovstack;
> +	const struct iovec *msg_iov;
> +	ktime_t batch_deadline, *timeout = NULL;
> +	ssize_t ret = 0;
> +	ssize_t batch_success = 0;
> +	unsigned int i;
> +
> +	if (!attrs->vlen || attrs->vlen > UIO_MAXIOV || start_idx > attrs->vlen)
> +		return -EINVAL;
> +
> +	ret = __import_iovec(ITER_SOURCE,
> +			     attrs->msg_attrs_vec, attrs->vlen,
> +			     ARRAY_SIZE(outer_iovstack), &free_iov, &iter,
> +			     compat);
> +	if (ret < 0)
> +		return ret;
> +	msg_iov = iter_iov(&iter);
> +
> +	if (ts) {
> +		batch_deadline = timespec64_to_ktime(*ts);
> +		timeout = &batch_deadline;
> +	}
> +
> +	for (i = start_idx; i < attrs->vlen; i++) {
> +		struct mq_msg_attrs desc = {};
> +		unsigned int msg_prio = 0;
> +
> +		ret = mq_mmsg_copy_msg_attr(&desc, &msg_iov[i], compat);
> +		if (ret < 0)
> +			break;
> +
> +		if (!desc.msg_prio) {
> +			ret = -EINVAL;
> +			break;
> +		}
> +		if (get_user(msg_prio, desc.msg_prio)) {
> +			ret = -EFAULT;
> +			break;
> +		}
> +
> +		ret = do_mq_sendmsg(mqdes, desc.msg_ptr, desc.msg_len,
> +						msg_prio, ts, timeout);
> +
> +		if (ret < 0) {
> +			if (attrs->ret && put_user(ret, attrs->ret)) {
> +				kfree(free_iov);
> +				return -EFAULT;
> +			}
> +		break;
> +		}
> +		batch_success++;
> +	}
> +
> +	if (!(batch_success != attrs->vlen)) {
> +		if (attrs->ret && put_user(0, attrs->ret)) {
> +			kfree(free_iov);
> +			return -EFAULT;
> +			}
> +	}
> +	kfree(free_iov);
> +	return mq_mmsg_done_or_error(batch_success, ret < 0 ? ret : batch_success);
> +}
> +
>  SYSCALL_DEFINE5(mq_timedsend, mqd_t, mqdes, const char __user *, u_msg_ptr,
>  		size_t, msg_len, unsigned int, msg_prio,
>  		const struct __kernel_timespec __user *, u_abs_timeout)
> @@ -1244,6 +1598,27 @@ SYSCALL_DEFINE5(mq_timedsend, mqd_t, mqdes, const char __user *, u_msg_ptr,
>  	return do_mq_timedsend(mqdes, u_msg_ptr, msg_len, msg_prio, p);
>  }
>  
> +SYSCALL_DEFINE5(mq_sendmmsg, mqd_t, mqdes, struct mq_mmsg_attrs __user *, attrs,
> +			unsigned int, attrs_len, unsigned long, start_idx,
> +			const struct __kernel_timespec __user *, u_abs_timeout)
> +{
> +	struct mq_mmsg_attrs kattrs = {};
> +	struct timespec64 ts, *p = NULL;
> +	int ret;
> +
> +	ret = mq_mmsg_copy_attrs_from_user(&kattrs, attrs, attrs_len);
> +	if (ret)
> +		return ret;
> +	if (u_abs_timeout) {
> +		int res = prepare_timeout(u_abs_timeout, &ts);
> +
> +		if (res)
> +			return res;
> +		p = &ts;
> +	}
> +	return do_mq_sendmmsg(mqdes, &kattrs, start_idx, p, false);
> +}
> +
>  SYSCALL_DEFINE5(mq_timedreceive, mqd_t, mqdes, char __user *, u_msg_ptr,
>  		size_t, msg_len, unsigned int __user *, u_msg_prio,
>  		const struct __kernel_timespec __user *, u_abs_timeout)
> @@ -1251,6 +1626,7 @@ SYSCALL_DEFINE5(mq_timedreceive, mqd_t, mqdes, char __user *, u_msg_ptr,
>  	struct timespec64 ts, *p = NULL;
>  	if (u_abs_timeout) {
>  		int res = prepare_timeout(u_abs_timeout, &ts);
> +
>  		if (res)
>  			return res;
>  		p = &ts;
> @@ -1258,6 +1634,26 @@ SYSCALL_DEFINE5(mq_timedreceive, mqd_t, mqdes, char __user *, u_msg_ptr,
>  	return do_mq_timedreceive(mqdes, u_msg_ptr, msg_len, u_msg_prio, p);
>  }
>  
> +SYSCALL_DEFINE6(mq_recvmmsg, mqd_t, mqdes, struct mq_mmsg_attrs __user *, attrs,
> +			unsigned int, attrs_len, unsigned int, flags, unsigned long, start_idx,
> +			const struct __kernel_timespec __user *, u_abs_timeout)
> +{
> +	struct mq_mmsg_attrs kattrs = {};
> +	struct timespec64 ts, *p = NULL;
> +	int ret;
> +
> +	ret = mq_mmsg_copy_attrs_from_user(&kattrs, attrs, attrs_len);
> +	if (ret)
> +		return ret;
> +	if (u_abs_timeout) {
> +		int res = prepare_timeout(u_abs_timeout, &ts);
> +
> +		if (res)
> +			return res;
> +		p = &ts;
> +	}
> +	return do_mq_recvmmsg(mqdes, &kattrs, flags, start_idx, p, false);
> +}
>  /*
>   * Notes: the case when user wants us to deregister (with NULL as pointer)
>   * and he isn't currently owner of notification, will be silently discarded.
> @@ -1451,6 +1847,67 @@ SYSCALL_DEFINE3(mq_getsetattr, mqd_t, mqdes,
>  
>  #ifdef CONFIG_COMPAT
>  
> +COMPAT_SYSCALL_DEFINE5(mq_sendmmsg, mqd_t, mqdes, struct compat_mq_mmsg_attrs __user *, attrs,
> +				unsigned int, attrs_len, unsigned int, flags,
> +				const struct __kernel_timespec __user *, u_abs_timeout)
> +{
> +	struct mq_mmsg_attrs kattrs = {};
> +	struct timespec64 ts, *p = NULL;
> +	int ret;
> +
> +	ret = mq_mmsg_copy_compat_attrs(&kattrs, attrs, attrs_len);
> +	if (ret)
> +		return ret;
> +
> +	if (u_abs_timeout) {
> +		int res = prepare_timeout(u_abs_timeout, &ts);
> +
> +		if (res)
> +			return res;
> +		p = &ts;
> +	}
> +
> +	return do_mq_sendmmsg(mqdes, &kattrs, flags, p, true);
> +}
> +
> +COMPAT_SYSCALL_DEFINE6(mq_recvmmsg, mqd_t, mqdes, struct compat_mq_mmsg_attrs __user *, attrs,
> +				unsigned int, attrs_len, unsigned int, flags, unsigned long, start_idx,
> +				const struct __kernel_timespec __user *, u_abs_timeout)
> +{
> +	struct mq_mmsg_attrs kattrs = {};
> +	struct timespec64 ts, *p = NULL;
> +	int ret;
> +
> +	ret = mq_mmsg_copy_compat_attrs(&kattrs, attrs, attrs_len);
> +	if (ret)
> +		return ret;
> +
> +	if (u_abs_timeout) {
> +		int res = prepare_timeout(u_abs_timeout, &ts);
> +
> +		if (res)
> +			return res;
> +		p = &ts;
> +	}
> +
> +	return do_mq_recvmmsg(mqdes, &kattrs, flags, start_idx, p, true);
> +}
> +
> +#endif
> +
> +#ifdef CONFIG_COMPAT_32BIT_TIME
> +static int compat_prepare_timeout(const struct old_timespec32 __user *p,
> +								struct timespec64 *ts)
> +{
> +	if (get_old_timespec32(ts, p))
> +		return -EFAULT;
> +	if (!timespec64_valid(ts))
> +		return -EINVAL;
> +	return 0;
> +}

Maybe we need #endif here?

> +
> +#ifdef CONFIG_COMPAT
> +
>  struct compat_mq_attr {
>  	compat_long_t mq_flags;      /* message queue flags		     */
>  	compat_long_t mq_maxmsg;     /* maximum number of messages	     */
> @@ -1541,18 +1998,8 @@ COMPAT_SYSCALL_DEFINE3(mq_getsetattr, mqd_t, mqdes,
>  		return -EFAULT;
>  	return 0;
>  }
> -#endif
>  
> -#ifdef CONFIG_COMPAT_32BIT_TIME
> -static int compat_prepare_timeout(const struct old_timespec32 __user *p,
> -				   struct timespec64 *ts)
> -{
> -	if (get_old_timespec32(ts, p))
> -		return -EFAULT;
> -	if (!timespec64_valid(ts))
> -		return -EINVAL;
> -	return 0;
> -}
> +#endif
>  
>  SYSCALL_DEFINE5(mq_timedsend_time32, mqd_t, mqdes,
>  		const char __user *, u_msg_ptr,
> @@ -1583,6 +2030,7 @@ SYSCALL_DEFINE5(mq_timedreceive_time32, mqd_t, mqdes,
>  	}
>  	return do_mq_timedreceive(mqdes, u_msg_ptr, msg_len, u_msg_prio, p);
>  }
> +
>  #endif
>  
>  static const struct inode_operations mqueue_dir_inode_operations = {
> diff --git a/ipc/msg.c b/ipc/msg.c
> index 62996b97f0ac..6392b11dd7f7 100644
> --- a/ipc/msg.c
> +++ b/ipc/msg.c
> @@ -1156,7 +1156,7 @@ static long do_msgrcv(int msqid, void __user *buf, size_t bufsz, long msgtyp, in
>  			 * not update queue parameters.
>  			 */
>  			if (msgflg & MSG_COPY) {
> -				msg = copy_msg(msg, copy);
> +				msg = copy_msg(msg, copy, msg->m_ts);
>  				goto out_unlock0;
>  			}
>  
> diff --git a/ipc/msgutil.c b/ipc/msgutil.c
> index e28f0cecb2ec..25729e96f83c 100644
> --- a/ipc/msgutil.c
> +++ b/ipc/msgutil.c
> @@ -51,7 +51,7 @@ static int __init init_msg_buckets(void)
>  }
>  subsys_initcall(init_msg_buckets);
>  
> -static struct msg_msg *alloc_msg(size_t len)
> +struct msg_msg *alloc_msg(size_t len)
>  {
>  	struct msg_msg *msg;
>  	struct msg_msgseg **pseg;
> @@ -122,39 +122,36 @@ struct msg_msg *load_msg(const void __user *src, size_t len)
>  	free_msg(msg);
>  	return ERR_PTR(err);
>  }
> -#ifdef CONFIG_CHECKPOINT_RESTORE
> -struct msg_msg *copy_msg(struct msg_msg *src, struct msg_msg *dst)
> +
> +struct msg_msg *copy_msg(struct msg_msg *src, struct msg_msg *dst, size_t len)
>  {
> -	struct msg_msgseg *dst_pseg, *src_pseg;
> -	size_t len = src->m_ts;
> -	size_t alen;
> +	struct msg_msgseg *src_seg, *dst_seg;
> +	size_t remaining, chunk;
>  
> -	if (src->m_ts > dst->m_ts)
> +	if (len > src->m_ts)
>  		return ERR_PTR(-EINVAL);
> -
> -	alen = min(len, DATALEN_MSG);
> -	memcpy(dst + 1, src + 1, alen);
> -
> -	for (dst_pseg = dst->next, src_pseg = src->next;
> -	     src_pseg != NULL;
> -	     dst_pseg = dst_pseg->next, src_pseg = src_pseg->next) {
> -
> -		len -= alen;
> -		alen = min(len, DATALEN_SEG);
> -		memcpy(dst_pseg + 1, src_pseg + 1, alen);
> +	chunk = min(len, DATALEN_MSG);
> +	memcpy(dst + 1, src + 1, chunk);
> +	remaining = len - chunk;
> +	src_seg = src->next;
> +	dst_seg = dst->next;
> +	while (remaining > 0 && src_seg && dst_seg) {
> +
> +		chunk = min(remaining, DATALEN_SEG);
> +		memcpy(dst_seg + 1, src_seg + 1, chunk);
> +		remaining -= chunk;
> +		src_seg = src_seg->next;
> +		dst_seg = dst_seg->next;
>  	}
>  
> +	if (remaining != 0)
> +		return ERR_PTR(-EINVAL);
>  	dst->m_type = src->m_type;
> -	dst->m_ts = src->m_ts;
> -
> +	dst->m_ts   = src->m_ts;
>  	return dst;
> -}
> -#else
> -struct msg_msg *copy_msg(struct msg_msg *src, struct msg_msg *dst)
> -{
> -	return ERR_PTR(-ENOSYS);
> -}
> -#endif
> +
> + }

nit: excess newline above and strange } alignment.

> +
>  int store_msg(void __user *dest, struct msg_msg *msg, size_t len)
>  {
>  	size_t alen;
> diff --git a/ipc/util.h b/ipc/util.h
> index a55d6cebe6d3..374abeee79b3 100644
> --- a/ipc/util.h
> +++ b/ipc/util.h
> @@ -197,8 +197,9 @@ int ipc_parse_version(int *cmd);
>  
>  extern void free_msg(struct msg_msg *msg);
>  extern struct msg_msg *load_msg(const void __user *src, size_t len);
> -extern struct msg_msg *copy_msg(struct msg_msg *src, struct msg_msg *dst);
> +extern struct msg_msg *copy_msg(struct msg_msg *src, struct msg_msg *dst, size_t len);
>  extern int store_msg(void __user *dest, struct msg_msg *msg, size_t len);
> +extern struct msg_msg *alloc_msg(size_t len);
>  
>  static inline int ipc_checkid(struct kern_ipc_perm *ipcp, int id)
>  {
> diff --git a/kernel/sys_ni.c b/kernel/sys_ni.c
> index add3032da16f..8219d76c72a0 100644
> --- a/kernel/sys_ni.c
> +++ b/kernel/sys_ni.c
> @@ -392,5 +392,11 @@ COND_SYSCALL(setuid16);
>  COND_SYSCALL(rseq);
>  COND_SYSCALL(rseq_slice_yield);
>  
> +/* ipc */
> +COND_SYSCALL(mq_recvmmsg);
> +COND_SYSCALL_COMPAT(mq_recvmmsg);
> +COND_SYSCALL(mq_sendmmsg);
> +COND_SYSCALL_COMPAT(mq_sendmmsg);
> +
>  COND_SYSCALL(uretprobe);
>  COND_SYSCALL(uprobe);

-- 
Best regards, Pavel Tikhomirov
Senior Software Developer, Virtuozzo.