Re: [PATCH RFC v2 08/13] hw/virtio/vhost-user: send isolation regions to device

Akihiko Odaki <[email protected]>
Newsgroups dev.linux.lists.virtio-fs,org.nongnu.qemu-devel
Message-ID <[email protected]>
On 2026/08/18 14:12, Connor Kite wrote:
> Adds features to fill a vhost_user_set_mem_table message with the
> addresses of isolation memory regions corresponding to bounce buffers
> and vrings.
> 
> Signed-off-by: Connor Kite <[email protected]>
> ---
>   hw/virtio/vhost-user.c | 118 ++++++++++++++++++++++++++++++++++++++++++++++++-
>   1 file changed, 117 insertions(+), 1 deletion(-)
> 
> diff --git a/hw/virtio/vhost-user.c b/hw/virtio/vhost-user.c
> index 7e9233e174..e1e5cba53d 100644
> --- a/hw/virtio/vhost-user.c
> +++ b/hw/virtio/vhost-user.c
> @@ -1140,7 +1140,77 @@ static void cleanup_isolation_regions(struct vhost_dev *dev)
>       }
>   }
>   
> -__attribute__((unused))
> +typedef struct {
> +    VhostUserMsg *msg;
> +    struct vhost_user *u;
> +    int *fds;
> +    size_t fds_size;
> +    uint64_t vring_iova;
> +    size_t vring_size;
> +    bool vring_node_visited;
> +    size_t *idx;
> +} IOVATreeTraversalArgs;
> +
> +static gboolean vhost_user_fill_msg_reg_from_tree(gpointer key,
> +                                                  gpointer value,
> +                                                  gpointer data)
> +{
> +    IOVATreeTraversalArgs *args = data;
> +    struct vhost_user *u;
> +    VhostUserMsg *msg = args->msg;
> +    DMAMap *map = key;
> +    uint64_t offset;
> +
> +    assert(args && key && value);
> +    if (!args->vring_node_visited) {
> +        args->vring_node_visited = true;
> +        return false;
> +    }
> +
> +    u = args->u;
> +    assert(*args->idx < args->fds_size);
> +
> +    args->fds[*args->idx] = args->u->iso_mem_ctx.fd;
> +
> +    /*
> +     * If the number of regions is fixed, it would be wasteful to use one for
> +     * only the vrings. The first vhost_iova_tree element is always
> +     * reserved for the vrings, so we can simply combine the first and second
> +     * elements, which are contiguous in IOVA space, when sending regions to
> +     * the backend.
> +     */
> +    if (*args->idx == 0) {
> +        offset = u->iso_mem_ctx.vring_hva_addr - u->iso_mem_ctx.shared_mem_addr;
> +        msg->payload.memory.regions[*args->idx].userspace_addr =
> +            args->vring_iova;
> +        /*
> +         * The size from the iova tree is inclusive, so 1 is added to it.
> +         * args->vring_size is exclusive, so no addition is required.
> +         */
> +        msg->payload.memory.regions[*args->idx].memory_size =
> +            args->vring_size + map->size + 1;
> +        msg->payload.memory.regions[*args->idx].guest_phys_addr =
> +            args->vring_iova;
> +        msg->payload.memory.regions[*args->idx].mmap_offset = offset;
> +    } else {
> +        /* Use 128 bit operation in unlikely case of negative iso_iova_offset */
> +        offset = int128_get64(int128_add(int128_make64(map->iova),
> +                                        u->iso_mem_ctx.iso_iova_offset)) -
> +                (uint64_t)u->iso_mem_ctx.shared_mem_addr;
> +
> +        msg->payload.memory.regions[*args->idx].userspace_addr = map->iova;
> +        msg->payload.memory.regions[*args->idx].memory_size = map->size + 1;
> +        msg->payload.memory.regions[*args->idx].guest_phys_addr = map->iova;
> +        msg->payload.memory.regions[*args->idx].mmap_offset = offset;
> +    }
> +
> +    assert(offset + msg->payload.memory.regions[*args->idx].memory_size <=
> +        u->iso_mem_ctx.size);
> +    (*args->idx)++;
> +
> +    return false;
> +}
> +
>   static int init_isolation_regions(struct vhost_dev *dev,
>                                     VhostUserMsg *msg,
>                                     int *fds, size_t *fd_num)
> @@ -1159,6 +1229,7 @@ static int init_isolation_regions(struct vhost_dev *dev,
>       DMAMap *map;
>       DMAMap vring_map;
>       int r;
> +    IOVATreeTraversalArgs trav_args;
>   
>       msg->hdr.request = VHOST_USER_SET_MEM_TABLE;
>   
> @@ -1244,6 +1315,27 @@ static int init_isolation_regions(struct vhost_dev *dev,
>           }
>       }
>   
> +    *fd_num = 0;
> +    trav_args.idx = fd_num;
> +    trav_args.fds = fds;
> +    trav_args.msg = msg;
> +    trav_args.u = u;
> +    trav_args.vring_iova = vring_map.iova;
> +    trav_args.vring_size = total_vring_size;
> +    trav_args.fds_size = nregions;
> +    trav_args.vring_node_visited = false;
> +
> +    vhost_iova_tree_foreach(u->iso_mem_ctx.tree,
> +                            vhost_user_fill_msg_reg_from_tree, &trav_args);
> +
> +    msg->payload.memory.nregions = *fd_num;
> +
> +    assert(*fd_num == nregions);
> +
> +    msg->hdr.size = sizeof(msg->payload.memory.nregions);
> +    msg->hdr.size += sizeof(msg->payload.memory.padding);
> +    msg->hdr.size += *fd_num * sizeof(VhostUserMemoryRegion);
> +
>       return 0;
>   }
>   
> @@ -1251,6 +1343,7 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
>                                       struct vhost_memory *mem)
>   {
>       struct vhost_user *u = dev->opaque;
> +    bool memory_isolation = u->user->memory_isolation;
>       int fds[VHOST_MEMORY_BASELINE_NREGIONS];
>       size_t fd_num = 0;
>       bool do_postcopy = u->postcopy_listen && u->postcopy_fd.handler;
> @@ -1262,6 +1355,11 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
>       int ret;
>   
>       if (do_postcopy) {
> +        /* Postcopy is not supported with memory isolation yet */
> +        if (memory_isolation) {
> +            return -1;
> +        }
> +
>           /*
>            * Postcopy has enough differences that it's best done in it's own
>            * version
> @@ -1278,6 +1376,24 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
>           msg.hdr.flags |= VHOST_USER_NEED_REPLY_MASK;
>       }
>   
> +    if (memory_isolation) {
> +        ret = init_isolation_regions(dev, &msg, fds, &fd_num);
> +        if (ret < 0) {
> +            return ret;
> +        }
> +
> +        ret = vhost_user_write(dev, &msg, fds, fd_num);
> +        if (ret < 0) {
> +            return ret;
> +        }
> +
> +        if (reply_supported) {
> +            return process_message_reply(dev, &msg);
> +        }
> +
> +        return 0;
> +    }
> +

This bypasses vhost_user_fill_set_mem_table_msg() below and suppresses 
the error saying: "Failed initializing vhost-user memory map, consider 
using -object memory-backend-file share=on"

However, vhost-user doesn't work even with memory isolation in such a 
case because vhost_user_no_private_memslots() always returns true and 
that makes vhost_section() filter private memory out. This restriction 
is unnecessary with memory isolation and can be removed instead of 
restoring the error.

This also lacks handling of VHOST_USER_PROTOCOL_F_CONFIGURE_MEM_SLOTS, 
and will overflow fds and VhostUserMemory::regions when 
VHOST_USER_GET_MAX_MEM_SLOTS is larger than VHOST_MEMORY_BASELINE_NREGIONS.

Regards,
Akihiko Odaki

>       if (config_mem_slots) {
>           ret = vhost_user_add_remove_regions(dev, &msg, reply_supported, false);
>           if (ret < 0) {
>
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.