Re: [PATCH RFC v2 08/13] hw/virtio/vhost-user: send isolation regions to device
Akihiko Odaki <[email protected]>
| Newsgroups | dev.linux.lists.virtio-fs,org.nongnu.qemu-devel |
|---|---|
| Message-ID | <[email protected]> |
On 2026/08/18 14:12, Connor Kite wrote: > Adds features to fill a vhost_user_set_mem_table message with the > addresses of isolation memory regions corresponding to bounce buffers > and vrings. > > Signed-off-by: Connor Kite <[email protected]> > --- > hw/virtio/vhost-user.c | 118 ++++++++++++++++++++++++++++++++++++++++++++++++- > 1 file changed, 117 insertions(+), 1 deletion(-) > > diff --git a/hw/virtio/vhost-user.c b/hw/virtio/vhost-user.c > index 7e9233e174..e1e5cba53d 100644 > --- a/hw/virtio/vhost-user.c > +++ b/hw/virtio/vhost-user.c > @@ -1140,7 +1140,77 @@ static void cleanup_isolation_regions(struct vhost_dev *dev) > } > } > > -__attribute__((unused)) > +typedef struct { > + VhostUserMsg *msg; > + struct vhost_user *u; > + int *fds; > + size_t fds_size; > + uint64_t vring_iova; > + size_t vring_size; > + bool vring_node_visited; > + size_t *idx; > +} IOVATreeTraversalArgs; > + > +static gboolean vhost_user_fill_msg_reg_from_tree(gpointer key, > + gpointer value, > + gpointer data) > +{ > + IOVATreeTraversalArgs *args = data; > + struct vhost_user *u; > + VhostUserMsg *msg = args->msg; > + DMAMap *map = key; > + uint64_t offset; > + > + assert(args && key && value); > + if (!args->vring_node_visited) { > + args->vring_node_visited = true; > + return false; > + } > + > + u = args->u; > + assert(*args->idx < args->fds_size); > + > + args->fds[*args->idx] = args->u->iso_mem_ctx.fd; > + > + /* > + * If the number of regions is fixed, it would be wasteful to use one for > + * only the vrings. The first vhost_iova_tree element is always > + * reserved for the vrings, so we can simply combine the first and second > + * elements, which are contiguous in IOVA space, when sending regions to > + * the backend. > + */ > + if (*args->idx == 0) { > + offset = u->iso_mem_ctx.vring_hva_addr - u->iso_mem_ctx.shared_mem_addr; > + msg->payload.memory.regions[*args->idx].userspace_addr = > + args->vring_iova; > + /* > + * The size from the iova tree is inclusive, so 1 is added to it. > + * args->vring_size is exclusive, so no addition is required. > + */ > + msg->payload.memory.regions[*args->idx].memory_size = > + args->vring_size + map->size + 1; > + msg->payload.memory.regions[*args->idx].guest_phys_addr = > + args->vring_iova; > + msg->payload.memory.regions[*args->idx].mmap_offset = offset; > + } else { > + /* Use 128 bit operation in unlikely case of negative iso_iova_offset */ > + offset = int128_get64(int128_add(int128_make64(map->iova), > + u->iso_mem_ctx.iso_iova_offset)) - > + (uint64_t)u->iso_mem_ctx.shared_mem_addr; > + > + msg->payload.memory.regions[*args->idx].userspace_addr = map->iova; > + msg->payload.memory.regions[*args->idx].memory_size = map->size + 1; > + msg->payload.memory.regions[*args->idx].guest_phys_addr = map->iova; > + msg->payload.memory.regions[*args->idx].mmap_offset = offset; > + } > + > + assert(offset + msg->payload.memory.regions[*args->idx].memory_size <= > + u->iso_mem_ctx.size); > + (*args->idx)++; > + > + return false; > +} > + > static int init_isolation_regions(struct vhost_dev *dev, > VhostUserMsg *msg, > int *fds, size_t *fd_num) > @@ -1159,6 +1229,7 @@ static int init_isolation_regions(struct vhost_dev *dev, > DMAMap *map; > DMAMap vring_map; > int r; > + IOVATreeTraversalArgs trav_args; > > msg->hdr.request = VHOST_USER_SET_MEM_TABLE; > > @@ -1244,6 +1315,27 @@ static int init_isolation_regions(struct vhost_dev *dev, > } > } > > + *fd_num = 0; > + trav_args.idx = fd_num; > + trav_args.fds = fds; > + trav_args.msg = msg; > + trav_args.u = u; > + trav_args.vring_iova = vring_map.iova; > + trav_args.vring_size = total_vring_size; > + trav_args.fds_size = nregions; > + trav_args.vring_node_visited = false; > + > + vhost_iova_tree_foreach(u->iso_mem_ctx.tree, > + vhost_user_fill_msg_reg_from_tree, &trav_args); > + > + msg->payload.memory.nregions = *fd_num; > + > + assert(*fd_num == nregions); > + > + msg->hdr.size = sizeof(msg->payload.memory.nregions); > + msg->hdr.size += sizeof(msg->payload.memory.padding); > + msg->hdr.size += *fd_num * sizeof(VhostUserMemoryRegion); > + > return 0; > } > > @@ -1251,6 +1343,7 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev, > struct vhost_memory *mem) > { > struct vhost_user *u = dev->opaque; > + bool memory_isolation = u->user->memory_isolation; > int fds[VHOST_MEMORY_BASELINE_NREGIONS]; > size_t fd_num = 0; > bool do_postcopy = u->postcopy_listen && u->postcopy_fd.handler; > @@ -1262,6 +1355,11 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev, > int ret; > > if (do_postcopy) { > + /* Postcopy is not supported with memory isolation yet */ > + if (memory_isolation) { > + return -1; > + } > + > /* > * Postcopy has enough differences that it's best done in it's own > * version > @@ -1278,6 +1376,24 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev, > msg.hdr.flags |= VHOST_USER_NEED_REPLY_MASK; > } > > + if (memory_isolation) { > + ret = init_isolation_regions(dev, &msg, fds, &fd_num); > + if (ret < 0) { > + return ret; > + } > + > + ret = vhost_user_write(dev, &msg, fds, fd_num); > + if (ret < 0) { > + return ret; > + } > + > + if (reply_supported) { > + return process_message_reply(dev, &msg); > + } > + > + return 0; > + } > + This bypasses vhost_user_fill_set_mem_table_msg() below and suppresses the error saying: "Failed initializing vhost-user memory map, consider using -object memory-backend-file share=on" However, vhost-user doesn't work even with memory isolation in such a case because vhost_user_no_private_memslots() always returns true and that makes vhost_section() filter private memory out. This restriction is unnecessary with memory isolation and can be removed instead of restoring the error. This also lacks handling of VHOST_USER_PROTOCOL_F_CONFIGURE_MEM_SLOTS, and will overflow fds and VhostUserMemory::regions when VHOST_USER_GET_MAX_MEM_SLOTS is larger than VHOST_MEMORY_BASELINE_NREGIONS. Regards, Akihiko Odaki > if (config_mem_slots) { > ret = vhost_user_add_remove_regions(dev, &msg, reply_supported, false); > if (ret < 0) { >