[PATCH RFC v2 08/13] hw/virtio/vhost-user: send isolation regions to device

Connor Kite <[email protected]>
Newsgroups org.nongnu.qemu-devel,dev.linux.lists.virtio-fs
Message-ID <[email protected]>
Adds features to fill a vhost_user_set_mem_table message with the
addresses of isolation memory regions corresponding to bounce buffers
and vrings.

Signed-off-by: Connor Kite <[email protected]>
---
 hw/virtio/vhost-user.c | 118 ++++++++++++++++++++++++++++++++++++++++++++++++-
 1 file changed, 117 insertions(+), 1 deletion(-)

diff --git a/hw/virtio/vhost-user.c b/hw/virtio/vhost-user.c
index 7e9233e174..e1e5cba53d 100644
--- a/hw/virtio/vhost-user.c
+++ b/hw/virtio/vhost-user.c
@@ -1140,7 +1140,77 @@ static void cleanup_isolation_regions(struct vhost_dev *dev)
     }
 }
 
-__attribute__((unused))
+typedef struct {
+    VhostUserMsg *msg;
+    struct vhost_user *u;
+    int *fds;
+    size_t fds_size;
+    uint64_t vring_iova;
+    size_t vring_size;
+    bool vring_node_visited;
+    size_t *idx;
+} IOVATreeTraversalArgs;
+
+static gboolean vhost_user_fill_msg_reg_from_tree(gpointer key,
+                                                  gpointer value,
+                                                  gpointer data)
+{
+    IOVATreeTraversalArgs *args = data;
+    struct vhost_user *u;
+    VhostUserMsg *msg = args->msg;
+    DMAMap *map = key;
+    uint64_t offset;
+
+    assert(args && key && value);
+    if (!args->vring_node_visited) {
+        args->vring_node_visited = true;
+        return false;
+    }
+
+    u = args->u;
+    assert(*args->idx < args->fds_size);
+
+    args->fds[*args->idx] = args->u->iso_mem_ctx.fd;
+
+    /*
+     * If the number of regions is fixed, it would be wasteful to use one for
+     * only the vrings. The first vhost_iova_tree element is always
+     * reserved for the vrings, so we can simply combine the first and second
+     * elements, which are contiguous in IOVA space, when sending regions to
+     * the backend.
+     */
+    if (*args->idx == 0) {
+        offset = u->iso_mem_ctx.vring_hva_addr - u->iso_mem_ctx.shared_mem_addr;
+        msg->payload.memory.regions[*args->idx].userspace_addr =
+            args->vring_iova;
+        /*
+         * The size from the iova tree is inclusive, so 1 is added to it.
+         * args->vring_size is exclusive, so no addition is required.
+         */
+        msg->payload.memory.regions[*args->idx].memory_size =
+            args->vring_size + map->size + 1;
+        msg->payload.memory.regions[*args->idx].guest_phys_addr =
+            args->vring_iova;
+        msg->payload.memory.regions[*args->idx].mmap_offset = offset;
+    } else {
+        /* Use 128 bit operation in unlikely case of negative iso_iova_offset */
+        offset = int128_get64(int128_add(int128_make64(map->iova),
+                                        u->iso_mem_ctx.iso_iova_offset)) -
+                (uint64_t)u->iso_mem_ctx.shared_mem_addr;
+
+        msg->payload.memory.regions[*args->idx].userspace_addr = map->iova;
+        msg->payload.memory.regions[*args->idx].memory_size = map->size + 1;
+        msg->payload.memory.regions[*args->idx].guest_phys_addr = map->iova;
+        msg->payload.memory.regions[*args->idx].mmap_offset = offset;
+    }
+
+    assert(offset + msg->payload.memory.regions[*args->idx].memory_size <=
+        u->iso_mem_ctx.size);
+    (*args->idx)++;
+
+    return false;
+}
+
 static int init_isolation_regions(struct vhost_dev *dev,
                                   VhostUserMsg *msg,
                                   int *fds, size_t *fd_num)
@@ -1159,6 +1229,7 @@ static int init_isolation_regions(struct vhost_dev *dev,
     DMAMap *map;
     DMAMap vring_map;
     int r;
+    IOVATreeTraversalArgs trav_args;
 
     msg->hdr.request = VHOST_USER_SET_MEM_TABLE;
 
@@ -1244,6 +1315,27 @@ static int init_isolation_regions(struct vhost_dev *dev,
         }
     }
 
+    *fd_num = 0;
+    trav_args.idx = fd_num;
+    trav_args.fds = fds;
+    trav_args.msg = msg;
+    trav_args.u = u;
+    trav_args.vring_iova = vring_map.iova;
+    trav_args.vring_size = total_vring_size;
+    trav_args.fds_size = nregions;
+    trav_args.vring_node_visited = false;
+
+    vhost_iova_tree_foreach(u->iso_mem_ctx.tree,
+                            vhost_user_fill_msg_reg_from_tree, &trav_args);
+
+    msg->payload.memory.nregions = *fd_num;
+
+    assert(*fd_num == nregions);
+
+    msg->hdr.size = sizeof(msg->payload.memory.nregions);
+    msg->hdr.size += sizeof(msg->payload.memory.padding);
+    msg->hdr.size += *fd_num * sizeof(VhostUserMemoryRegion);
+
     return 0;
 }
 
@@ -1251,6 +1343,7 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
                                     struct vhost_memory *mem)
 {
     struct vhost_user *u = dev->opaque;
+    bool memory_isolation = u->user->memory_isolation;
     int fds[VHOST_MEMORY_BASELINE_NREGIONS];
     size_t fd_num = 0;
     bool do_postcopy = u->postcopy_listen && u->postcopy_fd.handler;
@@ -1262,6 +1355,11 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
     int ret;
 
     if (do_postcopy) {
+        /* Postcopy is not supported with memory isolation yet */
+        if (memory_isolation) {
+            return -1;
+        }
+
         /*
          * Postcopy has enough differences that it's best done in it's own
          * version
@@ -1278,6 +1376,24 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
         msg.hdr.flags |= VHOST_USER_NEED_REPLY_MASK;
     }
 
+    if (memory_isolation) {
+        ret = init_isolation_regions(dev, &msg, fds, &fd_num);
+        if (ret < 0) {
+            return ret;
+        }
+
+        ret = vhost_user_write(dev, &msg, fds, fd_num);
+        if (ret < 0) {
+            return ret;
+        }
+
+        if (reply_supported) {
+            return process_message_reply(dev, &msg);
+        }
+
+        return 0;
+    }
+
     if (config_mem_slots) {
         ret = vhost_user_add_remove_regions(dev, &msg, reply_supported, false);
         if (ret < 0) {

-- 
2.43.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.