On 18.08.26 07:12, Connor Kite wrote:
Adds features to fill a vhost_user_set_mem_table message with the
addresses of isolation memory regions corresponding to bounce buffers
and vrings.
Signed-off-by: Connor Kite <[email protected]>
---
hw/virtio/vhost-user.c | 118 ++++++++++++++++++++++++++++++++++++++++++++++++-
1 file changed, 117 insertions(+), 1 deletion(-)
diff --git a/hw/virtio/vhost-user.c b/hw/virtio/vhost-user.c
index 7e9233e174..e1e5cba53d 100644
--- a/hw/virtio/vhost-user.c
+++ b/hw/virtio/vhost-user.c
@@ -1140,7 +1140,77 @@ static void cleanup_isolation_regions(struct vhost_dev
*dev)
}
}
-__attribute__((unused))
+typedef struct {
+ VhostUserMsg *msg;
+ struct vhost_user *u;
+ int *fds;
+ size_t fds_size;
+ uint64_t vring_iova;
+ size_t vring_size;
+ bool vring_node_visited;
+ size_t *idx;
+} IOVATreeTraversalArgs;
+
+static gboolean vhost_user_fill_msg_reg_from_tree(gpointer key,
+ gpointer value,
+ gpointer data)
+{
+ IOVATreeTraversalArgs *args = data;
+ struct vhost_user *u;
+ VhostUserMsg *msg = args->msg;
+ DMAMap *map = key;
+ uint64_t offset;
+
+ assert(args && key && value);
+ if (!args->vring_node_visited) {
+ args->vring_node_visited = true;
+ return false;
+ }
+
+ u = args->u;
+ assert(*args->idx < args->fds_size);
+
+ args->fds[*args->idx] = args->u->iso_mem_ctx.fd;
+
+ /*
+ * If the number of regions is fixed, it would be wasteful to use one for
+ * only the vrings. The first vhost_iova_tree element is always
+ * reserved for the vrings, so we can simply combine the first and second
+ * elements, which are contiguous in IOVA space, when sending regions to
+ * the backend.
+ */
I am not sure it makes sense to optimize here if it makes the code more
complicated and the eventual goal would rather be to have a dedicated
memory area for the device I/O memory rather than a full mirror of guest
memory. Specifically because that full mirror already is quite wasteful,
so… it was my understanding that the full mirror is mostly for testing
anyway.
Hanna
+ if (*args->idx == 0) {
+ offset = u->iso_mem_ctx.vring_hva_addr -
u->iso_mem_ctx.shared_mem_addr;
+ msg->payload.memory.regions[*args->idx].userspace_addr =
+ args->vring_iova;
+ /*
+ * The size from the iova tree is inclusive, so 1 is added to it.
+ * args->vring_size is exclusive, so no addition is required.
+ */
+ msg->payload.memory.regions[*args->idx].memory_size =
+ args->vring_size + map->size + 1;
+ msg->payload.memory.regions[*args->idx].guest_phys_addr =
+ args->vring_iova;
+ msg->payload.memory.regions[*args->idx].mmap_offset = offset;
+ } else {
+ /* Use 128 bit operation in unlikely case of negative iso_iova_offset
*/
+ offset = int128_get64(int128_add(int128_make64(map->iova),
+ u->iso_mem_ctx.iso_iova_offset)) -
+ (uint64_t)u->iso_mem_ctx.shared_mem_addr;
+
+ msg->payload.memory.regions[*args->idx].userspace_addr = map->iova;
+ msg->payload.memory.regions[*args->idx].memory_size = map->size + 1;
+ msg->payload.memory.regions[*args->idx].guest_phys_addr = map->iova;
+ msg->payload.memory.regions[*args->idx].mmap_offset = offset;
+ }
+
+ assert(offset + msg->payload.memory.regions[*args->idx].memory_size <=
+ u->iso_mem_ctx.size);
+ (*args->idx)++;
+
+ return false;
+}
+
static int init_isolation_regions(struct vhost_dev *dev,
VhostUserMsg *msg,
int *fds, size_t *fd_num)
@@ -1159,6 +1229,7 @@ static int init_isolation_regions(struct vhost_dev *dev,
DMAMap *map;
DMAMap vring_map;
int r;
+ IOVATreeTraversalArgs trav_args;
msg->hdr.request = VHOST_USER_SET_MEM_TABLE;
@@ -1244,6 +1315,27 @@ static int init_isolation_regions(struct vhost_dev *dev,
}
}
+ *fd_num = 0;
+ trav_args.idx = fd_num;
+ trav_args.fds = fds;
+ trav_args.msg = msg;
+ trav_args.u = u;
+ trav_args.vring_iova = vring_map.iova;
+ trav_args.vring_size = total_vring_size;
+ trav_args.fds_size = nregions;
+ trav_args.vring_node_visited = false;
+
+ vhost_iova_tree_foreach(u->iso_mem_ctx.tree,
+ vhost_user_fill_msg_reg_from_tree, &trav_args);
+
+ msg->payload.memory.nregions = *fd_num;
+
+ assert(*fd_num == nregions);
+
+ msg->hdr.size = sizeof(msg->payload.memory.nregions);
+ msg->hdr.size += sizeof(msg->payload.memory.padding);
+ msg->hdr.size += *fd_num * sizeof(VhostUserMemoryRegion);
+
return 0;
}
@@ -1251,6 +1343,7 @@ static int vhost_user_set_mem_table(struct vhost_dev *dev,
struct vhost_memory *mem)
{
struct vhost_user *u = dev->opaque;
+ bool memory_isolation = u->user->memory_isolation;
int fds[VHOST_MEMORY_BASELINE_NREGIONS];
size_t fd_num = 0;
bool do_postcopy = u->postcopy_listen && u->postcopy_fd.handler;
@@ -1262,6 +1355,11 @@ static int vhost_user_set_mem_table(struct vhost_dev
*dev,
int ret;
if (do_postcopy) {
+ /* Postcopy is not supported with memory isolation yet */
+ if (memory_isolation) {
+ return -1;
+ }
+
/*
* Postcopy has enough differences that it's best done in it's own
* version
@@ -1278,6 +1376,24 @@ static int vhost_user_set_mem_table(struct vhost_dev
*dev,
msg.hdr.flags |= VHOST_USER_NEED_REPLY_MASK;
}
+ if (memory_isolation) {
+ ret = init_isolation_regions(dev, &msg, fds, &fd_num);
+ if (ret < 0) {
+ return ret;
+ }
+
+ ret = vhost_user_write(dev, &msg, fds, fd_num);
+ if (ret < 0) {
+ return ret;
+ }
+
+ if (reply_supported) {
+ return process_message_reply(dev, &msg);
+ }
+
+ return 0;
+ }
+
if (config_mem_slots) {
ret = vhost_user_add_remove_regions(dev, &msg, reply_supported,
false);
if (ret < 0) {