In fast snapshot load, we would like to serve faults as soon as possible
hence loading pages directly instead of requesting a source

Add postcopy_mapped_ram_load_page() function which serves single page
fault by reading the snapshot file. It uses bitmap_test_and_clear_atomic
on pending_bmap to coordinate between threads so each page is loaded
exactly once. Non-zero pages are read using qemu_get_buffer_at into a
temporary page (for loading page atomically), which is then placed using
postcopy_place_page. Zero pages are placed directly using
postcopy_place_page_zero.

Update postcopy_ram_fault_thread to call postcopy_mapped_ram_load_page
instead of requesting source in case of fast snapshot load. to_src_file
check is bypassed in fast snapshot load case as there is no source.

Call try_mark_postcopy_blocktime_begin on every page fault to support
postcopy-blocktime.

Allocate another channel in postcopy_temp_pages_setup(like the preempt
case), for both the fault thread and eager thread to load pages
independently.

Add function ramblock_file_bitmap_page_is_nonzero() which searches a
range of bits corresponding to a page for a set bit. This is just a bit
check for normal pages but for hugepages it checks the range to see if
any part of page is non zero.

Signed-off-by: Aadeshveer Singh <[email protected]>
---
 migration/postcopy-ram.c | 121 +++++++++++++++++++++++++++++++++------
 migration/ram.c          |  11 +++-
 migration/ram.h          |   2 +
 3 files changed, 116 insertions(+), 18 deletions(-)

diff --git a/migration/postcopy-ram.c b/migration/postcopy-ram.c
index 2e2c9fae10..723070b5cd 100644
--- a/migration/postcopy-ram.c
+++ b/migration/postcopy-ram.c
@@ -949,6 +949,68 @@ int postcopy_wake_shared(struct PostCopyFD *pcfd,
                        pagesize);
 }
 
+/**
+ * postcopy_mapped_ram_load_page() - Load a page to given host address.
+ * @mis: Migration Incoming State.
+ * @rb: RAMBlock from where page is loaded.
+ * @rb_offset: Offset of page in RAMBlock.
+ * @haddr: Base of page where to load in page.
+ * @channel: Used to identify between threads and use corresponding temp.
+ * @errp: Set error in case of failure
+ *
+ * Load a page from RAMBlock at offset at given host address. Used by postcopy
+ * ram fault thread and eager thread in fast snapshot load case.
+ *
+ * Return: True on success.
+ */
+static bool postcopy_mapped_ram_load_page(MigrationIncomingState *mis,
+                                          RAMBlock *rb, ram_addr_t rb_offset,
+                                          uint64_t haddr, int channel,
+                                          Error **errp)
+{
+    void *place_source = mis->postcopy_tmp_pages[channel].tmp_huge_page;
+    size_t page;
+    size_t read;
+
+    page = rb_offset / qemu_ram_pagesize(rb);
+
+    if (bitmap_test_and_clear_atomic(rb->pending_bmap, page, 1)) {
+        if (ramblock_file_bitmap_page_is_nonzero(rb, page)) {
+            /*
+             * This can happen concurrently, but it's thread-safe because
+             * qemu_get_buffer_at() is thread-safe, and the caller will be 
using
+             * different temporary buffers.
+             */
+            read = qemu_get_buffer_at(mis->from_src_file, place_source,
+                                      qemu_ram_pagesize(rb),
+                                      rb->pages_offset + rb_offset, errp);
+
+            if (read != qemu_ram_pagesize(rb)) {
+                error_prepend(errp, "Could not read page %zu from RAM Block 
%s",
+                              page, rb->idstr);
+                return false;
+            }
+
+            if (postcopy_place_page(mis, (void *)haddr, place_source, rb)) {
+                error_setg(errp,
+                           "Failed to place page %zu from RAM Block %s at "
+                           "address %" PRIu64,
+                           page, rb->idstr, haddr);
+                return false;
+            }
+        } else {
+            if (postcopy_place_page_zero(mis, (void *)haddr, rb)) {
+                error_setg(errp,
+                           "Failed to place zero page %zu from RAM Block %s at 
"
+                           "address %" PRIu64,
+                           page, rb->idstr, haddr);
+                return false;
+            }
+        }
+    }
+    return true;
+}
+
 /*
  * NOTE: @tid is only used when postcopy-blocktime feature is enabled, and
  * also optional: when zero is provided, the fault accounting will be ignored.
@@ -1310,6 +1372,7 @@ static void *postcopy_ram_fault_thread(void *opaque)
     int ret;
     size_t index;
     RAMBlock *rb = NULL;
+    Error *local_err = NULL;
 
     trace_postcopy_ram_fault_thread_entry();
     rcu_register_thread();
@@ -1351,11 +1414,13 @@ static void *postcopy_ram_fault_thread(void *opaque)
             break;
         }
 
-        if (!mis->to_src_file) {
+        if (!migrate_mapped_ram() && !mis->to_src_file) {
             /*
-             * Possibly someone tells us that the return path is
-             * broken already using the event. We should hold until
-             * the channel is rebuilt.
+             * Possibly someone tells us that the return path is broken already
+             * using the event. We should hold until the channel is rebuilt.
+             * Fast snapshot load doesn't support pause and recover, because
+             * it's not necessary: we can fail right away when QEMU just booted
+             * with nothing to lose.
              */
             postcopy_pause_fault_thread(mis);
         }
@@ -1418,18 +1483,37 @@ static void *postcopy_ram_fault_thread(void *opaque)
                                                 qemu_ram_get_idstr(rb),
                                                 rb_offset,
                                                 msg.arg.pagefault.feat.ptid);
+
+            if (migrate_mapped_ram()) {
+                /* Load page directly in case of fast snapshot load */
+
+                uintptr_t aligned = (uintptr_t)ROUND_DOWN(
+                    msg.arg.pagefault.address, qemu_ram_pagesize(rb));
+
+                if (try_mark_postcopy_blocktime_begin(
+                        mis, rb, rb_offset, (uintptr_t)aligned,
+                        msg.arg.pagefault.feat.ptid)) {
+                    if (!postcopy_mapped_ram_load_page(
+                            mis, rb, rb_offset, aligned, RAM_CHANNEL_POSTCOPY,
+                            &local_err)) {
+                        error_report_err(local_err);
+                        break;
+                    }
+                }
+            } else {
 retry:
-            /*
-             * Send the request to the source - we want to request one
-             * of our host page sizes (which is >= TPS)
-             */
-            ret = postcopy_request_page(mis, rb, rb_offset,
-                                        msg.arg.pagefault.address,
-                                        msg.arg.pagefault.feat.ptid);
-            if (ret) {
-                /* May be network failure, try to wait for recovery */
-                postcopy_pause_fault_thread(mis);
-                goto retry;
+                /*
+                 * Send the request to the source - we want to request one
+                 * of our host page sizes (which is >= TPS)
+                 */
+                ret = postcopy_request_page(mis, rb, rb_offset,
+                                            msg.arg.pagefault.address,
+                                            msg.arg.pagefault.feat.ptid);
+                if (ret) {
+                    /* May be network failure, try to wait for recovery */
+                    postcopy_pause_fault_thread(mis);
+                    goto retry;
+                }
             }
         }
 
@@ -1501,8 +1585,11 @@ static int 
postcopy_temp_pages_setup(MigrationIncomingState *mis, Error **errp)
     unsigned i, channels;
     void *temp_page;
 
-    if (migrate_postcopy_preempt()) {
-        /* If preemption enabled, need extra channel for urgent requests */
+    if (migrate_postcopy_preempt() || migrate_mapped_ram()) {
+        /*
+         * If preemption enabled or it is fast snapshot load, need extra 
channel
+         * for urgent requests/faults
+         */
         mis->postcopy_channels = RAM_CHANNEL_MAX;
     } else {
         /* Both precopy/postcopy on the same channel */
diff --git a/migration/ram.c b/migration/ram.c
index 330fceaa43..4ab8e0e750 100644
--- a/migration/ram.c
+++ b/migration/ram.c
@@ -269,12 +269,21 @@ static void ramblock_pending_bmap_init(void)
 
     RAMBLOCK_FOREACH_NOT_IGNORED(rb) {
         assert(!rb->pending_bmap);
-        size_t size = rb->max_length >> qemu_target_page_bits();
+        size_t size = rb->max_length / qemu_ram_pagesize(rb);
         rb->pending_bmap = bitmap_new(size);
         bitmap_set(rb->pending_bmap, 0, size);
     }
 }
 
+bool ramblock_file_bitmap_page_is_nonzero(RAMBlock *rb, uint64_t page_idx)
+{
+    int page_bits = qemu_ram_pagesize(rb) / qemu_target_page_size();
+    uint64_t bmap_page_start = page_idx * page_bits;
+    uint64_t bmap_page_end = bmap_page_start + page_bits;
+    return find_next_bit(rb->file_bmap, bmap_page_end, bmap_page_start) !=
+           bmap_page_end;
+}
+
 static void ramblock_recv_map_init(void)
 {
     RAMBlock *rb;
diff --git a/migration/ram.h b/migration/ram.h
index 41697a7599..7e2eac58d3 100644
--- a/migration/ram.h
+++ b/migration/ram.h
@@ -95,6 +95,8 @@ void ram_handle_zero(void *host, uint64_t size);
 void ram_transferred_add(uint64_t bytes);
 void ram_release_page(const char *rbname, uint64_t offset);
 
+bool ramblock_file_bitmap_page_is_nonzero(RAMBlock *rb, uint64_t page_idx);
+
 int ramblock_recv_bitmap_test(RAMBlock *rb, void *host_addr);
 bool ramblock_recv_bitmap_test_byte_offset(RAMBlock *rb, uint64_t byte_offset);
 void ramblock_recv_bitmap_set(RAMBlock *rb, void *host_addr);
-- 
2.55.0


Reply via email to