On Wed, 30 Sep 2026 22:06:15 +0800 Muchun Song <[email protected]> wrote:

> This version is based on mm-new commit a878c908dc92.
> 
> This series is split out from the earlier, larger series "mm: Generalize
> HVO for HugeTLB and device DAX" [1]. While the parent series generalizes
> vmemmap optimization across HugeTLB and device DAX, this subset addresses
> a single, self-contained step: switching device DAX to the section-based
> sparse-vmemmap optimization infrastructure introduced for HugeTLB.
> 
> After the HugeTLB conversion, optimized vmemmap state is described by
> the memory section and the sparse-vmemmap population path can allocate or
> reuse shared tail vmemmap pages based on that metadata. Device DAX still
> uses the older DAX-specific population model, including a separate tail
> vmemmap page reservation and architecture-specific logic to locate or
> populate reusable tail pages.
> 
> This series makes device DAX use the same section-based model. Device DAX
> records the compound page order from pgmap->vmemmap_shift in section
> metadata before vmemmap population, uses the common per-zone shared tail
> vmemmap page, and drops the extra reserved tail page. The powerpc radix
> path is updated to use the same shared tail-page helper, so the generic
> and powerpc DAX paths follow the same reservation model.
> 

Thanks.  I updated mm-unstable to this version.

> 
> v6:
> - Use try_get_page() to prevent the shared device DAX tail-page
>   reference count from overflowing (suggested by Andrew Morton,
>   reported by Sashiko)
> - Move constant declarations to the top of their functions and fold
>   the shared tail-page array lookup into vmemmap_tails() (suggested by
>   David Hildenbrand)
> - Clarify the DAX population flag description and why optimized tail
>   pages must not be poisoned (suggested by David Hildenbrand)
> - Collect Acked-by tags from David Hildenbrand
> - Rebase onto mm/mm-new


Here's how v6 altered mm.git:


 arch/powerpc/mm/book3s64/radix_pgtable.c |    8 +++--
 mm/sparse-vmemmap.c                      |   32 ++++++++++++---------
 2 files changed, 25 insertions(+), 15 deletions(-)

--- a/arch/powerpc/mm/book3s64/radix_pgtable.c~b
+++ a/arch/powerpc/mm/book3s64/radix_pgtable.c
@@ -1042,13 +1042,17 @@ static pte_t * __meminit radix__vmemmap_
                        /*
                         * When a PTE/PMD entry is freed from the init_mm
                         * there's a free_pages() call to this page allocated
-                        * above. Thus this get_page() is paired with the
+                        * above. Thus this try_get_page() is paired with the
                         * put_page_testzero() on the freeing path.
                         * This can only called by certain ZONE_DEVICE path,
                         * and through vmemmap_populate_compound_pages() when
                         * slab is available.
+                        *
+                        * Use try_get_page() to prevent the shared page 
refcount
+                        * from overflowing.
                         */
-                       get_page(reuse);
+                       if (!try_get_page(reuse))
+                               return NULL;
                        p = page_to_virt(reuse);
                        pr_debug("Tail page reuse vmemmap mapping\n");
                }
--- a/mm/sparse-vmemmap.c~b
+++ a/mm/sparse-vmemmap.c
@@ -170,10 +170,14 @@ static void * __meminit vmemmap_alloc_bl
 #ifdef CONFIG_VMEMMAP_OPTIMIZATION
 #define VMEMMAP_OPTIMIZATION_NR_ORDERS (MAX_FOLIO_ORDER - 
VMEMMAP_OPTIMIZATION_MIN_ORDER + 1)
 
-static __ref struct page **vmemmap_tails_alloc(struct zone *zone)
+static __ref struct page **vmemmap_tails(struct zone *zone)
 {
-       struct page **pages;
-       const size_t size = array_size(VMEMMAP_OPTIMIZATION_NR_ORDERS, 
sizeof(*pages));
+       const size_t size = array_size(VMEMMAP_OPTIMIZATION_NR_ORDERS,
+                                      sizeof(*zone->vmemmap_tails));
+       struct page **pages = READ_ONCE(zone->vmemmap_tails);
+
+       if (pages)
+               return pages;
 
        pages = slab_is_available() ? kzalloc_objs(*pages, 
VMEMMAP_OPTIMIZATION_NR_ORDERS) :
                memblock_alloc(size, __alignof__(*pages));
@@ -193,19 +197,19 @@ static __ref struct page **vmemmap_tails
 
 struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone 
*zone)
 {
-       void *addr;
-       struct page *page, **pages;
        const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER;
+       struct page *page, **pages;
+       void *addr;
 
        if (WARN_ON_ONCE(idx >= VMEMMAP_OPTIMIZATION_NR_ORDERS))
                return NULL;
 
-       pages = READ_ONCE(zone->vmemmap_tails) ? : vmemmap_tails_alloc(zone);
+       pages = vmemmap_tails(zone);
        if (!pages)
                return NULL;
 
        page = READ_ONCE(pages[idx]);
-       if (likely(page))
+       if (page)
                return page;
 
        addr = vmemmap_alloc_block(PAGE_SIZE, zone_to_nid(zone));
@@ -218,9 +222,9 @@ struct page __ref *vmemmap_shared_tail_p
                atomic_set(&page->_mapcount, -1);
                set_page_node(page, zone_to_nid(zone));
                set_page_zone(page, zone_idx(zone));
-               prep_compound_tail(page, NULL, order);
                if (zone_is_zone_device(zone))
                        __SetPageReserved(page);
+               prep_compound_tail(page, NULL, order);
        }
 
        page = virt_to_page(addr);
@@ -528,12 +532,12 @@ static int __meminit vmemmap_populate_co
                                                     unsigned long end, int 
node,
                                                     struct dev_pagemap *pgmap)
 {
+       const unsigned long flags = VMEMMAP_POPULATE_DAX;
+       const unsigned int order = pfn_to_section_compound_order(start_pfn);
        unsigned long size, addr;
        pte_t *pte;
-       int rc;
-       unsigned long flags = VMEMMAP_POPULATE_DAX;
        struct page *page;
-       unsigned int order = pfn_to_section_compound_order(start_pfn);
+       int rc;
 
        page = vmemmap_shared_tail_page(order, device_zone(node));
        if (!page)
@@ -891,8 +895,10 @@ int __meminit sparse_add_section(int nid
 
        ms = __nr_to_section(section_nr);
        /*
-        * Poison uninitialized struct pages in order to catch invalid flags
-        * combinations.
+        * Poison uninitialized struct pages to catch invalid flag combinations.
+        *
+        * Tail struct pages in a vmemmap-optimized section are initialized and
+        * shared during vmemmap population, so they must not be overwritten 
here.
         */
        if (!section_vmemmap_optimizable(ms))
                page_init_poison(memmap, sizeof(struct page) * nr_pages);
_


Reply via email to