> From: Konstantin Ananyev [mailto:[email protected]]
> Sent: Friday, 18 September 2026 14.44
> 
> > Replaced the fixed size object array in the per-lcore local cache
> > with a variable size array, allocated at mempool creation.
> > For faster indexing into the per-lcore array of caches, pre-calculate
> > the size (in bytes) of the per-lcore local cache.
> >
> > Using a variable size array makes the RTE_MEMPOOL_CACHE_MAX_SIZE
> build
> > time configuration parameter superfluous,
> > but it was kept for compatibility purposes, as it is often used for
> > the cache_size parameter when creating mempools.
> >
> > Signed-off-by: Morten Brørup <[email protected]>
> > ---
> > Supersedes: patch-169235
> > ("[v2] mempool: no cache size limit")
> > ---
> >  app/test/test_mempool.c                |   2 +-
> >  doc/guides/rel_notes/release_26_11.rst |  12 ++-
> >  lib/mempool/rte_mempool.c              | 118 ++++++++++++++---------
> --
> >  lib/mempool/rte_mempool.h              |  17 ++--
> >  4 files changed, 86 insertions(+), 63 deletions(-)
> >
> > diff --git a/app/test/test_mempool.c b/app/test/test_mempool.c
> > index 6ff8746474..8a21b385d9 100644
> > --- a/app/test/test_mempool.c
> > +++ b/app/test/test_mempool.c
> > @@ -193,7 +193,7 @@ static int
> > test_mempool_creation_with_exceeded_cache_size(void)
> >     mp_cov = rte_mempool_create("test_cache_too_big",
> >             MEMPOOL_SIZE,
> >             MEMPOOL_ELT_SIZE,
> > -           RTE_MEMPOOL_CACHE_MAX_SIZE + 32, 0,
> > +           MEMPOOL_SIZE + 32, 0,
> >             NULL, NULL,
> >             my_obj_init, NULL,
> >             SOCKET_ID_ANY, 0);
> > diff --git a/doc/guides/rel_notes/release_26_11.rst
> > b/doc/guides/rel_notes/release_26_11.rst
> > index 59b8514360..64f6802e58 100644
> > --- a/doc/guides/rel_notes/release_26_11.rst
> > +++ b/doc/guides/rel_notes/release_26_11.rst
> > @@ -97,10 +97,20 @@ API Changes
> >     Also, make sure to start the actual text at the margin.
> >     =======================================================
> >
> > +* mempool: When creating a mempool, the cache size can be freely
> specified
> > +  (although still not exceed the number of elements),
> > +  and is no longer limited by the ``RTE_MEMPOOL_CACHE_MAX_SIZE``
> build time
> > configuration parameter.
> > +  Although ``RTE_MEMPOOL_CACHE_MAX_SIZE`` has lost its original
> meaning,
> > +  it was kept for compatibility purposes,
> > +  as it is often used for the ``cache_size`` parameter when creating
> mempools.
> > +
> > +* mempool: Updated the ``rte_mempool`` structure as follows:
> > +  - Added the ``sizeof_cache_per_lcore`` field, for indexing into
> the per-lcore
> > local cache.
> > +
> >  * mempool: Updated the ``rte_mempool_cache`` structure as follows:
> >    - Removed the deprecated and obsolete ``flushthresh`` field.
> >    - Removed the ``unused`` field.
> > -  - Reduced the size of the ``objs`` array from
> > ``RTE_MEMPOOL_CACHE_MAX_SIZE`` * 2 to
> > ``RTE_MEMPOOL_CACHE_MAX_SIZE``.
> > +  - Changed the ``objs`` array from fixed size to variable size.
> >
> >  ABI Changes
> >  -----------
> > diff --git a/lib/mempool/rte_mempool.c b/lib/mempool/rte_mempool.c
> > index 211763aced..069aabaa70 100644
> > --- a/lib/mempool/rte_mempool.c
> > +++ b/lib/mempool/rte_mempool.c
> > @@ -759,20 +759,21 @@ mempool_cache_init(struct rte_mempool_cache
> > *cache, uint32_t size)
> >  /*
> >   * Create and initialize a cache for objects that are retrieved from
> and
> >   * returned to an underlying mempool. This structure is identical to
> the
> > - * local_cache[lcore_id] pointed to by the mempool structure.
> > + * local_cache entry pointed to by the mempool structure.
> >   */
> >  RTE_EXPORT_SYMBOL(rte_mempool_cache_create)
> >  struct rte_mempool_cache *
> >  rte_mempool_cache_create(uint32_t size, int socket_id)
> >  {
> >     struct rte_mempool_cache *cache;
> > +   size_t sizeof_cache = sizeof(struct rte_mempool_cache) + size *
> > sizeof(void *);
> >
> > -   if (size == 0 || size > RTE_MEMPOOL_CACHE_MAX_SIZE) {
> > +   if (size == 0 || sizeof_cache > UINT32_MAX) {
> >             rte_errno = EINVAL;
> >             return NULL;
> >     }
> >
> > -   cache = rte_zmalloc_socket("MEMPOOL_CACHE", sizeof(*cache),
> > +   cache = rte_zmalloc_socket("MEMPOOL_CACHE", sizeof_cache,
> >                               RTE_CACHE_LINE_SIZE, socket_id);
> >     if (cache == NULL) {
> >             RTE_MEMPOOL_LOG(ERR, "Cannot allocate mempool cache.");
> > @@ -811,10 +812,9 @@ rte_mempool_create_empty(const char *name,
> > unsigned n, unsigned elt_size,
> >     struct rte_mempool *mp = NULL;
> >     struct rte_tailq_entry *te = NULL;
> >     const struct rte_memzone *mz = NULL;
> > -   size_t mempool_size;
> > +   size_t mempool_size, sizeof_cache_per_lcore;
> >     unsigned int mz_flags =
> > RTE_MEMZONE_1GB|RTE_MEMZONE_SIZE_HINT_ONLY;
> >     struct rte_mempool_objsz objsz;
> > -   unsigned lcore_id;
> >     int ret;
> >
> >     /* compilation-time checks */
> > @@ -822,6 +822,8 @@ rte_mempool_create_empty(const char *name,
> > unsigned n, unsigned elt_size,
> >                       RTE_CACHE_LINE_MASK) != 0);
> >     RTE_BUILD_BUG_ON((sizeof(struct rte_mempool_cache) &
> >                       RTE_CACHE_LINE_MASK) != 0);
> > +   RTE_BUILD_BUG_ON(offsetof(struct rte_mempool_cache, objs) !=
> > +                     sizeof(struct rte_mempool_cache));
> >  #ifdef RTE_LIBRTE_MEMPOOL_STATS
> >     RTE_BUILD_BUG_ON((sizeof(struct rte_mempool_debug_stats) &
> >                       RTE_CACHE_LINE_MASK) != 0);
> > @@ -838,7 +840,18 @@ rte_mempool_create_empty(const char *name,
> > unsigned n, unsigned elt_size,
> >     }
> >
> >     /* asked cache too big */
> > -   if (cache_size > RTE_MEMPOOL_CACHE_MAX_SIZE ||
> > +   sizeof_cache_per_lcore = 0;
> > +   if (cache_size != 0) {
> > +           sizeof_cache_per_lcore = sizeof(struct rte_mempool_cache);
> > +           sizeof_cache_per_lcore +=
> > RTE_CACHE_LINE_ROUNDUP(cache_size * sizeof(void *));
> > +           /*
> > +            * Add padding, to guard against false sharing-like effects
> > +            * on systems with a next-N-lines hardware prefetcher, when
> > +            * accessing objects at the end of the cache.
> > +            */
> > +           sizeof_cache_per_lcore += RTE_CACHE_GUARD_LINES *
> > RTE_CACHE_LINE_SIZE;
> > +   }
> > +   if (sizeof_cache_per_lcore > UINT32_MAX ||
> >         cache_size > n) {
> >             rte_errno = EINVAL;
> >             return NULL;
> > @@ -890,8 +903,7 @@ rte_mempool_create_empty(const char *name,
> > unsigned n, unsigned elt_size,
> >
> >     mempool_size = sizeof(struct rte_mempool);
> >     mempool_size += private_data_size;
> > -   if (cache_size != 0)
> > -           mempool_size += RTE_MAX_LCORE * sizeof(struct
> > rte_mempool_cache);
> > +   mempool_size += RTE_MAX_LCORE * sizeof_cache_per_lcore;
> >
> >     ret = snprintf(mz_name, sizeof(mz_name),
> > RTE_MEMPOOL_MZ_FORMAT, name);
> >     if (ret < 0 || ret >= (int)sizeof(mz_name)) {
> > @@ -918,7 +930,6 @@ rte_mempool_create_empty(const char *name,
> > unsigned n, unsigned elt_size,
> >     mp->elt_size = objsz.elt_size;
> >     mp->header_size = objsz.header_size;
> >     mp->trailer_size = objsz.trailer_size;
> > -   /* Size of default caches, zero means disabled. */
> >     mp->cache_size = cache_size;
> >     mp->private_data_size = private_data_size;
> >     STAILQ_INIT(&mp->elt_list);
> > @@ -942,18 +953,17 @@ rte_mempool_create_empty(const char *name,
> > unsigned n, unsigned elt_size,
> >             goto exit_unlock;
> >     }
> >
> > -   /*
> > -    * local_cache pointer is set even if cache_size is zero.
> > -    * The local_cache points to just past the private data.
> > -    */
> > -   mp->local_cache = (struct rte_mempool_cache *)
> > -           RTE_PTR_ADD(mp, sizeof(struct rte_mempool) +
> > private_data_size);
> > -
> > -   /* Init all default caches. */
> > +   /* local_cache pointer is only set if per-lcore local cache is
> present */
> >     if (cache_size != 0) {
> > -           for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++)
> > -                   mempool_cache_init(&mp->local_cache[lcore_id],
> > -                                      cache_size);
> > +           mp->local_cache = (struct rte_mempool_cache *)
> > +                   RTE_PTR_ADD(mp, sizeof(struct rte_mempool) +
> > private_data_size);
> > +           mp->sizeof_cache_per_lcore = sizeof_cache_per_lcore;
> > +
> > +           /* Init all default caches. */
> > +           struct rte_mempool_cache *cache = mp->local_cache;
> > +           for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                           lcore_id++, cache = RTE_PTR_ADD(cache,
> > sizeof_cache_per_lcore))
> > +                   mempool_cache_init(cache, cache_size);
> >     }
> >
> >     te->data = mp;
> > @@ -1016,16 +1026,17 @@ RTE_EXPORT_SYMBOL(rte_mempool_avail_count)
> >  unsigned int
> >  rte_mempool_avail_count(const struct rte_mempool *mp)
> >  {
> > -   unsigned count;
> > -   unsigned lcore_id;
> > +   unsigned int count;
> >
> >     count = rte_mempool_ops_get_count(mp);
> >
> > -   if (mp->cache_size == 0)
> > +   if (mp->local_cache == NULL)
> >             return count;
> >
> > -   for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++)
> > -           count += mp->local_cache[lcore_id].len;
> > +   const struct rte_mempool_cache *cache = mp->local_cache;
> > +   for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                   lcore_id++, cache = RTE_PTR_ADD(cache, mp-
> > >sizeof_cache_per_lcore))
> > +           count += cache->len;
> >
> >     /*
> >      * due to race condition (access to len is not locked), the
> > @@ -1053,11 +1064,11 @@ rte_mempool_stats_reset(struct rte_mempool
> > *mp)
> >
> >  #ifdef RTE_LIBRTE_MEMPOOL_STATS
> >     memset(&mp->stats, 0, sizeof(mp->stats));
> > -   if (mp->cache_size != 0) {
> > -           for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > lcore_id++) {
> > -                   memset(&mp->local_cache[lcore_id].stats, 0,
> > -                                   
> > sizeof(mp->local_cache[lcore_id].stats));
> > -           }
> > +   if (mp->local_cache != NULL) {
> > +           struct rte_mempool_cache *cache = mp->local_cache;
> > +           for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                           lcore_id++, cache = RTE_PTR_ADD(cache, mp-
> > >sizeof_cache_per_lcore))
> > +                   memset(&cache->stats, 0, sizeof(cache->stats));
> >     }
> >
> >     RTE_MEMPOOL_LOG(DEBUG, "<%s>@%p: statistics reset", mp->name,
> > mp);
> > @@ -1071,18 +1082,18 @@ rte_mempool_stats_reset(struct rte_mempool
> > *mp)
> >  static unsigned
> >  rte_mempool_dump_cache(FILE *f, const struct rte_mempool *mp)
> >  {
> > -   unsigned lcore_id;
> > -   unsigned count = 0;
> > -   unsigned cache_count;
> > +   unsigned int count = 0;
> >
> >     fprintf(f, "  internal cache infos (hide zero value items):\n");
> >     fprintf(f, "    cache_size=%"PRIu32"\n", mp->cache_size);
> >
> > -   if (mp->cache_size == 0)
> > +   if (mp->local_cache == NULL)
> >             return count;
> >
> > -   for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++) {
> > -           cache_count = mp->local_cache[lcore_id].len;
> > +   const struct rte_mempool_cache *cache = mp->local_cache;
> > +   for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                   lcore_id++, cache = RTE_PTR_ADD(cache, mp-
> > >sizeof_cache_per_lcore)) {
> > +           unsigned int cache_count = cache->len;
> >             if (cache_count == 0)
> >                     continue;
> >             fprintf(f, "    cache_count[%u]=%"PRIu32"\n",
> > @@ -1223,15 +1234,13 @@ mempool_audit_cookies(struct rte_mempool *mp)
> >  static void
> >  mempool_audit_cache(const struct rte_mempool *mp)
> >  {
> > -   unsigned lcore_id;
> > -
> > -   if (mp->cache_size == 0)
> > +   if (mp->local_cache == NULL)
> >             return;
> >
> > -   for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++) {
> > -           const struct rte_mempool_cache *cache;
> > -           cache = &mp->local_cache[lcore_id];
> > -           if (cache->size > RTE_DIM(cache->objs)) {
> > +   const struct rte_mempool_cache *cache = mp->local_cache;
> > +   for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                   lcore_id++, cache = RTE_PTR_ADD(cache, mp-
> > >sizeof_cache_per_lcore)) {
> > +           if (cache->size > mp->size) {
> >                     RTE_MEMPOOL_LOG(CRIT, "badness on cache[%u] size",
> > lcore_id);
> >                     rte_panic("MEMPOOL: invalid cache[%u] size\n",
> > lcore_id);
> >             }
> > @@ -1324,13 +1333,15 @@ rte_mempool_dump(FILE *f, struct rte_mempool
> > *mp)
> >             sum.get_success_blks += mp-
> >stats[lcore_id].get_success_blks;
> >             sum.get_fail_blks += mp->stats[lcore_id].get_fail_blks;
> >     }
> > -   if (mp->cache_size != 0) {
> > +   if (mp->local_cache != NULL) {
> >             /* Add the statistics stored in the mempool caches. */
> > -           for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++) {
> > -                   sum.put_bulk += mp-
> > >local_cache[lcore_id].stats.put_bulk;
> > -                   sum.put_objs += mp-
> > >local_cache[lcore_id].stats.put_objs;
> > -                   sum.get_success_bulk += mp-
> > >local_cache[lcore_id].stats.get_success_bulk;
> > -                   sum.get_success_objs += mp-
> > >local_cache[lcore_id].stats.get_success_objs;
> > +           const struct rte_mempool_cache *cache = mp->local_cache;
> > +           for (lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                   lcore_id++, cache = RTE_PTR_ADD(cache, mp-
> > >sizeof_cache_per_lcore)) {
> > +                   sum.put_bulk += cache->stats.put_bulk;
> > +                   sum.put_objs += cache->stats.put_objs;
> > +                   sum.get_success_bulk += cache-
> >stats.get_success_bulk;
> > +                   sum.get_success_objs += cache-
> >stats.get_success_objs;
> >             }
> >     }
> >     fprintf(f, "  stats:\n");
> > @@ -1627,10 +1638,11 @@ mempool_info_cb(struct rte_mempool *mp, void
> > *arg)
> >                               mp->populated_size);
> >
> >     cache_count = 0;
> > -   if (mp->cache_size > 0) {
> > -           int lcore_id;
> > -           for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++)
> > -                   cache_count += mp->local_cache[lcore_id].len;
> > +   if (mp->local_cache != NULL) {
> > +           const struct rte_mempool_cache *cache = mp->local_cache;
> > +           for (unsigned int lcore_id = 0; lcore_id < RTE_MAX_LCORE;
> > +                           lcore_id++, cache = RTE_PTR_ADD(cache, mp-
> > >sizeof_cache_per_lcore))
> > +                   cache_count += cache->len;
> >     }
> >     rte_tel_data_add_dict_uint(info->d, "total_cache_count",
> cache_count);
> >     common_count = rte_mempool_ops_get_count(mp);
> > diff --git a/lib/mempool/rte_mempool.h b/lib/mempool/rte_mempool.h
> > index 5a81e53a9f..5fc7cdb79c 100644
> > --- a/lib/mempool/rte_mempool.h
> > +++ b/lib/mempool/rte_mempool.h
> > @@ -105,8 +105,7 @@ struct __rte_cache_aligned rte_mempool_cache {
> >     } stats;                        /**< Statistics */
> >  #endif
> >     /** Cache objects */
> > -   alignas(RTE_CACHE_LINE_SIZE) void
> > *objs[RTE_MEMPOOL_CACHE_MAX_SIZE];
> > -   RTE_CACHE_GUARD;
> > +   alignas(RTE_CACHE_LINE_SIZE) void *objs[];
> 
> LGTM to me in general.
> Again, just as a thought:
> Would it be plausible to have a pointer per lcore instead?
> i.e.:
> alignas(RTE_CACHE_LINE_SIZE) void *objs[RTE_MAX_LCORE];
> That way we can avoid multiply op in rte_mempool_default_cache(),
> also it will allow us to allocate lcore cache on demand (only for
> enabled lcores).
> As a downside - sizeof(rte_mempool_cache) will grow by
> (RTE_MAX_LCORE - 1) * sizeof(uintptr_t), but that's probably not a big
> deal.

The multiply op is not expensive.

This patch has:
struct rte_mempool_cache *cache =
        (struct rte_mempool_cache *)
        RTE_PTR_ADD(
        mp->local_cache,
        lcore_id * (size_t)mp->sizeof_cache_per_lcore);

Which compiles to something like:
ptr = LOAD(mp + offsetof(cache_ptr));
off = LOAD(mp + offsetof(sizeof_cache_per_lcore));
off *= lcore_id;
ptr += off;


Yes, having an array of pointers to the caches in the mempool would be a good 
alternative:

struct __rte_cache_aligned rte_mempool {
-       struct rte_mempool_cache *local_cache; /**< Per-lcore local cache */

[...]
        struct rte_mempool_debug_stats stats[RTE_MAX_LCORE + 1];
#endif
+       /** Per-lcore local cache */
+       struct rte_mempool_cache *local_cache[RTE_MAX_LCORE];
};


Lookup for that would become simple:
struct rte_mempool_cache *cache =
        mp->local_cache[lcore_id];

Which compiles to something like:
off = lcore_id * sizeof(void*); // Shift operation
off += offsetof(cache_ptr_array);
ptr = LOAD(mp + off);

Yes, I suppose that would be faster.
I'll take a stab at it.

Caches for all lcores must be allocated at mempool creation. Or we would need 
special handling to allocate/free the relevant per-lcore cache for all 
registered mempools whenever a new lcore is registered for use at runtime, e.g. 
by rte_thread_register()/unregister().

> 
> >  };
> >
> >  /**
> > @@ -265,6 +264,7 @@ struct __rte_cache_aligned rte_mempool {
> >     int32_t ops_index;
> >
> >     struct rte_mempool_cache *local_cache; /**< Per-lcore local cache
> */
> > +   uint32_t sizeof_cache_per_lcore; /**< Multiplier for indexing
> into the
> > local cache. */
> >
> >     uint32_t populated_size;         /**< Number of populated
> objects. */
> >     struct rte_mempool_objhdr_list elt_list; /**< List of objects in
> pool */
> > @@ -1050,8 +1050,7 @@ rte_mempool_free(struct rte_mempool *mp);
> >   * @param cache_size
> >   *   If cache_size is non-zero, the rte_mempool library will try to
> >   *   limit the accesses to the common lockless pool, by maintaining
> a
> > - *   per-lcore object cache. This argument must be lower or equal to
> > - *   RTE_MEMPOOL_CACHE_MAX_SIZE and n.
> > + *   per-lcore object cache. This argument must be lower or equal to
> n.
> >   *   The access to the per-lcore table is of course
> >   *   faster than the multi-producer/consumer pool. The cache can be
> >   *   disabled if the cache_size argument is set to 0; it can be
> useful to
> > @@ -1371,15 +1370,17 @@ rte_mempool_cache_create(uint32_t size, int
> > socket_id)
> >  static __rte_always_inline struct rte_mempool_cache *
> >  rte_mempool_default_cache(struct rte_mempool *mp, unsigned lcore_id)
> >  {
> > -   if (unlikely(mp->cache_size == 0))
> > +   if (unlikely(mp->local_cache == NULL))
> >             return NULL;
> >
> >     if (unlikely(lcore_id == LCORE_ID_ANY))
> >             return NULL;
> >
> > -   rte_mempool_trace_default_cache(mp, lcore_id,
> > -           &mp->local_cache[lcore_id]);
> > -   return &mp->local_cache[lcore_id];
> > +   struct rte_mempool_cache *cache = (struct rte_mempool_cache
> > *)RTE_PTR_ADD(mp->local_cache,
> > +                   lcore_id * (size_t)mp->sizeof_cache_per_lcore);
> > +   rte_mempool_trace_default_cache(mp, lcore_id, cache);
> > +   __rte_assume(cache != NULL);
> > +   return cache;
> >  }
> >
> >  /**
> > --
> > 2.43.0

Reply via email to