From: "Morten Brørup" <mb@smartsharesystems.com>
To: <dev@dpdk.org>
Subject: RE: [RFC PATCH] pile stack and mempool driver (resend)
Date: Sat, 1 Aug 2026 10:10:14 +0200 [thread overview]
Message-ID: <98CBD80474FA8B44BF855DF32C47DC35F659AD@smartserver.smartshare.dk> (raw)
In-Reply-To: <20260801065358.981232-1-mb@smartsharesystems.com>
PLEASE IGNORE.
(Resent due to internal mail problems.)
Venlig hilsen / Kind regards,
-Morten Brørup
> -----Original Message-----
> From: Morten Brørup [mailto:mb@smartsharesystems.com]
> Sent: Saturday, 1 August 2026 08.54
> To: dev@dpdk.org
> Cc: Morten Brørup
> Subject: [RFC PATCH] pile stack and mempool driver (resend)
>
> Early submission of:
> - some mempool optimizations,
> - a new mempool "pile" driver, and
> - its underlying "pile" stack implementation.
>
> For community feedback and CI test.
>
> Needless to say, this must be separated into a series of patches.
> For now, I'm submitting a snapshot of work in progress.
>
> Some performance numbers from mempool_perf_autotest_2cores, all
> with cache=1024 cores=2 n_keep=32768:
>
> start performance test (using ring_mp_mc, with cache)
> n_get_bulk= 64 n_put_bulk= 64 constant_n=0 rate_persec= 753985338
> n_get_bulk=256 n_put_bulk=256 constant_n=0 rate_persec= 755805913
>
> start performance test for lf_stack (with cache)
> n_get_bulk= 64 n_put_bulk= 64 constant_n=0 rate_persec= 29132352
> n_get_bulk=256 n_put_bulk=256 constant_n=0 rate_persec= 29276708
>
> start performance test for pile (with cache)
> n_get_bulk= 64 n_put_bulk= 64 constant_n=0 rate_persec= 560159479
> n_get_bulk=256 n_put_bulk=256 constant_n=0 rate_persec= 557910933
>
> Hat tip to Bruce for bringing attention to the ring not being the
> optimal mempool driver.
>
> Signed-off-by: Morten Brørup <mb@smartsharesystems.com>
> ---
> app/test/test_mempool.c | 3 +-
> app/test/test_stack.c | 70 ++++-
> app/test/test_stack_perf.c | 15 +-
> config/rte_config.h | 5 +-
> doc/guides/prog_guide/stack_lib.rst | 67 ++++-
> drivers/mempool/stack/rte_mempool_stack.c | 40 +++
> drivers/net/bonding/rte_eth_bond_pmd.c | 2 +-
> drivers/net/intel/cpfl/cpfl_rxtx.h | 2 +-
> drivers/net/sxe2/sxe2_txrx_vec_avx512.c | 3 +-
> drivers/net/tap/rte_eth_tap.c | 2 +-
> lib/eal/include/rte_common.h | 12 +
> lib/eal/x86/include/rte_memcpy.h | 29 ++
> lib/mempool/mempool_trace.h | 1 -
> lib/mempool/rte_mempool.c | 51 ++--
> lib/mempool/rte_mempool.h | 74 +++--
> lib/stack/meson.build | 3 +-
> lib/stack/rte_stack.c | 18 +-
> lib/stack/rte_stack.h | 77 +++++-
> lib/stack/rte_stack_lf.h | 6 +-
> lib/stack/rte_stack_lf_c11.h | 2 +-
> lib/stack/rte_stack_lf_generic.h | 2 +-
> lib/stack/rte_stack_lf_stubs.h | 2 +-
> lib/stack/rte_stack_pile.c | 33 +++
> lib/stack/rte_stack_pile.h | 316 ++++++++++++++++++++++
> lib/stack/rte_stack_std.h | 41 ++-
> 25 files changed, 756 insertions(+), 120 deletions(-)
> create mode 100644 lib/stack/rte_stack_pile.c
> create mode 100644 lib/stack/rte_stack_pile.h
>
> diff --git a/app/test/test_mempool.c b/app/test/test_mempool.c
> index e54249ce61..76d45cea2a 100644
> --- a/app/test/test_mempool.c
> +++ b/app/test/test_mempool.c
> @@ -112,8 +112,7 @@ test_mempool_basic(struct rte_mempool *mp, int
> use_external_cache)
> GOTO_ERR(ret, out);
>
> printf("get private data\n");
> - if (rte_mempool_get_priv(mp) != (char *)mp +
> - RTE_MEMPOOL_HEADER_SIZE(mp, mp->cache_size))
> + if (rte_mempool_get_priv(mp) != (char *)mp + sizeof(struct
> rte_mempool))
> GOTO_ERR(ret, out);
>
> #ifndef RTE_EXEC_ENV_FREEBSD /* rte_mem_virt2iova() not supported on
> bsd */
> diff --git a/app/test/test_stack.c b/app/test/test_stack.c
> index 5517982774..ac52f1c048 100644
> --- a/app/test/test_stack.c
> +++ b/app/test/test_stack.c
> @@ -81,13 +81,29 @@ test_stack_push_pop(struct rte_stack *s, void
> **obj_table, unsigned int bulk_sz)
> }
> }
>
> - for (i = 0; i < STACK_SIZE; i++) {
> - if (obj_table[i] != popped_objs[STACK_SIZE - i - 1]) {
> - printf("[%s():%u] Incorrect value %p at index
> 0x%x\n",
> - __func__, __LINE__,
> - popped_objs[STACK_SIZE - i - 1], i);
> - rte_free(popped_objs);
> - return -1;
> + if (!(s->flags & RTE_STACK_F_PILE)) {
> + for (i = 0; i < STACK_SIZE; i++) {
> + if (obj_table[i] != popped_objs[STACK_SIZE - i - 1])
> {
> + printf("[%s():%u] Incorrect value %p at index
> 0x%x\n",
> + __func__, __LINE__,
> + popped_objs[STACK_SIZE - i - 1], i);
> + rte_free(popped_objs);
> + return -1;
> + }
> + }
> + }
> +
> + if ((s->flags & RTE_STACK_F_PILE) && (bulk_sz &
> (RTE_STACK_PILE_BULK_SIZE - 1)) == 0) {
> + for (i = 0; i < STACK_SIZE; i += RTE_STACK_PILE_BULK_SIZE)
> {
> + if (memcmp(&obj_table[i],
> + &popped_objs[STACK_SIZE -
> RTE_STACK_PILE_BULK_SIZE - i],
> + RTE_STACK_PILE_BULK_SIZE) != 0) {
> + printf("[%s():%u] Incorrect values %p at index
> 0x%x with bulk size %u\n",
> + __func__, __LINE__,
> + popped_objs[STACK_SIZE -
> RTE_STACK_PILE_BULK_SIZE - i], i, bulk_sz);
> + rte_free(popped_objs);
> + return -1;
> + }
> }
> }
>
> @@ -152,12 +168,26 @@ test_stack_basic(uint32_t flags)
> goto fail_test;
> }
>
> - ret = rte_stack_push(s, obj_table, 2 * STACK_SIZE);
> - if (ret != 0) {
> - printf("[%s():%u] Excess objects push succeeded\n",
> - __func__, __LINE__);
> - goto fail_test;
> +__rte_diagnostic_push
> +#pragma GCC diagnostic ignored "-Warray-bounds"
> +#pragma GCC diagnostic ignored "-Wstringop-overread"
> + if (!(s->flags & RTE_STACK_F_PILE)) {
> + ret = rte_stack_push(s, obj_table, 2 * STACK_SIZE);
> + if (ret != 0) {
> + printf("[%s():%u] Excess objects push succeeded\n",
> + __func__, __LINE__);
> + goto fail_test;
> + }
> }
> + if (s->flags & RTE_STACK_F_PILE) {
> + ret = rte_stack_push(s, obj_table, STACK_SIZE *
> RTE_STACK_PILE_BULK_SIZE + 1);
> + if (ret != 0) {
> + printf("[%s():%u] Excess objects push succeeded\n",
> + __func__, __LINE__);
> + goto fail_test;
> + }
> + }
> +__rte_diagnostic_pop
>
> ret = rte_stack_pop(s, obj_table, 1);
> if (ret != 0) {
> @@ -181,14 +211,14 @@ test_stack_name_reuse(uint32_t flags)
> {
> struct rte_stack *s[2];
>
> - s[0] = rte_stack_create("test", STACK_SIZE, rte_socket_id(),
> flags);
> + s[0] = rte_stack_create(__func__, STACK_SIZE, rte_socket_id(),
> flags);
> if (s[0] == NULL) {
> printf("[%s():%u] Failed to create a stack\n",
> __func__, __LINE__);
> return -1;
> }
>
> - s[1] = rte_stack_create("test", STACK_SIZE, rte_socket_id(),
> flags);
> + s[1] = rte_stack_create(__func__, STACK_SIZE, rte_socket_id(),
> flags);
> if (s[1] != NULL) {
> printf("[%s():%u] Failed to detect re-used name\n",
> __func__, __LINE__);
> @@ -300,6 +330,7 @@ stack_thread_push_pop(__rte_unused void *args)
> __func__, __LINE__, num);
> return -1;
> }
> + rte_compiler_barrier();
> }
>
> return 0;
> @@ -384,5 +415,16 @@ test_lf_stack(void)
> #endif
> }
>
> +static int
> +test_pile(void)
> +{
> +#if defined(RTE_STACK_PILE_SUPPORTED)
> + return __test_stack(RTE_STACK_F_PILE);
> +#else
> + return TEST_SKIPPED;
> +#endif
> +}
> +
> REGISTER_FAST_TEST(stack_autotest, NOHUGE_SKIP, ASAN_OK, test_stack);
> REGISTER_FAST_TEST(stack_lf_autotest, NOHUGE_SKIP, ASAN_OK,
> test_lf_stack);
> +REGISTER_FAST_TEST(stack_pile_autotest, NOHUGE_SKIP, ASAN_OK,
> test_pile);
> diff --git a/app/test/test_stack_perf.c b/app/test/test_stack_perf.c
> index 3f17a2606c..586410671f 100644
> --- a/app/test/test_stack_perf.c
> +++ b/app/test/test_stack_perf.c
> @@ -14,14 +14,14 @@
> #include "test.h"
>
> #define STACK_NAME "STACK_PERF"
> -#define MAX_BURST 32
> +#define MAX_BURST RTE_MEMPOOL_CACHE_MAX_SIZE / 2
> #define STACK_SIZE (RTE_MAX_LCORE * MAX_BURST)
>
> /*
> * Push/pop bulk sizes, marked volatile so they aren't treated as
> compile-time
> * constants.
> */
> -static volatile unsigned int bulk_sizes[] = {8, MAX_BURST};
> +static volatile unsigned int bulk_sizes[] = {1, 8, 32, MAX_BURST};
>
> static RTE_ATOMIC(uint32_t) lcore_barrier;
>
> @@ -354,5 +354,16 @@ test_lf_stack_perf(void)
> #endif
> }
>
> +static int
> +test_pile_perf(void)
> +{
> +#if defined(RTE_STACK_PILE_SUPPORTED)
> + return __test_stack_perf(RTE_STACK_F_PILE);
> +#else
> + return TEST_SKIPPED;
> +#endif
> +}
> +
> REGISTER_PERF_TEST(stack_perf_autotest, test_stack_perf);
> REGISTER_PERF_TEST(stack_lf_perf_autotest, test_lf_stack_perf);
> +REGISTER_PERF_TEST(stack_pile_perf_autotest, test_pile_perf);
> diff --git a/config/rte_config.h b/config/rte_config.h
> index 0447cdf2ad..03350660e4 100644
> --- a/config/rte_config.h
> +++ b/config/rte_config.h
> @@ -56,7 +56,7 @@
> #define RTE_CONTIGMEM_DEFAULT_BUF_SIZE (512*1024*1024)
>
> /* mempool defines */
> -#define RTE_MEMPOOL_CACHE_MAX_SIZE 512
> +#define RTE_MEMPOOL_CACHE_MAX_SIZE 1024
> /* RTE_LIBRTE_MEMPOOL_STATS is not set */
> /* RTE_LIBRTE_MEMPOOL_DEBUG is not set */
>
> @@ -64,6 +64,9 @@
> #define RTE_MBUF_DEFAULT_MEMPOOL_OPS "ring_mp_mc"
> /* RTE_MBUF_HISTORY_DEBUG is not set */
>
> +/* stack defines */
> +#define RTE_STACK_PILE_BULK_SIZE 32
> +
> /* ether defines */
> #define RTE_MAX_QUEUES_PER_PORT 1024
> #define RTE_ETHDEV_RXTX_CALLBACKS 1
> diff --git a/doc/guides/prog_guide/stack_lib.rst
> b/doc/guides/prog_guide/stack_lib.rst
> index fdf056730c..9b473030d3 100644
> --- a/doc/guides/prog_guide/stack_lib.rst
> +++ b/doc/guides/prog_guide/stack_lib.rst
> @@ -1,5 +1,6 @@
> .. SPDX-License-Identifier: BSD-3-Clause
> Copyright(c) 2019 Intel Corporation.
> + Copyright(c) 2026 SmartShare Systems.
>
> Stack Library
> =============
> @@ -9,9 +10,10 @@ stack of pointers.
>
> The stack library provides the following basic operations:
>
> -* Create a uniquely named stack of a user-specified size and using a
> +* Create a uniquely named stack (or pile) of a user-specified size
> and using a
> user-specified socket, with either standard (lock-based) or lock-
> free
> behavior.
> + The pile resembles a lock-free stack, but is not strictly LIFO.
>
> * Push and pop a burst of one or more stack objects (pointers).
> These functions are multi-thread safe.
> @@ -25,8 +27,9 @@ The stack library provides the following basic
> operations:
> Implementation
> --------------
>
> -The library supports two types of stacks: standard (lock-based) and
> lock-free.
> -Both types use the same set of interfaces, but their implementations
> differ.
> +The library supports three types of stacks: standard (lock-based),
> lock-free,
> +and pile (lock-free, not strictly LIFO, optimized for bulk
> operations).
> +All types use the same set of interfaces, but their implementations
> differ.
>
> .. _Stack_Library_Std_Stack:
>
> @@ -64,7 +67,7 @@ The linked list elements themselves are maintained in
> a lock-free LIFO, and are
> allocated before stack pushes and freed after stack pops. Since the
> stack has a
> fixed maximum depth, these elements do not need to be dynamically
> created.
>
> -The lock-free behavior is selected by passing the *RTE_STACK_F_LF*
> flag to
> +The lock-free behavior is selected by passing the ``RTE_STACK_F_LF``
> flag to
> ``rte_stack_create()``.
>
> Preventing the ABA problem
> @@ -86,3 +89,59 @@ both pop stale data and incorrectly change the head
> pointer. By adding a
> modification counter that is updated on every push and pop as part of
> the
> compare-and-swap, the algorithm can detect when the list changes even
> if the
> head pointer remains the same.
> +
> +.. _Stack_Library_Pile:
> +
> +Pile
> +~~~~
> +
> +The pile is a stack-like implementation, optimized for bulk
> operations.
> +It is only LIFO on bulk level, not on object level; i.e. arrays of
> bulks are
> +pushed and popped in LIFO manner, but objects within each bulk are not
> ordered
> +as expected by a stack.
> +
> +The pile implementation generally resembles that of the lock-free
> stack.
> +In addition to the lock-free stack's linked list of solo (single-
> object) elements,
> +it also contains a linked list of bulk (multi-object) elements.
> +And similar to the linked list of free elements, it contains two
> linked lists of
> +free elements, one for each element type (bulk and solo).
> +The lock-free property means that multiple threads can push and pop
> simultaneously.
> +One thread being preempted/delayed in a push or pop operation will not
> +impede the forward progress of any other thread.
> +
> +Push operations are performed by splitting the burst in two: objects
> fitting into
> +bulk elements, and any remaining objects (after filling bulk elements)
> into
> +solo elements, and then performaing two lock-free push operations,
> +one for each element type (solo and bulk).
> +
> +Pop operations are performed by splitting the burst in two: objects
> fitting into
> +bulk elements, and any remaining objects (not filling a bulk element)
> into
> +solo elements. Two lock-free pop operations are performed,
> +first for bulk elements, and then for solo elements.
> +If the pop operation for bulk elements fails, it keeps retrying,
> requesting one
> +less bulk element. The number of solo elements in the following
> request is
> +correspondingly increased.
> +
> +The pile's lock-free list push and pop operations use the lock-free
> stack's
> +implementations (and uses type casting to mimick C++ class
> inheritance).
> +
> +The linked list elements themselves are maintained in two lock-free
> LIFOs,
> +one for bulk elements and one for solo elements, and are
> +allocated before pushes and freed after pops. Since the pile has a
> +fixed maximum depth, these elements do not need to be dynamically
> created.
> +
> +The pile behavior is selected by passing the ``RTE_STACK_F_PILE`` flag
> to
> +``rte_stack_create()``.
> +
> +The pile bulk size can be changed by modifying
> ``RTE_STACK_PILE_BULK_SIZE`` in
> +``config/rte_config.h``.
> +For optimal performance when using the pile mempool driver, the
> +mempool cache size / 2 should be divisible by the pile bulk size.
> +
> +Note:
> +The pile is designed and optimized for use with bulks of objects.
> +Bursts not a multiple of the bulk size are still handled in a lock-
> free,
> +forward-progress-guaranteed manner. However, pop operations may
> exhibit
> +significantly lower performance in instances where the optimal number
> of
> +bulk elements is unavailable, and it is necessary to retry (fetching
> +increasingly fewer bulk elements and correspondingly more solo
> elements).
> diff --git a/drivers/mempool/stack/rte_mempool_stack.c
> b/drivers/mempool/stack/rte_mempool_stack.c
> index 1476905227..7467b8b39e 100644
> --- a/drivers/mempool/stack/rte_mempool_stack.c
> +++ b/drivers/mempool/stack/rte_mempool_stack.c
> @@ -41,6 +41,36 @@ lf_stack_alloc(struct rte_mempool *mp)
> return __stack_alloc(mp, RTE_STACK_F_LF);
> }
>
> +static int
> +pile_alloc(struct rte_mempool *mp)
> +{
> + return __stack_alloc(mp, RTE_STACK_F_PILE);
> +}
> +
> +static int
> +pile_enqueue(struct rte_mempool *mp, void * const *obj_table,
> + unsigned int n)
> +{
> + struct rte_stack *s = mp->pool_data;
> +
> + RTE_ASSERT(s != NULL);
> + RTE_ASSERT(obj_table != NULL);
> +
> + return __rte_stack_pile_push(s, obj_table, n) == 0 ? -ENOBUFS :
> 0;
> +}
> +
> +static int
> +pile_dequeue(struct rte_mempool *mp, void **obj_table,
> + unsigned int n)
> +{
> + struct rte_stack *s = mp->pool_data;
> +
> + RTE_ASSERT(s != NULL);
> + RTE_ASSERT(obj_table != NULL);
> +
> + return __rte_stack_pile_pop(s, obj_table, n) == 0 ? -ENOBUFS : 0;
> +}
> +
> static int
> stack_enqueue(struct rte_mempool *mp, void * const *obj_table,
> unsigned int n)
> @@ -93,5 +123,15 @@ static struct rte_mempool_ops ops_lf_stack = {
> .get_count = stack_get_count
> };
>
> +static struct rte_mempool_ops ops_pile = {
> + .name = "pile",
> + .alloc = pile_alloc,
> + .free = stack_free,
> + .enqueue = pile_enqueue,
> + .dequeue = pile_dequeue,
> + .get_count = stack_get_count
> +};
> +
> RTE_MEMPOOL_REGISTER_OPS(ops_stack);
> RTE_MEMPOOL_REGISTER_OPS(ops_lf_stack);
> +RTE_MEMPOOL_REGISTER_OPS(ops_pile);
> diff --git a/drivers/net/bonding/rte_eth_bond_pmd.c
> b/drivers/net/bonding/rte_eth_bond_pmd.c
> index 6a4f997b5a..92d7f4f4ef 100644
> --- a/drivers/net/bonding/rte_eth_bond_pmd.c
> +++ b/drivers/net/bonding/rte_eth_bond_pmd.c
> @@ -1702,7 +1702,7 @@ member_configure_slow_queue(struct rte_eth_dev
> *bonding_eth_dev,
> snprintf(mem_name, RTE_DIM(mem_name),
> "member_port%u_slow_pool",
> member_id);
> port->slow_pool = rte_pktmbuf_pool_create(mem_name, 8191,
> - 250, 0, RTE_MBUF_DEFAULT_BUF_SIZE,
> + 256, 0, RTE_MBUF_DEFAULT_BUF_SIZE,
> member_eth_dev->data->numa_node);
>
> /* Any memory allocation failure in initialization is
> critical because
> diff --git a/drivers/net/intel/cpfl/cpfl_rxtx.h
> b/drivers/net/intel/cpfl/cpfl_rxtx.h
> index 52cdecac88..faf28fd489 100644
> --- a/drivers/net/intel/cpfl/cpfl_rxtx.h
> +++ b/drivers/net/intel/cpfl/cpfl_rxtx.h
> @@ -25,7 +25,7 @@
> #define CPFL_P2P_QUEUE_GRP_ID 1
> #define CPFL_P2P_DESC_LEN 16
> #define CPFL_P2P_NB_MBUF 4096
> -#define CPFL_P2P_CACHE_SIZE 250
> +#define CPFL_P2P_CACHE_SIZE 256
> #define CPFL_P2P_MBUF_SIZE 2048
> #define CPFL_P2P_RING_BUF 128
>
> diff --git a/drivers/net/sxe2/sxe2_txrx_vec_avx512.c
> b/drivers/net/sxe2/sxe2_txrx_vec_avx512.c
> index a830c7a33b..4ded5cb63e 100644
> --- a/drivers/net/sxe2/sxe2_txrx_vec_avx512.c
> +++ b/drivers/net/sxe2/sxe2_txrx_vec_avx512.c
> @@ -67,11 +67,12 @@ static __rte_always_inline int32_t
> sxe2_tx_bufs_free_vec_avx512(struct sxe2_tx_q
> }
> cache->len += rs_thresh;
>
> - if (cache->len >= cache->flushthresh) {
> + if (cache->len >= cache->size) {
> (void)rte_mempool_ops_enqueue_bulk(mp,
> &cache->objs[cache->size], cache->len -
> cache->size);
> cache->len = cache->size;
> }
> +
> goto done;
> }
>
> diff --git a/drivers/net/tap/rte_eth_tap.c
> b/drivers/net/tap/rte_eth_tap.c
> index b93452f168..b3142561c2 100644
> --- a/drivers/net/tap/rte_eth_tap.c
> +++ b/drivers/net/tap/rte_eth_tap.c
> @@ -61,7 +61,7 @@
> #define TAP_MAX_MAC_ADDRS 16
> #define TAP_GSO_MBUFS_PER_CORE 128
> #define TAP_GSO_MBUF_SEG_SIZE 128
> -#define TAP_GSO_MBUF_CACHE_SIZE 4
> +#define TAP_GSO_MBUF_CACHE_SIZE 32
> #define TAP_GSO_MBUFS_NUM \
> (TAP_GSO_MBUFS_PER_CORE * TAP_GSO_MBUF_CACHE_SIZE)
>
> diff --git a/lib/eal/include/rte_common.h
> b/lib/eal/include/rte_common.h
> index 79d2a0ab93..0fd0906506 100644
> --- a/lib/eal/include/rte_common.h
> +++ b/lib/eal/include/rte_common.h
> @@ -567,6 +567,15 @@ static void
> __attribute__((destructor(RTE_PRIO(prio)), used)) func(void)
> #define __rte_assume(condition) __assume(condition)
> #endif
>
> +/**
> + * Alignment hint precondition
> + */
> +#ifdef RTE_TOOLCHAIN_MSVC
> +#define __rte_assume_aligned(ptr, alignment) (ptr)
> +#else
> +#define __rte_assume_aligned(ptr, alignment)
> __builtin_assume_aligned(ptr, alignment)
> +#endif
> +
> /**
> * Disable AddressSanitizer on some code
> */
> @@ -775,6 +784,9 @@ rte_is_aligned(const void * const __rte_restrict
> ptr, const unsigned int align)
> /** Force minimum cache line alignment. */
> #define __rte_cache_min_aligned __rte_aligned(RTE_CACHE_LINE_MIN_SIZE)
>
> +/** Cache alignment hint precondition */
> +#define __rte_assume_cache_aligned(ptr) __rte_assume_aligned(ptr,
> RTE_CACHE_LINE_SIZE)
> +
> #define _RTE_CACHE_GUARD_HELPER2(unique) \
> alignas(RTE_CACHE_LINE_SIZE) \
> char cache_guard_ ## unique[RTE_CACHE_LINE_SIZE *
> RTE_CACHE_GUARD_LINES]
> diff --git a/lib/eal/x86/include/rte_memcpy.h
> b/lib/eal/x86/include/rte_memcpy.h
> index 8ed8c55010..3fe1e8d247 100644
> --- a/lib/eal/x86/include/rte_memcpy.h
> +++ b/lib/eal/x86/include/rte_memcpy.h
> @@ -707,6 +707,35 @@ rte_memcpy(void *__rte_restrict dst, const void
> *__rte_restrict src, size_t n)
> #endif
> return dst;
> }
> + /* Common way for small copy size of 64-byte blocks */
> +#if defined __AVX512F__ && defined RTE_MEMCPY_AVX512
> + if (__rte_constant(n) && (n & 63) == 0 && n <= 512) {
> +#elif defined RTE_MEMCPY_AVX
> + if (__rte_constant(n) && (n & 63) == 0 && n <= 256) {
> +#else /* SSE implementation */
> + if (__rte_constant(n) && (n & 63) == 0 && n <= 512) {
> +#endif
> + void *ret = dst;
> +
> + if (n & 512) {
> + rte_mov256((uint8_t *)dst + 0 * 256, (const uint8_t
> *)src + 0 * 256);
> + rte_mov256((uint8_t *)dst + 1 * 256, (const uint8_t
> *)src + 1 * 256);
> + }
> + if (n & 256) {
> + rte_mov256((uint8_t *)dst, (const uint8_t *)src);
> + src = (const uint8_t *)src + 256;
> + dst = (uint8_t *)dst + 256;
> + }
> + if (n & 128) {
> + rte_mov128((uint8_t *)dst, (const uint8_t *)src);
> + src = (const uint8_t *)src + 128;
> + dst = (uint8_t *)dst + 128;
> + }
> + if (n & 64)
> + rte_mov64((uint8_t *)dst, (const uint8_t *)src);
> +
> + return ret;
> + }
>
> /* Implementation for size > 64 bytes depends on alignment with
> vector register size. */
> if (!(((uintptr_t)dst | (uintptr_t)src) & ALIGNMENT_MASK))
> diff --git a/lib/mempool/mempool_trace.h b/lib/mempool/mempool_trace.h
> index 23cda1473c..60e47cf67b 100644
> --- a/lib/mempool/mempool_trace.h
> +++ b/lib/mempool/mempool_trace.h
> @@ -119,7 +119,6 @@ RTE_TRACE_POINT(
> rte_trace_point_emit_i32(socket_id);
> rte_trace_point_emit_ptr(cache);
> rte_trace_point_emit_u32(cache->len);
> - rte_trace_point_emit_u32(cache->flushthresh);
> )
>
> RTE_TRACE_POINT(
> diff --git a/lib/mempool/rte_mempool.c b/lib/mempool/rte_mempool.c
> index 817e2b8dc1..457ef8fd1b 100644
> --- a/lib/mempool/rte_mempool.c
> +++ b/lib/mempool/rte_mempool.c
> @@ -753,14 +753,13 @@ static void
> mempool_cache_init(struct rte_mempool_cache *cache, uint32_t size)
> {
> cache->size = size;
> - cache->flushthresh = size; /* Obsolete; for API/ABI compatibility
> purposes only */
> cache->len = 0;
> }
>
> /*
> * Create and initialize a cache for objects that are retrieved from
> and
> * returned to an underlying mempool. This structure is identical to
> the
> - * local_cache[lcore_id] pointed to by the mempool structure.
> + * local_cache[lcore_id] entry in the mempool structure.
> */
> RTE_EXPORT_SYMBOL(rte_mempool_cache_create)
> struct rte_mempool_cache *
> @@ -838,9 +837,21 @@ rte_mempool_create_empty(const char *name,
> unsigned n, unsigned elt_size,
> return NULL;
> }
>
> + /*
> + * Alignment requirement for performance optimized move within
> the mempool cache.
> + * @ref rte_mempool_do_generic_put() implementation.
> + */
> + if (cache_size & 31) {
> + unsigned int rounded = RTE_ALIGN_MUL_CEIL(cache_size, 32);
> + RTE_MEMPOOL_LOG(WARNING, "%s cache size %u not divisible by
> 32, using %u instead.",
> + name, cache_size, rounded);
> + cache_size = rounded;
> + }
> +
> /* asked cache too big */
> if (cache_size > RTE_MEMPOOL_CACHE_MAX_SIZE ||
> cache_size > n) {
> + RTE_MEMPOOL_LOG(ERR, "Cache size too big.");
> rte_errno = EINVAL;
> return NULL;
> }
> @@ -884,7 +895,7 @@ rte_mempool_create_empty(const char *name, unsigned
> n, unsigned elt_size,
> goto exit_unlock;
> }
>
> - mempool_size = RTE_MEMPOOL_HEADER_SIZE(mp, cache_size);
> + mempool_size = sizeof(struct rte_mempool);
> mempool_size += private_data_size;
> mempool_size = RTE_ALIGN_CEIL(mempool_size, RTE_MEMPOOL_ALIGN);
>
> @@ -900,7 +911,7 @@ rte_mempool_create_empty(const char *name, unsigned
> n, unsigned elt_size,
>
> /* init the mempool structure */
> mp = mz->addr;
> - memset(mp, 0, RTE_MEMPOOL_HEADER_SIZE(mp, cache_size));
> + memset(mp, 0, mempool_size);
> ret = strlcpy(mp->name, name, sizeof(mp->name));
> if (ret < 0 || ret >= (int)sizeof(mp->name)) {
> rte_errno = ENAMETOOLONG;
> @@ -937,13 +948,6 @@ rte_mempool_create_empty(const char *name,
> unsigned n, unsigned elt_size,
> goto exit_unlock;
> }
>
> - /*
> - * local_cache pointer is set even if cache_size is zero.
> - * The local_cache points to just past the elt_pa[] array.
> - */
> - mp->local_cache = (struct rte_mempool_cache *)
> - RTE_PTR_ADD(mp, RTE_MEMPOOL_HEADER_SIZE(mp, 0));
> -
> /* Init all default caches. */
> if (cache_size != 0) {
> for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++)
> @@ -1197,6 +1201,7 @@ mempool_obj_audit(struct rte_mempool *mp,
> __rte_unused void *opaque,
> RTE_MEMPOOL_CHECK_COOKIES(mp, &obj, 1, 2);
> }
>
> +/* check cookies before and after objects */
> static void
> mempool_audit_cookies(struct rte_mempool *mp)
> {
> @@ -1213,23 +1218,28 @@ mempool_audit_cookies(struct rte_mempool *mp)
> #define mempool_audit_cookies(mp) do {} while(0)
> #endif
>
> -/* check cookies before and after objects */
> +/* check cache size consistency */
> static void
> mempool_audit_cache(const struct rte_mempool *mp)
> {
> - /* check cache size consistency */
> unsigned lcore_id;
> + const uint32_t cache_size = mp->cache_size;
>
> - if (mp->cache_size == 0)
> - return;
> + if (cache_size > RTE_MEMPOOL_CACHE_MAX_SIZE) {
> + RTE_MEMPOOL_LOG(CRIT, "badness on cache size");
> + rte_panic("MEMPOOL: invalid cache size\n");
> + }
>
> for (lcore_id = 0; lcore_id < RTE_MAX_LCORE; lcore_id++) {
> const struct rte_mempool_cache *cache;
> cache = &mp->local_cache[lcore_id];
> - if (cache->len > RTE_DIM(cache->objs)) {
> - RTE_MEMPOOL_LOG(CRIT, "badness on cache[%u]",
> - lcore_id);
> - rte_panic("MEMPOOL: invalid cache len\n");
> + if (cache->size != cache_size) {
> + RTE_MEMPOOL_LOG(CRIT, "badness on cache[%u] size",
> lcore_id);
> + rte_panic("MEMPOOL: invalid cache[%u] size\n",
> lcore_id);
> + }
> + if (cache->len > cache_size) {
> + RTE_MEMPOOL_LOG(CRIT, "badness on cache[%u] len",
> lcore_id);
> + rte_panic("MEMPOOL: invalid cache[%u] len\n",
> lcore_id);
> }
> }
> }
> @@ -1241,9 +1251,6 @@ rte_mempool_audit(struct rte_mempool *mp)
> {
> mempool_audit_cache(mp);
> mempool_audit_cookies(mp);
> -
> - /* For case where mempool DEBUG is not set, and cache size is 0
> */
> - RTE_SET_USED(mp);
> }
>
> /* dump the status of the mempool on the console */
> diff --git a/lib/mempool/rte_mempool.h b/lib/mempool/rte_mempool.h
> index 50d958c7c6..49e8401280 100644
> --- a/lib/mempool/rte_mempool.h
> +++ b/lib/mempool/rte_mempool.h
> @@ -89,14 +89,14 @@ struct __rte_cache_aligned rte_mempool_debug_stats
> {
> */
> struct __rte_cache_aligned rte_mempool_cache {
> uint32_t size; /**< Size of the cache */
> - uint32_t flushthresh; /**< Obsolete; for API/ABI compatibility
> purposes only */
> uint32_t len; /**< Current cache count */
> #ifdef RTE_LIBRTE_MEMPOOL_STATS
> - uint32_t unused;
> /*
> * Alternative location for the most frequently updated mempool
> statistics (per-lcore),
> * providing faster update access when using a mempool cache.
> + * Note: 16-byte aligned for optimal SIMD access, when updating
> pairs of counters.
> */
> + alignas(16)
> struct {
> uint64_t put_bulk; /**< Number of puts. */
> uint64_t put_objs; /**< Number of objects
> successfully put. */
> @@ -104,15 +104,9 @@ struct __rte_cache_aligned rte_mempool_cache {
> uint64_t get_success_objs; /**< Objects successfully
> allocated. */
> } stats; /**< Statistics */
> #endif
> - /**
> - * Cache objects
> - *
> - * Note:
> - * Cache is allocated at double size for API/ABI compatibility
> purposes only.
> - * When reducing its size at an API/ABI breaking release,
> - * remember to add a cache guard after it.
> - */
> - alignas(RTE_CACHE_LINE_SIZE) void
> *objs[RTE_MEMPOOL_CACHE_MAX_SIZE * 2];
> + /** Cache objects */
> + alignas(RTE_CACHE_LINE_SIZE) void
> *objs[RTE_MEMPOOL_CACHE_MAX_SIZE];
> + RTE_CACHE_GUARD;
> };
>
> /**
> @@ -240,8 +234,7 @@ struct __rte_cache_aligned rte_mempool {
> unsigned int flags; /**< Flags of the mempool. */
> int socket_id; /**< Socket id passed at create.
> */
> uint32_t size; /**< Max size of the mempool. */
> - uint32_t cache_size;
> - /**< Size of per-lcore default local cache. */
> + uint32_t cache_size; /**< Size of per-lcore default
> local cache. */
>
> uint32_t elt_size; /**< Size of an element. */
> uint32_t header_size; /**< Size of header (before
> elt). */
> @@ -257,13 +250,13 @@ struct __rte_cache_aligned rte_mempool {
> */
> int32_t ops_index;
>
> - struct rte_mempool_cache *local_cache; /**< Per-lcore local cache
> */
> -
> uint32_t populated_size; /**< Number of populated
> objects. */
> struct rte_mempool_objhdr_list elt_list; /**< List of objects in
> pool */
> uint32_t nb_mem_chunks; /**< Number of memory chunks */
> struct rte_mempool_memhdr_list mem_list; /**< List of memory
> chunks */
>
> + struct rte_mempool_cache local_cache[RTE_MAX_LCORE]; /**< Per-
> lcore local cache */
> +
> #ifdef RTE_LIBRTE_MEMPOOL_STATS
> /** Per-lcore statistics.
> *
> @@ -271,6 +264,8 @@ struct __rte_cache_aligned rte_mempool {
> */
> struct rte_mempool_debug_stats stats[RTE_MAX_LCORE + 1];
> #endif
> +
> + /* Private data are located immediately after the mempool
> structure. */
> };
>
> /** Spreading among memory channels not required. */
> @@ -362,18 +357,6 @@ struct __rte_cache_aligned rte_mempool {
> #define RTE_MEMPOOL_CACHE_STAT_ADD(cache, name, n) do {} while (0)
> #endif
>
> -/**
> - * @internal Calculate the size of the mempool header.
> - *
> - * @param mp
> - * Pointer to the memory pool.
> - * @param cs
> - * Size of the per-lcore cache.
> - */
> -#define RTE_MEMPOOL_HEADER_SIZE(mp, cs) \
> - (sizeof(*(mp)) + (((cs) == 0) ? 0 : \
> - (sizeof(struct rte_mempool_cache) * RTE_MAX_LCORE)))
> -
> /* return the header of a mempool object (internal) */
> static inline struct rte_mempool_objhdr *
> rte_mempool_get_header(void *obj)
> @@ -718,7 +701,7 @@ struct __rte_cache_aligned rte_mempool_ops {
> rte_mempool_dequeue_contig_blocks_t dequeue_contig_blocks;
> };
>
> -#define RTE_MEMPOOL_MAX_OPS_IDX 16 /**< Max registered ops structs */
> +#define RTE_MEMPOOL_MAX_OPS_IDX 32 /**< Max registered ops structs */
>
> /**
> * Structure storing the table of registered ops structs, each of
> which contain
> @@ -1049,7 +1032,7 @@ rte_mempool_free(struct rte_mempool *mp);
> * If cache_size is non-zero, the rte_mempool library will try to
> * limit the accesses to the common lockless pool, by maintaining a
> * per-lcore object cache. This argument must be lower or equal to
> - * RTE_MEMPOOL_CACHE_MAX_SIZE and n.
> + * RTE_MEMPOOL_CACHE_MAX_SIZE and n, and it must be divisible by 32.
> * The access to the per-lcore table is of course
> * faster than the multi-producer/consumer pool. The cache can be
> * disabled if the cache_size argument is set to 0; it can be useful
> to
> @@ -1368,15 +1351,16 @@ rte_mempool_cache_free(struct rte_mempool_cache
> *cache);
> static __rte_always_inline struct rte_mempool_cache *
> rte_mempool_default_cache(struct rte_mempool *mp, unsigned lcore_id)
> {
> - if (unlikely(mp->cache_size == 0))
> + if (unlikely(lcore_id == LCORE_ID_ANY))
> return NULL;
>
> - if (unlikely(lcore_id == LCORE_ID_ANY))
> + struct rte_mempool_cache *cache = &mp->local_cache[lcore_id];
> +
> + if (unlikely(cache->size == 0))
> return NULL;
>
> - rte_mempool_trace_default_cache(mp, lcore_id,
> - &mp->local_cache[lcore_id]);
> - return &mp->local_cache[lcore_id];
> + rte_mempool_trace_default_cache(mp, lcore_id, cache);
> + return cache;
> }
>
> /**
> @@ -1445,9 +1429,22 @@ rte_mempool_do_generic_put(struct rte_mempool
> *mp, void * const *obj_table,
> * are more hot, from the upper half of the cache.
> */
> __rte_assume(cache->len > cache->size / 2);
> - rte_mempool_ops_enqueue_bulk(mp, &cache->objs[0], cache-
> >size / 2);
> - rte_memcpy(&cache->objs[0], &cache->objs[cache->size / 2],
> - sizeof(void *) * (cache->len - cache->size /
> 2));
> + rte_mempool_ops_enqueue_bulk(mp, cache->objs, cache->size /
> 2);
> + /*
> + * For improved rte_memcpy() performance, move down objects
> + * from CPU cache line aligned address in chunks of 32
> bytes.
> + * Note: For cache->objs[cache->size / 2] to be cache line
> aligned, cache->size
> + * must be divisible by 32 on 32-bit architecture with 64-
> byte cache line,
> + * divisible by 32 on 64-bit architecture with 128-byte
> cache line, and
> + * be divisible by 16 on 64-bit architecture with 64-byte
> cache line.
> + * For API consistency, require mempool cache size is
> divisible by 32.
> + */
> + const size_t move = RTE_ALIGN_MUL_CEIL(
> + sizeof(void *) * (cache->len - cache->size /
> 2), 32);
> + __rte_assume(move >= 32);
> + __rte_assume((move & 31) == 0);
> + rte_memcpy(cache->objs, __rte_assume_cache_aligned(&cache-
> >objs[cache->size / 2]),
> + move);
> cache_objs = &cache->objs[cache->len - cache->size / 2];
> cache->len = cache->len - cache->size / 2 + n;
> } else {
> @@ -1892,8 +1889,7 @@ void rte_mempool_audit(struct rte_mempool *mp);
> */
> static inline void *rte_mempool_get_priv(struct rte_mempool *mp)
> {
> - return (char *)mp +
> - RTE_MEMPOOL_HEADER_SIZE(mp, mp->cache_size);
> + return (char *)mp + sizeof(struct rte_mempool);
> }
>
> /**
> diff --git a/lib/stack/meson.build b/lib/stack/meson.build
> index 18177a742f..50e688522e 100644
> --- a/lib/stack/meson.build
> +++ b/lib/stack/meson.build
> @@ -1,7 +1,7 @@
> # SPDX-License-Identifier: BSD-3-Clause
> # Copyright(c) 2019 Intel Corporation
>
> -sources = files('rte_stack.c', 'rte_stack_std.c', 'rte_stack_lf.c')
> +sources = files('rte_stack.c', 'rte_stack_std.c', 'rte_stack_lf.c',
> 'rte_stack_pile.c')
> headers = files('rte_stack.h')
> # subheaders, not for direct inclusion by apps
> indirect_headers += files(
> @@ -10,4 +10,5 @@ indirect_headers += files(
> 'rte_stack_lf_generic.h',
> 'rte_stack_lf_c11.h',
> 'rte_stack_lf_stubs.h',
> + 'rte_stack_pile.h',
> )
> diff --git a/lib/stack/rte_stack.c b/lib/stack/rte_stack.c
> index 4c78fe4b4b..a4bbf8a4d7 100644
> --- a/lib/stack/rte_stack.c
> +++ b/lib/stack/rte_stack.c
> @@ -1,5 +1,6 @@
> /* SPDX-License-Identifier: BSD-3-Clause
> * Copyright(c) 2019 Intel Corporation
> + * Copyright(c) 2026 SmartShare Systems
> */
>
> #include <stdalign.h>
> @@ -32,6 +33,8 @@ rte_stack_init(struct rte_stack *s, unsigned int
> count, uint32_t flags)
>
> if (flags & RTE_STACK_F_LF)
> rte_stack_lf_init(s, count);
> + else if (flags & RTE_STACK_F_PILE)
> + rte_stack_pile_init(s, count);
> else
> rte_stack_std_init(s);
> }
> @@ -41,6 +44,8 @@ rte_stack_get_memsize(unsigned int count, uint32_t
> flags)
> {
> if (flags & RTE_STACK_F_LF)
> return rte_stack_lf_get_memsize(count);
> + else if (flags & RTE_STACK_F_PILE)
> + return rte_stack_pile_get_memsize(count);
> else
> return rte_stack_std_get_memsize(count);
> }
> @@ -58,7 +63,11 @@ rte_stack_create(const char *name, unsigned int
> count, int socket_id,
> unsigned int sz;
> int ret;
>
> - if (flags & ~(RTE_STACK_F_LF)) {
> + if (flags & ~(RTE_STACK_F_LF | RTE_STACK_F_PILE)) {
> + STACK_LOG_ERR("Unsupported stack flags %#x", flags);
> + return NULL;
> + }
> + if ((flags & RTE_STACK_F_LF) && (flags & RTE_STACK_F_PILE)) {
> STACK_LOG_ERR("Unsupported stack flags %#x", flags);
> return NULL;
> }
> @@ -73,6 +82,13 @@ rte_stack_create(const char *name, unsigned int
> count, int socket_id,
> return NULL;
> }
> #endif
> +#if !defined(RTE_STACK_PILE_SUPPORTED)
> + if (flags & RTE_STACK_F_PILE) {
> + STACK_LOG_ERR("Pile is not supported on your platform");
> + rte_errno = ENOTSUP;
> + return NULL;
> + }
> +#endif
>
> sz = rte_stack_get_memsize(count, flags);
>
> diff --git a/lib/stack/rte_stack.h b/lib/stack/rte_stack.h
> index fd17ac791d..bf64dbc7bc 100644
> --- a/lib/stack/rte_stack.h
> +++ b/lib/stack/rte_stack.h
> @@ -1,5 +1,6 @@
> /* SPDX-License-Identifier: BSD-3-Clause
> * Copyright(c) 2019 Intel Corporation
> + * Copyright(c) 2026 SmartShare Systems
> */
>
> /**
> @@ -28,11 +29,45 @@
> #define RTE_STACK_NAMESIZE (RTE_MEMZONE_NAMESIZE - \
> sizeof(RTE_STACK_MZ_PREFIX) + 1)
>
> +static_assert(((sizeof(void *) * RTE_STACK_PILE_BULK_SIZE) &
> RTE_CACHE_LINE_MASK) == 0,
> + "Pile bulk size must be divisible by CPU cache line size");
> +
> +/* Note: Also used as solo (single-object) pile element. */
> struct rte_stack_lf_elem {
> void *data; /**< Data pointer */
> struct rte_stack_lf_elem *next; /**< Next pointer */
> };
>
> +/*
> + * Bulk (multi-object) pile element.
> + * Inherited from the rte_stack_lf_elem (single-object) class,
> + * and extended with an array for holding a bulk of object pointers.
> + */
> +struct rte_stack_pile_bulk_elem {
> + /* The first part must be compatible with the rte_stack_lf_elem
> parent class. */
> + void *data; /**< Data pointer
> (unused) */
> + struct rte_stack_pile_bulk_elem *next; /**< Next pointer */
> + /* The second part differs. */
> + alignas(RTE_CACHE_LINE_SIZE)
> + void *objs[RTE_STACK_PILE_BULK_SIZE]; /**< Bulk (multi-
> object) pointers */
> +};
> +
> +static_assert(sizeof(struct rte_stack_lf_elem) ==
> + sizeof(struct rte_stack_lf_elem *) + sizeof(void*),
> + "Parent type has changed");
> +static_assert(RTE_SIZEOF_FIELD(struct rte_stack_lf_elem, next) ==
> + RTE_SIZEOF_FIELD(struct rte_stack_pile_bulk_elem, next),
> + "Inherited type mismatch");
> +static_assert(offsetof(struct rte_stack_lf_elem, next) ==
> + offsetof(struct rte_stack_pile_bulk_elem, next),
> + "Inherited type mismatch");
> +static_assert(RTE_SIZEOF_FIELD(struct rte_stack_lf_elem, data) ==
> + RTE_SIZEOF_FIELD(struct rte_stack_pile_bulk_elem, data),
> + "Inherited type mismatch");
> +static_assert(offsetof(struct rte_stack_lf_elem, data) ==
> + offsetof(struct rte_stack_pile_bulk_elem, data),
> + "Inherited type mismatch");
> +
> struct __rte_aligned(16) rte_stack_lf_head {
> struct rte_stack_lf_elem *top; /**< Stack top */
> uint64_t cnt; /**< Modification counter for avoiding ABA problem
> */
> @@ -51,12 +86,35 @@ struct rte_stack_lf_list {
> struct rte_stack_lf {
> /** LIFO list of elements */
> alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_list used;
> + RTE_CACHE_GUARD;
> /** LIFO list of free elements */
> alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_list free;
> + RTE_CACHE_GUARD;
> /** LIFO elements */
> alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_elem elems[];
> };
>
> +/* Pile structure containing three lock-free LIFO-like lists:
> + * - A list of elements, each element holding a bulk of pointers to
> objects.
> + * - A list of elements, each element holding one pointer to an
> object.
> + * - A list of free linked-list elements.
> + */
> +struct rte_stack_pile {
> + /** LIFO list of bulk (multi-object) elements */
> + alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_list bulk;
> + RTE_CACHE_GUARD;
> + /** LIFO list of solo (single-object) elements */
> + alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_list solo;
> + RTE_CACHE_GUARD;
> + /** LIFO list of free bulk elements */
> + alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_list free_bulk;
> + RTE_CACHE_GUARD;
> + /** LIFO list of free solo elements */
> + alignas(RTE_CACHE_LINE_SIZE) struct rte_stack_lf_list free_solo;
> + RTE_CACHE_GUARD;
> + /** LIFO elements follow, first bulk, then solo */
> +};
> +
> /* Structure containing the LIFO, its current length, and a lock for
> mutual
> * exclusion.
> */
> @@ -78,6 +136,7 @@ struct __rte_cache_aligned rte_stack {
> uint32_t flags; /**< Flags supplied at creation. */
> union {
> struct rte_stack_lf stack_lf; /**< Lock-free LIFO
> structure. */
> + struct rte_stack_pile stack_pile; /**< Lock-free pile
> (LIFO-like) structure. */
> struct rte_stack_std stack_std; /**< LIFO structure. */
> };
> };
> @@ -88,8 +147,16 @@ struct __rte_cache_aligned rte_stack {
> */
> #define RTE_STACK_F_LF 0x0001
>
> +/**
> + * The stack-like pile uses lock-free push and pop functions.
> + * It is optimized for bulks of objects, and is not strictly LIFO.
> + * This flag is only supported on x86_64 or arm64 platforms,
> currently.
> + */
> +#define RTE_STACK_F_PILE 0x0002
> +
> #include "rte_stack_std.h"
> #include "rte_stack_lf.h"
> +#include "rte_stack_pile.h"
>
> #ifdef __cplusplus
> extern "C" {
> @@ -108,13 +175,15 @@ extern "C" {
> * Actual number of objects pushed (either 0 or *n*).
> */
> static __rte_always_inline unsigned int
> -rte_stack_push(struct rte_stack *s, void * const *obj_table, unsigned
> int n)
> +rte_stack_push(struct rte_stack *s, void * const * __rte_restrict
> obj_table, unsigned int n)
> {
> RTE_ASSERT(s != NULL);
> RTE_ASSERT(obj_table != NULL);
>
> if (s->flags & RTE_STACK_F_LF)
> return __rte_stack_lf_push(s, obj_table, n);
> + else if (s->flags & RTE_STACK_F_PILE)
> + return __rte_stack_pile_push(s, obj_table, n);
> else
> return __rte_stack_std_push(s, obj_table, n);
> }
> @@ -132,13 +201,15 @@ rte_stack_push(struct rte_stack *s, void * const
> *obj_table, unsigned int n)
> * Actual number of objects popped (either 0 or *n*).
> */
> static __rte_always_inline unsigned int
> -rte_stack_pop(struct rte_stack *s, void **obj_table, unsigned int n)
> +rte_stack_pop(struct rte_stack *s, void ** __rte_restrict obj_table,
> unsigned int n)
> {
> RTE_ASSERT(s != NULL);
> RTE_ASSERT(obj_table != NULL);
>
> if (s->flags & RTE_STACK_F_LF)
> return __rte_stack_lf_pop(s, obj_table, n);
> + else if (s->flags & RTE_STACK_F_PILE)
> + return __rte_stack_pile_pop(s, obj_table, n);
> else
> return __rte_stack_std_pop(s, obj_table, n);
> }
> @@ -158,6 +229,8 @@ rte_stack_count(struct rte_stack *s)
>
> if (s->flags & RTE_STACK_F_LF)
> return __rte_stack_lf_count(s);
> + else if (s->flags & RTE_STACK_F_PILE)
> + return __rte_stack_pile_count(s);
> else
> return __rte_stack_std_count(s);
> }
> diff --git a/lib/stack/rte_stack_lf.h b/lib/stack/rte_stack_lf.h
> index f2b012cd0e..655620aaa9 100644
> --- a/lib/stack/rte_stack_lf.h
> +++ b/lib/stack/rte_stack_lf.h
> @@ -34,7 +34,7 @@
> */
> static __rte_always_inline unsigned int
> __rte_stack_lf_push(struct rte_stack *s,
> - void * const *obj_table,
> + void * const * __rte_restrict obj_table,
> unsigned int n)
> {
> struct rte_stack_lf_elem *tmp, *first, *last = NULL;
> @@ -71,7 +71,8 @@ __rte_stack_lf_push(struct rte_stack *s,
> * - Actual number of objects popped.
> */
> static __rte_always_inline unsigned int
> -__rte_stack_lf_pop(struct rte_stack *s, void **obj_table, unsigned int
> n)
> +__rte_stack_lf_pop(struct rte_stack *s, void ** __rte_restrict
> obj_table,
> + unsigned int n)
> {
> struct rte_stack_lf_elem *first, *last = NULL;
>
> @@ -79,6 +80,7 @@ __rte_stack_lf_pop(struct rte_stack *s, void
> **obj_table, unsigned int n)
> return 0;
>
> /* Pop n used elements */
> + __rte_assume(obj_table != NULL);
> first = __rte_stack_lf_pop_elems(&s->stack_lf.used,
> n, obj_table, &last);
> if (unlikely(first == NULL))
> diff --git a/lib/stack/rte_stack_lf_c11.h
> b/lib/stack/rte_stack_lf_c11.h
> index b97e02d6a1..501e985a49 100644
> --- a/lib/stack/rte_stack_lf_c11.h
> +++ b/lib/stack/rte_stack_lf_c11.h
> @@ -99,7 +99,7 @@ __rte_stack_lf_push_elems(struct rte_stack_lf_list
> *list,
> static __rte_always_inline struct rte_stack_lf_elem *
> __rte_stack_lf_pop_elems(struct rte_stack_lf_list *list,
> unsigned int num,
> - void **obj_table,
> + void ** __rte_restrict obj_table,
> struct rte_stack_lf_elem **last)
> {
> struct rte_stack_lf_head old_head;
> diff --git a/lib/stack/rte_stack_lf_generic.h
> b/lib/stack/rte_stack_lf_generic.h
> index cc69e4d168..c4cc4d2c03 100644
> --- a/lib/stack/rte_stack_lf_generic.h
> +++ b/lib/stack/rte_stack_lf_generic.h
> @@ -74,7 +74,7 @@ __rte_stack_lf_push_elems(struct rte_stack_lf_list
> *list,
> static __rte_always_inline struct rte_stack_lf_elem *
> __rte_stack_lf_pop_elems(struct rte_stack_lf_list *list,
> unsigned int num,
> - void **obj_table,
> + void ** __rte_restrict obj_table,
> struct rte_stack_lf_elem **last)
> {
> struct rte_stack_lf_head old_head;
> diff --git a/lib/stack/rte_stack_lf_stubs.h
> b/lib/stack/rte_stack_lf_stubs.h
> index a05abf1f1c..457a415ccf 100644
> --- a/lib/stack/rte_stack_lf_stubs.h
> +++ b/lib/stack/rte_stack_lf_stubs.h
> @@ -30,7 +30,7 @@ __rte_stack_lf_push_elems(struct rte_stack_lf_list
> *list,
> static __rte_always_inline struct rte_stack_lf_elem *
> __rte_stack_lf_pop_elems(struct rte_stack_lf_list *list,
> unsigned int num,
> - void **obj_table,
> + void ** __rte_restrict obj_table,
> struct rte_stack_lf_elem **last)
> {
> RTE_SET_USED(obj_table);
> diff --git a/lib/stack/rte_stack_pile.c b/lib/stack/rte_stack_pile.c
> new file mode 100644
> index 0000000000..eb42ba70a9
> --- /dev/null
> +++ b/lib/stack/rte_stack_pile.c
> @@ -0,0 +1,33 @@
> +/* SPDX-License-Identifier: BSD-3-Clause
> + * Copyright(c) 2026 SmartShare Systems
> + */
> +
> +#include "rte_stack.h"
> +
> +void
> +rte_stack_pile_init(struct rte_stack *s, unsigned int count)
> +{
> + unsigned int bulk = (count + RTE_STACK_PILE_BULK_SIZE - 1) /
> RTE_STACK_PILE_BULK_SIZE;
> + struct rte_stack_pile_bulk_elem * bulk_elems = (struct
> rte_stack_pile_bulk_elem *)(&s->stack_pile + 1);
> + struct rte_stack_lf_elem * solo_elems = (struct rte_stack_lf_elem
> *)&bulk_elems[bulk];
> + unsigned int i;
> +
> + for (i = 0; i < bulk; i++)
> + __rte_stack_pile_bulk_push_elems(&s->stack_pile.free_bulk,
> + &bulk_elems[i], &bulk_elems[i], 1);
> + for (i = 0; i < count; i++)
> + __rte_stack_lf_push_elems(&s->stack_pile.free_solo,
> + &solo_elems[i], &solo_elems[i], 1);
> +}
> +
> +ssize_t
> +rte_stack_pile_get_memsize(unsigned int count)
> +{
> + unsigned int bulk = (count + RTE_STACK_PILE_BULK_SIZE - 1) /
> RTE_STACK_PILE_BULK_SIZE;
> + ssize_t sz = sizeof(struct rte_stack); /* Already cache line
> aligned. */
> + sz += bulk * sizeof(struct rte_stack_pile_bulk_elem); /* Already
> cache line aligned. */
> + sz += RTE_CACHE_LINE_ROUNDUP(count * sizeof(struct
> rte_stack_lf_elem));
> + sz += RTE_CACHE_GUARD_LINES * RTE_CACHE_LINE_SIZE;
> +
> + return sz;
> +}
> diff --git a/lib/stack/rte_stack_pile.h b/lib/stack/rte_stack_pile.h
> new file mode 100644
> index 0000000000..d928515d18
> --- /dev/null
> +++ b/lib/stack/rte_stack_pile.h
> @@ -0,0 +1,316 @@
> +/* SPDX-License-Identifier: BSD-3-Clause
> + * Copyright(c) 2026 SmartShare Systems
> + */
> +
> +#ifndef _RTE_STACK_PILE_H_
> +#define _RTE_STACK_PILE_H_
> +
> +#if !(defined(RTE_ARCH_X86_64) || defined(RTE_ARCH_ARM64))
> +#include "rte_stack_lf_stubs.h"
> +#else
> +#ifdef RTE_USE_C11_MEM_MODEL
> +#include "rte_stack_lf_c11.h"
> +#else
> +#include "rte_stack_lf_generic.h"
> +#endif
> +
> +/**
> + * Indicates that RTE_STACK_F_PILE is supported.
> + */
> +#define RTE_STACK_PILE_SUPPORTED
> +#endif
> +
> +static __rte_always_inline unsigned int
> +__rte_stack_pile_count(struct rte_stack *s)
> +{
> + /* stack_lf_push() and stack_lf_pop() do not update the list's
> contents
> + * and stack_lf->len atomically, which can cause the list to
> appear
> + * shorter than it actually is if this function is called while
> other
> + * threads are modifying the list.
> + *
> + * However, given the inherently approximate nature of the
> get_count
> + * callback -- even if the list and its size were updated
> atomically,
> + * the size could change between when get_count executes and when
> the
> + * value is returned to the caller -- this is acceptable.
> + *
> + * The stack_lf->len updates are placed such that the list may
> appear to
> + * have fewer elements than it does, but will never appear to
> have more
> + * elements. If the mempool is near-empty to the point that this
> is a
> + * concern, the user should consider increasing the mempool size.
> + */
> +#ifdef RTE_USE_C11_MEM_MODEL
> + return RTE_MIN((unsigned int)s->capacity,
> + (unsigned int)rte_atomic_load_explicit(&s-
> >stack_pile.bulk.len,
> + rte_memory_order_relaxed) * RTE_STACK_PILE_BULK_SIZE
> +
> + (unsigned int)rte_atomic_load_explicit(&s-
> >stack_pile.solo.len,
> + rte_memory_order_relaxed));
> +#else
> + /* NOTE: review for potential ordering optimization */
> + return RTE_MIN((unsigned int)s->capacity,
> + (unsigned int)rte_atomic_load_explicit(&s-
> >stack_pile.bulk.len,
> + rte_memory_order_seq_cst) * RTE_STACK_PILE_BULK_SIZE
> +
> + (unsigned int)rte_atomic_load_explicit(&s-
> >stack_pile.solo.len,
> + rte_memory_order_seq_cst));
> +#endif
> +}
> +
> +static __rte_always_inline void
> +__rte_stack_pile_bulk_push_elems(struct rte_stack_lf_list *list,
> + struct rte_stack_pile_bulk_elem *first,
> + struct rte_stack_pile_bulk_elem *last,
> + unsigned int num)
> +{
> + __rte_stack_lf_push_elems(list,
> + (struct rte_stack_lf_elem *)first,
> + (struct rte_stack_lf_elem *)last,
> + num);
> +}
> +
> +static __rte_always_inline struct rte_stack_pile_bulk_elem *
> +__rte_stack_pile_bulk_pop_elems(struct rte_stack_lf_list *list,
> + unsigned int num,
> + void ** __rte_restrict obj_table,
> + struct rte_stack_pile_bulk_elem **last)
> +{
> + struct rte_stack_pile_bulk_elem *first = (struct
> rte_stack_pile_bulk_elem *)__rte_stack_lf_pop_elems(list, num, NULL,
> (struct rte_stack_lf_elem **)last);
> + if (first == NULL)
> + return NULL;
> +
> + if (obj_table != NULL) {
> + /* Traverse the list to copy the bulks. */
> + struct rte_stack_pile_bulk_elem *tmp = first;
> + for (unsigned int i = 0; i < num; i++, tmp = tmp->next)
> + rte_memcpy(&obj_table[i * RTE_STACK_PILE_BULK_SIZE],
> tmp->objs, sizeof(void *) * RTE_STACK_PILE_BULK_SIZE);
> + }
> +
> + return first;
> +}
> +
> +/**
> + * Push several objects on the pile (lock-free, MT-safe).
> + *
> + * @param pile
> + * A pointer to the pile structure.
> + * @param obj_table
> + * A pointer to a table of void * pointers (objects).
> + * @param n
> + * The number of objects to push on the pile from the obj_table.
> + * @return
> + * Actual number of objects pushed (either 0 or *n*).
> + */
> +static __rte_always_inline unsigned int
> +__rte_stack_pile_push(struct rte_stack *s,
> + void * const * __rte_restrict obj_table,
> + unsigned int n)
> +{
> + RTE_ASSERT(s != NULL);
> + RTE_ASSERT(obj_table != NULL);
> +
> + struct rte_stack_pile *pile = &s->stack_pile;
> + struct rte_stack_pile_bulk_elem *bulk_first = NULL, *bulk_last =
> NULL;
> + struct rte_stack_lf_elem *solo_first = NULL, *solo_last = NULL;
> + unsigned int n_bulk = n / RTE_STACK_PILE_BULK_SIZE;
> + unsigned int n_solo = n & (RTE_STACK_PILE_BULK_SIZE - 1);
> + unsigned int i;
> +
> + if (unlikely(n_bulk == 0)) {
> + if (unlikely(n_solo == 0))
> + return 0;
> + else
> + goto solo;
> + }
> +
> + /* Allocate n_bulk elements from the free list. */
> + bulk_first = __rte_stack_pile_bulk_pop_elems(&pile->free_bulk,
> n_bulk, NULL, &bulk_last);
> + if (unlikely(bulk_first == NULL))
> + return 0; /* Failed. */
> +
> + if (likely(n_solo == 0))
> + goto bulk;
> +
> +solo:
> + /* Allocate n_solo elements from the free list. */
> + solo_first = __rte_stack_lf_pop_elems(&pile->free_solo, n_solo,
> NULL, &solo_last);
> + if (unlikely(solo_first == NULL)) {
> + /* Failed. Roll back. */
> + if (n_bulk > 0)
> + __rte_stack_pile_bulk_push_elems(&pile->free_bulk,
> bulk_first, bulk_last, n_bulk);
> + return 0;
> + }
> +
> + /*
> + * Construct the solo elements.
> + * Copy the objects, but ignore the object order.
> + */
> + struct rte_stack_lf_elem *tmp_solo = solo_first;
> + __rte_assume(n_solo > 0);
> + __rte_assume(n_solo < RTE_STACK_PILE_BULK_SIZE);
> + for (i = 0; i < n_solo; i++, tmp_solo = tmp_solo->next)
> + tmp_solo->data = obj_table[n_bulk *
> RTE_STACK_PILE_BULK_SIZE + i];
> +
> + /* Push them to the solo list. */
> + __rte_stack_lf_push_elems(&pile->solo, solo_first, solo_last,
> n_solo);
> +
> + if (unlikely(n_bulk == 0))
> + return n; /* Done. */
> +
> +bulk:
> + /*
> + * Construct the bulk elements.
> + * Copy bulks in reverse order, but ignore the object order
> within each bulk.
> + */
> + struct rte_stack_pile_bulk_elem *tmp_bulk = bulk_first;
> + __rte_assume(n_bulk > 0);
> + for (i = 0; i < n_bulk; i++, tmp_bulk = tmp_bulk->next)
> + rte_memcpy(tmp_bulk->objs, &obj_table[(n_bulk - i - 1) *
> RTE_STACK_PILE_BULK_SIZE], sizeof(void *) * RTE_STACK_PILE_BULK_SIZE);
> +
> + /* Push them to the bulk list. */
> + __rte_stack_pile_bulk_push_elems(&pile->bulk, bulk_first,
> bulk_last, n_bulk);
> +
> + return n;
> +}
> +
> +/**
> + * Pop several objects from the pile (lock-free, MT-safe).
> + *
> + * @param pile
> + * A pointer to the pile structure.
> + * @param obj_table
> + * A pointer to a table of void * pointers (objects).
> + * @param n
> + * The number of objects to pull from the pile.
> + * @return
> + * Actual number of objects popped (either 0 or *n*).
> + */
> +static __rte_always_inline unsigned int
> +__rte_stack_pile_pop(struct rte_stack *s,
> + void ** __rte_restrict obj_table,
> + unsigned int n)
> +{
> + RTE_ASSERT(s != NULL);
> + RTE_ASSERT(obj_table != NULL);
> +
> + struct rte_stack_pile *pile = &s->stack_pile;
> + struct rte_stack_pile_bulk_elem *bulk_first = NULL, *bulk_last =
> NULL;
> + struct rte_stack_lf_elem *solo_first = NULL, *solo_last = NULL;
> + unsigned int n_bulk = n / RTE_STACK_PILE_BULK_SIZE;
> + unsigned int n_solo = n & (RTE_STACK_PILE_BULK_SIZE - 1);
> + unsigned int i;
> +
> + if (unlikely(n_bulk == 0)) {
> + if (unlikely(n_solo == 0))
> + return 0;
> + else
> + goto solo;
> + }
> +
> +bulk:
> + /* Fetch n_bulk * RTE_STACK_PILE_BULK_SIZE objects as bulk
> elements. */
> + bulk_first = __rte_stack_pile_bulk_pop_elems(&pile->bulk, n_bulk,
> obj_table, &bulk_last);
> + if (unlikely(bulk_first == NULL)) {
> + /* Not available. Retry with fewer bulk elements; objects
> to be fetched as solo elements instead. */
> + n_solo += RTE_STACK_PILE_BULK_SIZE;
> + n_bulk--;
> + if (n_bulk > 0)
> + goto bulk;
> + else
> + goto solo;
> + }
> +
> + if (likely(n_solo == 0))
> + goto done;
> +
> +solo:
> + /* Fetch n_solo objects as solo elements. */
> + solo_first = __rte_stack_lf_pop_elems(&pile->solo, n_solo,
> &obj_table[n_bulk * RTE_STACK_PILE_BULK_SIZE], &solo_last);
> + if (solo_first != NULL)
> + goto done;
> +
> + /* Solo elements not available. Try fragmentation. */
> + alignas(RTE_CACHE_LINE_SIZE) void *
> obj_frag[RTE_STACK_PILE_BULK_SIZE];
> + struct rte_stack_pile_bulk_elem *frag;
> +
> + /* Fetch a fragmentation element as a bulk element. */
> + frag = __rte_stack_pile_bulk_pop_elems(&pile->bulk, 1, obj_frag,
> NULL);
> + if (unlikely(frag == NULL)) {
> + /* Failed. Roll back. */
> + if (n_bulk > 0)
> + __rte_stack_pile_bulk_push_elems(&pile->bulk,
> bulk_first, bulk_last, n_bulk);
> + return 0;
> + }
> +
> + /* Get n_solo objects from the fragmentation element. */
> + __rte_assume(n_solo > 0);
> + __rte_assume(n_solo < RTE_STACK_PILE_BULK_SIZE);
> + for (i = 0; i < n_solo; i++)
> + obj_table[n_bulk * RTE_STACK_PILE_BULK_SIZE + i] =
> obj_frag[i];
> +
> + /* Fetch free elements for the excess objects. */
> + __rte_assume(RTE_STACK_PILE_BULK_SIZE - n_solo > 0);
> + __rte_assume(RTE_STACK_PILE_BULK_SIZE - n_solo <
> RTE_STACK_PILE_BULK_SIZE - 1);
> + solo_first = __rte_stack_lf_pop_elems(&pile->free_solo,
> RTE_STACK_PILE_BULK_SIZE - n_solo, NULL, &solo_last);
> + if (unlikely(solo_first == NULL)) {
> + /* Failed. Roll back. */
> + struct rte_stack_pile_bulk_elem *last;
> + if (n_bulk > 0) {
> + /* Attach the bulk elements after the fragmentation
> element. */
> + frag->next = bulk_first;
> + last = bulk_last;
> + } else
> + last = frag;
> + __rte_stack_pile_bulk_push_elems(&pile->bulk, frag, last, 1
> + n_bulk);
> + return 0;
> + }
> +
> + /* Construct the solo elements from the excess objects. */
> + struct rte_stack_lf_elem *tmp = solo_first;
> + __rte_assume(n_solo > 0);
> + __rte_assume(n_solo < RTE_STACK_PILE_BULK_SIZE);
> + for (i = n_solo; i < RTE_STACK_PILE_BULK_SIZE; i++, tmp = tmp-
> >next)
> + tmp->data = obj_frag[i];
> +
> + /* Push the excess objects as solo elements. */
> + __rte_stack_lf_push_elems(&pile->solo, solo_first, solo_last,
> RTE_STACK_PILE_BULK_SIZE - n_solo);
> + n_solo = 0;
> +
> + /* Add the fragmentation element in front of the bulk elements,
> so it can be freed. */
> + if (n_bulk > 0)
> + frag->next = bulk_first;
> + else
> + bulk_last = frag;
> + bulk_first = frag;
> + n_bulk++;
> +
> +done:
> + /* Success. Free the elements. */
> + if (n_bulk > 0)
> + __rte_stack_pile_bulk_push_elems(&pile->free_bulk,
> bulk_first, bulk_last, n_bulk);
> + if (n_solo > 0)
> + __rte_stack_lf_push_elems(&pile->free_solo, solo_first,
> solo_last, n_solo);
> +
> + return n;
> +}
> +
> +/**
> + * @internal Initialize a pile stack.
> + *
> + * @param s
> + * A pointer to the stack structure.
> + * @param count
> + * The size of the stack.
> + */
> +void
> +rte_stack_pile_init(struct rte_stack *s, unsigned int count);
> +
> +/**
> + * @internal Return the memory required for a pile stack.
> + *
> + * @param count
> + * The size of the stack.
> + * @return
> + * The bytes to allocate for a pile stack.
> + */
> +ssize_t
> +rte_stack_pile_get_memsize(unsigned int count);
> +
> +#endif /* _RTE_STACK_PILE_H_ */
> diff --git a/lib/stack/rte_stack_std.h b/lib/stack/rte_stack_std.h
> index ae28add5c4..003095a144 100644
> --- a/lib/stack/rte_stack_std.h
> +++ b/lib/stack/rte_stack_std.h
> @@ -6,6 +6,7 @@
> #define _RTE_STACK_STD_H_
>
> #include <rte_branch_prediction.h>
> +#include <rte_memcpy.h>
>
> /**
> * @internal Push several objects on the stack (MT-safe).
> @@ -20,27 +21,24 @@
> * Actual number of objects pushed (either 0 or *n*).
> */
> static __rte_always_inline unsigned int
> -__rte_stack_std_push(struct rte_stack *s, void * const *obj_table,
> +__rte_stack_std_push(struct rte_stack *s, void * const *
> __rte_restrict obj_table,
> unsigned int n)
> {
> - struct rte_stack_std *stack = &s->stack_std;
> - unsigned int index;
> - void **cache_objs;
> + struct rte_stack_std * __rte_restrict stack = &s->stack_std;
> + void ** __rte_restrict stack_objs;
>
> rte_spinlock_lock(&stack->lock);
> - cache_objs = &stack->objs[stack->len];
>
> - /* Is there sufficient space in the stack? */
> - if ((stack->len + n) > s->capacity) {
> + if (unlikely((stack->len + n) > s->capacity)) {
> + /* Insufficient room in the stack. */
> rte_spinlock_unlock(&stack->lock);
> return 0;
> }
>
> - /* Add elements back into the cache */
> - for (index = 0; index < n; ++index, obj_table++)
> - cache_objs[index] = *obj_table;
> -
> + /* Push objects to the stack */
> + stack_objs = &stack->objs[stack->len];
> stack->len += n;
> + rte_memcpy(stack_objs, obj_table, sizeof(void *) * n);
>
> rte_spinlock_unlock(&stack->lock);
> return n;
> @@ -59,28 +57,27 @@ __rte_stack_std_push(struct rte_stack *s, void *
> const *obj_table,
> * Actual number of objects popped (either 0 or *n*).
> */
> static __rte_always_inline unsigned int
> -__rte_stack_std_pop(struct rte_stack *s, void **obj_table, unsigned
> int n)
> +__rte_stack_std_pop(struct rte_stack *s, void ** __rte_restrict
> obj_table, unsigned int n)
> {
> - struct rte_stack_std *stack = &s->stack_std;
> - unsigned int index, len;
> - void **cache_objs;
> + struct rte_stack_std * __rte_restrict stack = &s->stack_std;
> + unsigned int index;
> + void ** __rte_restrict stack_objs;
>
> rte_spinlock_lock(&stack->lock);
>
> if (unlikely(n > stack->len)) {
> + /* Insufficient objects in the stack. */
> rte_spinlock_unlock(&stack->lock);
> return 0;
> }
>
> - cache_objs = stack->objs;
> -
> - for (index = 0, len = stack->len - 1; index < n;
> - ++index, len--, obj_table++)
> - *obj_table = cache_objs[len];
> -
> + /* Pop objects from the stack */
> + stack_objs = &stack->objs[stack->len];
> stack->len -= n;
> - rte_spinlock_unlock(&stack->lock);
> + for (index = 0; index < n; index++)
> + *obj_table++ = *--stack_objs;
>
> + rte_spinlock_unlock(&stack->lock);
> return n;
> }
>
> --
> 2.43.0
next prev parent reply other threads:[~2026-08-01 8:10 UTC|newest]
Thread overview: 3+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-01 6:53 [RFC PATCH] pile stack and mempool driver (resend) Morten Brørup
2026-08-01 8:10 ` Morten Brørup [this message]
2026-08-01 14:59 ` Stephen Hemminger
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=98CBD80474FA8B44BF855DF32C47DC35F659AD@smartserver.smartshare.dk \
--to=mb@smartsharesystems.com \
--cc=dev@dpdk.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox