* [PATCH] net/ice: add per-queue Tx rate limit support
@ 2026-09-10 10:04 Anurag Mandal
2026-09-10 10:29 ` Bruce Richardson
2026-09-17 6:38 ` [PATCH v2] " Anurag Mandal
0 siblings, 2 replies; 7+ messages in thread
From: Anurag Mandal @ 2026-09-10 10:04 UTC (permalink / raw)
To: dev; +Cc: bruce.richardson, anatoly.burakov, Anurag Mandal
The Tx rate can be limited per queue with
ethdev operation ``rte_eth_set_queue_rate_limit()``
and can be read through ``rte_eth_get_queue_rate_limit()``.
This feature uses the hardware packet pacing
mechanism to enforce a data rate on individual
Tx queues without tearing down the queue.
The rate is specified in Mbps.
ice_set_queue_rate_limit() applies the requested rate
as the EIR (maximum bandwidth) limit of the queue
scheduler node using ice_cfg_q_bw_lmt(),
converting the Mbps value taken by the API to the Kbps
expected by the scheduler.
A rate of 0 removes the limit and restores the default
bandwidth via ice_cfg_q_bw_dflt_lmt().
ice_get_queue_rate_limit() reads back the value cached
in the queue context by the scheduler on a successful
set, and reports 0 when the queue runs unlimited.
Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
---
doc/guides/nics/features/ice.ini | 1 +
doc/guides/rel_notes/release_26_11.rst | 3 +
drivers/net/intel/ice/ice_ethdev.c | 77 ++++++++++++++++++++++++++
3 files changed, 81 insertions(+)
diff --git a/doc/guides/nics/features/ice.ini b/doc/guides/nics/features/ice.ini
index 893d09e9ec..6b21d56144 100644
--- a/doc/guides/nics/features/ice.ini
+++ b/doc/guides/nics/features/ice.ini
@@ -30,6 +30,7 @@ RSS hash = Y
RSS key update = Y
RSS reta update = Y
VLAN filter = Y
+Rate limitation = Y
Traffic manager = Y
CRC offload = Y
VLAN offload = Y
diff --git a/doc/guides/rel_notes/release_26_11.rst b/doc/guides/rel_notes/release_26_11.rst
index 907f9013ff..1567f93563 100644
--- a/doc/guides/rel_notes/release_26_11.rst
+++ b/doc/guides/rel_notes/release_26_11.rst
@@ -64,6 +64,9 @@ New Features
* Renamed the ``enable_ptype_lldp`` devarg to ``enable_lldp``.
The old name is no longer accepted.
+* **Updated Intel ice driver.**
+
+ * Added support for Tx rate limiting per queue.
Removed Items
-------------
diff --git a/drivers/net/intel/ice/ice_ethdev.c b/drivers/net/intel/ice/ice_ethdev.c
index 76b8ff0a72..8ae2842098 100644
--- a/drivers/net/intel/ice/ice_ethdev.c
+++ b/drivers/net/intel/ice/ice_ethdev.c
@@ -212,6 +212,10 @@ static const uint32_t *ice_buffer_split_supported_hdr_ptypes_get(struct rte_eth_
size_t *no_of_elements);
static int ice_get_dcb_info(struct rte_eth_dev *dev, struct rte_eth_dcb_info *dcb_info);
static int ice_priority_flow_ctrl_set(struct rte_eth_dev *dev, struct rte_eth_pfc_conf *pfc_conf);
+static int ice_set_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t tx_rate);
+static int ice_get_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t *tx_rate);
static const struct rte_pci_id pci_id_ice_map[] = {
{ RTE_PCI_DEVICE(ICE_INTEL_VENDOR_ID, ICE_DEV_ID_E823L_BACKPLANE) },
@@ -353,6 +357,8 @@ static const struct eth_dev_ops ice_eth_dev_ops = {
.buffer_split_supported_hdr_ptypes_get = ice_buffer_split_supported_hdr_ptypes_get,
.get_dcb_info = ice_get_dcb_info,
.priority_flow_ctrl_set = ice_priority_flow_ctrl_set,
+ .set_queue_rate_limit = ice_set_queue_rate_limit,
+ .get_queue_rate_limit = ice_get_queue_rate_limit,
};
/* store statistics names and its offset in stats structure */
@@ -4205,6 +4211,77 @@ ice_priority_flow_ctrl_set(struct rte_eth_dev *dev, struct rte_eth_pfc_conf *pfc
return 0;
}
+static int
+ice_set_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t tx_rate)
+{
+ struct ice_pf *pf = ICE_DEV_PRIVATE_TO_PF(dev->data->dev_private);
+ struct ice_hw *hw = ICE_PF_TO_HW(pf);
+ struct ice_vsi *vsi = pf->main_vsi;
+ int ret;
+
+ if (queue_idx >= dev->data->nb_tx_queues) {
+ PMD_DRV_LOG(ERR, "Tx queue %u is out of range (%u configured)",
+ queue_idx, dev->data->nb_tx_queues);
+ return -EINVAL;
+ }
+
+ /* The scheduler node of a Tx queue only exists once the queue has been
+ * added to the Tx scheduler tree, which happens on queue start.
+ */
+ if (dev->data->tx_queue_state[queue_idx] != RTE_ETH_QUEUE_STATE_STARTED) {
+ PMD_DRV_LOG(ERR, "Tx queue %u must be started before setting its rate limit",
+ queue_idx);
+ return -EINVAL;
+ }
+
+ /* Rate is expressed in Mbps by the API, the scheduler uses Kbps. */
+ if (tx_rate > ICE_SCHED_MAX_BW / 1000) {
+ PMD_DRV_LOG(ERR, "Invalid Tx rate %u Mbps for queue %u, maximum is %u Mbps",
+ tx_rate, queue_idx, (uint32_t)(ICE_SCHED_MAX_BW / 1000));
+ return -EINVAL;
+ }
+
+ /* A rate of 0 removes the limit and restores the default bandwidth. */
+ if (tx_rate == 0)
+ ret = ice_cfg_q_bw_dflt_lmt(hw->port_info, vsi->idx, 0,
+ queue_idx, ICE_MAX_BW);
+ else
+ ret = ice_cfg_q_bw_lmt(hw->port_info, vsi->idx, 0, queue_idx,
+ ICE_MAX_BW, tx_rate * 1000);
+ if (ret) {
+ PMD_DRV_LOG(ERR, "Failed to set Tx rate limit on queue %u, error %d",
+ queue_idx, ret);
+ return -EIO;
+ }
+
+ return 0;
+}
+
+static int
+ice_get_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t *tx_rate)
+{
+ struct ice_pf *pf = ICE_DEV_PRIVATE_TO_PF(dev->data->dev_private);
+ struct ice_hw *hw = ICE_PF_TO_HW(pf);
+ struct ice_vsi *vsi = pf->main_vsi;
+ struct ice_q_ctx *q_ctx;
+
+ q_ctx = ice_get_lan_q_ctx(hw, vsi->idx, 0, queue_idx);
+ if (q_ctx == NULL) {
+ PMD_DRV_LOG(ERR, "Failed to get the context of Tx queue %u",
+ queue_idx);
+ return -EINVAL;
+ }
+
+ /* The scheduler caches the EIR limit in Kbps, and stores 0 when the
+ * queue runs at the default (unlimited) bandwidth.
+ */
+ *tx_rate = q_ctx->bw_t_info.eir_bw.bw / 1000;
+
+ return 0;
+}
+
static void
__vsi_queues_bind_intr(struct ice_vsi *vsi, uint16_t msix_vect,
int base_queue, int nb_queue)
--
2.34.1
^ permalink raw reply related [flat|nested] 7+ messages in thread
* Re: [PATCH] net/ice: add per-queue Tx rate limit support
2026-09-10 10:04 [PATCH] net/ice: add per-queue Tx rate limit support Anurag Mandal
@ 2026-09-10 10:29 ` Bruce Richardson
2026-09-15 10:19 ` Mandal, Anurag
2026-09-17 6:38 ` [PATCH v2] " Anurag Mandal
1 sibling, 1 reply; 7+ messages in thread
From: Bruce Richardson @ 2026-09-10 10:29 UTC (permalink / raw)
To: Anurag Mandal; +Cc: dev, anatoly.burakov
On Thu, Sep 10, 2026 at 10:04:32AM +0000, Anurag Mandal wrote:
> The Tx rate can be limited per queue with
> ethdev operation ``rte_eth_set_queue_rate_limit()``
> and can be read through ``rte_eth_get_queue_rate_limit()``.
>
> This feature uses the hardware packet pacing
> mechanism to enforce a data rate on individual
> Tx queues without tearing down the queue.
>
> The rate is specified in Mbps.
>
> ice_set_queue_rate_limit() applies the requested rate
> as the EIR (maximum bandwidth) limit of the queue
> scheduler node using ice_cfg_q_bw_lmt(),
> converting the Mbps value taken by the API to the Kbps
> expected by the scheduler.
> A rate of 0 removes the limit and restores the default
> bandwidth via ice_cfg_q_bw_dflt_lmt().
>
> ice_get_queue_rate_limit() reads back the value cached
> in the queue context by the scheduler on a successful
> set, and reports 0 when the queue runs unlimited.
>
> Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
> ---
> doc/guides/nics/features/ice.ini | 1 +
> doc/guides/rel_notes/release_26_11.rst | 3 +
> drivers/net/intel/ice/ice_ethdev.c | 77 ++++++++++++++++++++++++++
> 3 files changed, 81 insertions(+)
>
Is this functionality not overlapping with what the rte_rm APIs provide for
ice? Using the rte_rm hierarchies, it's possible to rate limit a queue, no?
/Bruce
^ permalink raw reply [flat|nested] 7+ messages in thread
* RE: [PATCH] net/ice: add per-queue Tx rate limit support
2026-09-10 10:29 ` Bruce Richardson
@ 2026-09-15 10:19 ` Mandal, Anurag
2026-09-15 17:13 ` Bruce Richardson
0 siblings, 1 reply; 7+ messages in thread
From: Mandal, Anurag @ 2026-09-15 10:19 UTC (permalink / raw)
To: Richardson, Bruce; +Cc: dev@dpdk.org, Burakov, Anatoly
> -----Original Message-----
> From: Richardson, Bruce <bruce.richardson@intel.com>
> Sent: 10 September 2026 16:00
> To: Mandal, Anurag <anurag.mandal@intel.com>
> Cc: dev@dpdk.org; Burakov, Anatoly <anatoly.burakov@intel.com>
> Subject: Re: [PATCH] net/ice: add per-queue Tx rate limit support
>
> On Thu, Sep 10, 2026 at 10:04:32AM +0000, Anurag Mandal wrote:
> > The Tx rate can be limited per queue with ethdev operation
> > ``rte_eth_set_queue_rate_limit()``
> > and can be read through ``rte_eth_get_queue_rate_limit()``.
> >
> > This feature uses the hardware packet pacing mechanism to enforce a
> > data rate on individual Tx queues without tearing down the queue.
> >
> > The rate is specified in Mbps.
> >
> > ice_set_queue_rate_limit() applies the requested rate as the EIR
> > (maximum bandwidth) limit of the queue scheduler node using
> > ice_cfg_q_bw_lmt(), converting the Mbps value taken by the API to the
> > Kbps expected by the scheduler.
> > A rate of 0 removes the limit and restores the default bandwidth via
> > ice_cfg_q_bw_dflt_lmt().
> >
> > ice_get_queue_rate_limit() reads back the value cached in the queue
> > context by the scheduler on a successful set, and reports 0 when the
> > queue runs unlimited.
> >
> > Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
> > ---
> > doc/guides/nics/features/ice.ini | 1 +
> > doc/guides/rel_notes/release_26_11.rst | 3 +
> > drivers/net/intel/ice/ice_ethdev.c | 77 ++++++++++++++++++++++++++
> > 3 files changed, 81 insertions(+)
> >
> Is this functionality not overlapping with what the rte_rm APIs provide for
> ice? Using the rte_rm hierarchies, it's possible to rate limit a queue, no?
>
> /Bruce
Hi Bruce,
I am guessing you meant rte_tm APIs instead of rte_rm.
Yes, the two paths ultimately program the same hardware field.
But, there are few reasons I still think the ethdev op is worth having:
1. rte_tm commit bounces the port. This does not.
So adjusting one queue's rate through rte_tm drops traffic on every queue and bounces the link.
2. VSI subtree is rebuilt for rte_tm :
a. Stop the port if running
b. Walk the VSI root up or down to the new layer, freeing sibling subtrees
c. free_sched_node_recursive() - tear down the existing scheduler subtree
d. create_sched_node_recursive() - rebuild it, ice_sched_add_elems() per node
e. Recompute pf->main_vsi->nb_qps, then ice_alloc_lan_q_ctx() to resize queue contexts
f. Restart the port
3. ixgbe & txgbe pmds also implement both.
Thanks,
Anurag M
^ permalink raw reply [flat|nested] 7+ messages in thread
* Re: [PATCH] net/ice: add per-queue Tx rate limit support
2026-09-15 10:19 ` Mandal, Anurag
@ 2026-09-15 17:13 ` Bruce Richardson
2026-09-17 6:37 ` Mandal, Anurag
0 siblings, 1 reply; 7+ messages in thread
From: Bruce Richardson @ 2026-09-15 17:13 UTC (permalink / raw)
To: Mandal, Anurag; +Cc: dev@dpdk.org, Burakov, Anatoly
On Tue, Sep 15, 2026 at 11:19:08AM +0100, Mandal, Anurag wrote:
> > -----Original Message-----
> > From: Richardson, Bruce <bruce.richardson@intel.com>
> > Sent: 10 September 2026 16:00
> > To: Mandal, Anurag <anurag.mandal@intel.com>
> > Cc: dev@dpdk.org; Burakov, Anatoly <anatoly.burakov@intel.com>
> > Subject: Re: [PATCH] net/ice: add per-queue Tx rate limit support
> >
> > On Thu, Sep 10, 2026 at 10:04:32AM +0000, Anurag Mandal wrote:
> > > The Tx rate can be limited per queue with ethdev operation
> > > ``rte_eth_set_queue_rate_limit()``
> > > and can be read through ``rte_eth_get_queue_rate_limit()``.
> > >
> > > This feature uses the hardware packet pacing mechanism to enforce a
> > > data rate on individual Tx queues without tearing down the queue.
> > >
> > > The rate is specified in Mbps.
> > >
> > > ice_set_queue_rate_limit() applies the requested rate as the EIR
> > > (maximum bandwidth) limit of the queue scheduler node using
> > > ice_cfg_q_bw_lmt(), converting the Mbps value taken by the API to the
> > > Kbps expected by the scheduler.
> > > A rate of 0 removes the limit and restores the default bandwidth via
> > > ice_cfg_q_bw_dflt_lmt().
> > >
> > > ice_get_queue_rate_limit() reads back the value cached in the queue
> > > context by the scheduler on a successful set, and reports 0 when the
> > > queue runs unlimited.
> > >
> > > Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
> > > ---
> > > doc/guides/nics/features/ice.ini | 1 +
> > > doc/guides/rel_notes/release_26_11.rst | 3 +
> > > drivers/net/intel/ice/ice_ethdev.c | 77 ++++++++++++++++++++++++++
> > > 3 files changed, 81 insertions(+)
> > >
> > Is this functionality not overlapping with what the rte_rm APIs provide for
> > ice? Using the rte_rm hierarchies, it's possible to rate limit a queue, no?
> >
> > /Bruce
>
> Hi Bruce,
>
> I am guessing you meant rte_tm APIs instead of rte_rm.
> Yes, the two paths ultimately program the same hardware field.
> But, there are few reasons I still think the ethdev op is worth having:
> 1. rte_tm commit bounces the port. This does not.
> So adjusting one queue's rate through rte_tm drops traffic on every queue and bounces the link.
> 2. VSI subtree is rebuilt for rte_tm :
> a. Stop the port if running
> b. Walk the VSI root up or down to the new layer, freeing sibling subtrees
> c. free_sched_node_recursive() - tear down the existing scheduler subtree
> d. create_sched_node_recursive() - rebuild it, ice_sched_add_elems() per node
> e. Recompute pf->main_vsi->nb_qps, then ice_alloc_lan_q_ctx() to resize queue contexts
> f. Restart the port
> 3. ixgbe & txgbe pmds also implement both.
>
Yes, your logic makes sense.
However, one final concern, it appears that this feature doesn't interact
in any way with the rte_tm one. Therefore, if a user configures a full
hierarchy using rte_tm, and then uses this new API to tweak the Tx rates on
queues, we could see problems later, e.g. losing all adjustments on apply
of a slightly different hierarchy etc.
If the two features don't interact well, we may need to put in place some
form of locking to ensure that you can't use one when you use the other.
What do you think?
/Bruce
^ permalink raw reply [flat|nested] 7+ messages in thread
* RE: [PATCH] net/ice: add per-queue Tx rate limit support
2026-09-15 17:13 ` Bruce Richardson
@ 2026-09-17 6:37 ` Mandal, Anurag
0 siblings, 0 replies; 7+ messages in thread
From: Mandal, Anurag @ 2026-09-17 6:37 UTC (permalink / raw)
To: Richardson, Bruce; +Cc: dev@dpdk.org, Burakov, Anatoly
> -----Original Message-----
> From: Richardson, Bruce <bruce.richardson@intel.com>
> Sent: 15 September 2026 22:43
> To: Mandal, Anurag <anurag.mandal@intel.com>
> Cc: dev@dpdk.org; Burakov, Anatoly <anatoly.burakov@intel.com>
> Subject: Re: [PATCH] net/ice: add per-queue Tx rate limit support
>
> On Tue, Sep 15, 2026 at 11:19:08AM +0100, Mandal, Anurag wrote:
> > > -----Original Message-----
> > > From: Richardson, Bruce <bruce.richardson@intel.com>
> > > Sent: 10 September 2026 16:00
> > > To: Mandal, Anurag <anurag.mandal@intel.com>
> > > Cc: dev@dpdk.org; Burakov, Anatoly <anatoly.burakov@intel.com>
> > > Subject: Re: [PATCH] net/ice: add per-queue Tx rate limit support
> > >
> > > On Thu, Sep 10, 2026 at 10:04:32AM +0000, Anurag Mandal wrote:
> > > > The Tx rate can be limited per queue with ethdev operation
> > > > ``rte_eth_set_queue_rate_limit()``
> > > > and can be read through ``rte_eth_get_queue_rate_limit()``.
> > > >
> > > > This feature uses the hardware packet pacing mechanism to enforce
> > > > a data rate on individual Tx queues without tearing down the queue.
> > > >
> > > > The rate is specified in Mbps.
> > > >
> > > > ice_set_queue_rate_limit() applies the requested rate as the EIR
> > > > (maximum bandwidth) limit of the queue scheduler node using
> > > > ice_cfg_q_bw_lmt(), converting the Mbps value taken by the API to
> > > > the Kbps expected by the scheduler.
> > > > A rate of 0 removes the limit and restores the default bandwidth
> > > > via ice_cfg_q_bw_dflt_lmt().
> > > >
> > > > ice_get_queue_rate_limit() reads back the value cached in the
> > > > queue context by the scheduler on a successful set, and reports 0
> > > > when the queue runs unlimited.
> > > >
> > > > Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
> > > > ---
> > > > doc/guides/nics/features/ice.ini | 1 +
> > > > doc/guides/rel_notes/release_26_11.rst | 3 +
> > > > drivers/net/intel/ice/ice_ethdev.c | 77
> ++++++++++++++++++++++++++
> > > > 3 files changed, 81 insertions(+)
> > > >
> > > Is this functionality not overlapping with what the rte_rm APIs
> > > provide for ice? Using the rte_rm hierarchies, it's possible to rate limit a
> queue, no?
> > >
> > > /Bruce
> >
> > Hi Bruce,
> >
> > I am guessing you meant rte_tm APIs instead of rte_rm.
> > Yes, the two paths ultimately program the same hardware field.
> > But, there are few reasons I still think the ethdev op is worth having:
> > 1. rte_tm commit bounces the port. This does not.
> > So adjusting one queue's rate through rte_tm drops traffic on every queue
> and bounces the link.
> > 2. VSI subtree is rebuilt for rte_tm :
> > a. Stop the port if running
> > b. Walk the VSI root up or down to the new layer, freeing sibling
> subtrees
> > c. free_sched_node_recursive() - tear down the existing scheduler
> subtree
> > d. create_sched_node_recursive() - rebuild it, ice_sched_add_elems()
> per node
> > e. Recompute pf->main_vsi->nb_qps, then ice_alloc_lan_q_ctx() to
> resize queue contexts
> > f. Restart the port
> > 3. ixgbe & txgbe pmds also implement both.
> >
> Yes, your logic makes sense.
>
> However, one final concern, it appears that this feature doesn't interact in any
> way with the rte_tm one. Therefore, if a user configures a full hierarchy using
> rte_tm, and then uses this new API to tweak the Tx rates on queues, we could
> see problems later, e.g. losing all adjustments on apply of a slightly different
> hierarchy etc.
>
> If the two features don't interact well, we may need to put in place some form
> of locking to ensure that you can't use one when you use the other.
> What do you think?
>
> /Bruce
Hi Bruce,
Thank you for the review as well as feedback.
I have addressed them in v2.
Kindly check.
Thank you.
Regards,
Anurag M
^ permalink raw reply [flat|nested] 7+ messages in thread
* [PATCH v2] net/ice: add per-queue Tx rate limit support
2026-09-10 10:04 [PATCH] net/ice: add per-queue Tx rate limit support Anurag Mandal
2026-09-10 10:29 ` Bruce Richardson
@ 2026-09-17 6:38 ` Anurag Mandal
2026-09-18 10:14 ` Bruce Richardson
1 sibling, 1 reply; 7+ messages in thread
From: Anurag Mandal @ 2026-09-17 6:38 UTC (permalink / raw)
To: dev; +Cc: bruce.richardson, anatoly.burakov, Anurag Mandal
The Tx rate can be limited per queue with
ethdev operation ``rte_eth_set_queue_rate_limit()``
and can be read through ``rte_eth_get_queue_rate_limit()``.
This feature uses the hardware packet pacing
mechanism to enforce a data rate on individual
Tx queues without tearing down the queue or
bouncing the port.
The rate is specified in Mbps.
ice_set_queue_rate_limit() applies the requested rate
as the EIR (maximum bandwidth) limit of the queue
scheduler node using ice_cfg_q_bw_lmt(),
converting the Mbps value taken by the API to the Kbps
expected by the scheduler.
A rate of 0 removes the limit and restores the default
bandwidth via ice_cfg_q_bw_dflt_lmt().
ice_get_queue_rate_limit() reads back the value cached
in the queue context by the scheduler on a successful
set, and reports 0 when the queue runs unlimited.
This interface and the Traffic Management API are
mutually exclusive.
Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
---
V2: Addressed Bruce Richardson's feedback
- The per-queue Tx rate limit and Traffic Management APIs are mutually exclusive.
doc/guides/nics/features/ice.ini | 1 +
doc/guides/nics/ice.rst | 26 +++++++
doc/guides/rel_notes/release_26_11.rst | 2 +
drivers/net/intel/ice/ice_ethdev.c | 100 +++++++++++++++++++++++++
drivers/net/intel/ice/ice_ethdev.h | 2 +
drivers/net/intel/ice/ice_tm.c | 14 ++++
6 files changed, 145 insertions(+)
diff --git a/doc/guides/nics/features/ice.ini b/doc/guides/nics/features/ice.ini
index 589916b2c1..2bf1b2a42d 100644
--- a/doc/guides/nics/features/ice.ini
+++ b/doc/guides/nics/features/ice.ini
@@ -30,6 +30,7 @@ RSS hash = Y
RSS key update = Y
RSS reta update = Y
VLAN filter = Y
+Rate limitation = Y
Traffic manager = Y
CRC offload = Y
VLAN offload = Y
diff --git a/doc/guides/nics/ice.rst b/doc/guides/nics/ice.rst
index bdb3d61c23..bfb67a2cd3 100644
--- a/doc/guides/nics/ice.rst
+++ b/doc/guides/nics/ice.rst
@@ -925,6 +925,32 @@ Additional Options
192.168.0.2', dst="192.168.0.3")/TCP(flags='S')/Raw(load='XXXXXXXXXX'), \
iface="enp24s0f0", count=10)
+Per-Queue Tx Rate Limiting
+~~~~~~~~~~~~~~~~~~~~~~~~~~
+
+The maximum Tx rate of an individual queue can be capped using
+``rte_eth_set_queue_rate_limit()``, and read back using
+``rte_eth_get_queue_rate_limit()``.
+The rate is given in Mbps, and a rate of 0 removes the limit.
+
+The limit is applied to the queue's node in the Tx scheduler tree,
+which only exists while the queue is running,
+so the queue must be started before its rate can be set,
+and a configured rate is lost when the queue is stopped.
+
+This interface and the Traffic Management API are mutually exclusive,
+because both configure the bandwidth of the same scheduler nodes:
+
+* setting a queue rate limit fails while a Traffic Management
+ hierarchy is committed;
+
+* committing a Traffic Management hierarchy fails while any queue has
+ a rate limit set through ``rte_eth_set_queue_rate_limit()``.
+
+To switch from one to the other, clear the existing configuration first,
+either by setting the rate of each limited queue back to 0,
+or by deleting the Traffic Management hierarchy.
+
Sample Application Notes
------------------------
diff --git a/doc/guides/rel_notes/release_26_11.rst b/doc/guides/rel_notes/release_26_11.rst
index e725933fa7..19365fd0f2 100644
--- a/doc/guides/rel_notes/release_26_11.rst
+++ b/doc/guides/rel_notes/release_26_11.rst
@@ -75,6 +75,8 @@ New Features
paths, enabling QinQ tag insertion and outer IPv4/UDP checksum
offloads on those paths.
+ * Added support for Tx rate limiting per queue.
+
Removed Items
-------------
diff --git a/drivers/net/intel/ice/ice_ethdev.c b/drivers/net/intel/ice/ice_ethdev.c
index 22578c9f28..dfee07c457 100644
--- a/drivers/net/intel/ice/ice_ethdev.c
+++ b/drivers/net/intel/ice/ice_ethdev.c
@@ -212,6 +212,10 @@ static const uint32_t *ice_buffer_split_supported_hdr_ptypes_get(struct rte_eth_
size_t *no_of_elements);
static int ice_get_dcb_info(struct rte_eth_dev *dev, struct rte_eth_dcb_info *dcb_info);
static int ice_priority_flow_ctrl_set(struct rte_eth_dev *dev, struct rte_eth_pfc_conf *pfc_conf);
+static int ice_set_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t tx_rate);
+static int ice_get_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t *tx_rate);
static const struct rte_pci_id pci_id_ice_map[] = {
{ RTE_PCI_DEVICE(ICE_INTEL_VENDOR_ID, ICE_DEV_ID_E823L_BACKPLANE) },
@@ -353,6 +357,8 @@ static const struct eth_dev_ops ice_eth_dev_ops = {
.buffer_split_supported_hdr_ptypes_get = ice_buffer_split_supported_hdr_ptypes_get,
.get_dcb_info = ice_get_dcb_info,
.priority_flow_ctrl_set = ice_priority_flow_ctrl_set,
+ .set_queue_rate_limit = ice_set_queue_rate_limit,
+ .get_queue_rate_limit = ice_get_queue_rate_limit,
};
/* store statistics names and its offset in stats structure */
@@ -4207,6 +4213,100 @@ ice_priority_flow_ctrl_set(struct rte_eth_dev *dev, struct rte_eth_pfc_conf *pfc
return 0;
}
+static int
+ice_set_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t tx_rate)
+{
+ struct ice_pf *pf = ICE_DEV_PRIVATE_TO_PF(dev->data->dev_private);
+ struct ice_hw *hw = ICE_PF_TO_HW(pf);
+ struct ice_vsi *vsi = pf->main_vsi;
+ int ret;
+
+ if (queue_idx >= dev->data->nb_tx_queues) {
+ PMD_DRV_LOG(ERR, "Tx queue %u is out of range (%u configured)",
+ queue_idx, dev->data->nb_tx_queues);
+ return -EINVAL;
+ }
+
+ /*
+ * A committed TM hierarchy owns the bandwidth of every scheduler node
+ * and reapplies it on each commit, so the two interfaces are exclusive.
+ */
+ if (pf->tm_conf.committed) {
+ PMD_DRV_LOG(ERR, "Tx rate limit cannot be set while a traffic manager hierarchy is committed");
+ return -EBUSY;
+ }
+
+ /*
+ * The scheduler node of a Tx queue only exists once the queue has been
+ * added to the Tx scheduler tree, which happens on queue start.
+ */
+ if (dev->data->tx_queue_state[queue_idx] != RTE_ETH_QUEUE_STATE_STARTED) {
+ PMD_DRV_LOG(ERR, "Tx queue %u must be started before setting its rate limit",
+ queue_idx);
+ return -EINVAL;
+ }
+
+ /* Rate is expressed in Mbps by the API, the scheduler uses Kbps. */
+ if (tx_rate > ICE_SCHED_MAX_BW / 1000) {
+ PMD_DRV_LOG(ERR, "Invalid Tx rate %u Mbps for queue %u, maximum is %u Mbps",
+ tx_rate, queue_idx, (uint32_t)(ICE_SCHED_MAX_BW / 1000));
+ return -EINVAL;
+ }
+
+ /* A rate of 0 removes the limit and restores the default bandwidth. */
+ if (tx_rate == 0)
+ ret = ice_cfg_q_bw_dflt_lmt(hw->port_info, vsi->idx, 0,
+ queue_idx, ICE_MAX_BW);
+ else
+ ret = ice_cfg_q_bw_lmt(hw->port_info, vsi->idx, 0, queue_idx,
+ ICE_MAX_BW, tx_rate * 1000);
+ if (ret) {
+ PMD_DRV_LOG(ERR, "Failed to set Tx rate limit on queue %u, error %d",
+ queue_idx, ret);
+ return -EIO;
+ }
+
+ return 0;
+}
+
+/*
+ * Returns the rate limit currently programmed on a Tx queue, 0 if unlimited.
+ * The scheduler caches the requested rate, but that cache outlives the queue
+ * node, which is destroyed on queue stop and recreated with the default
+ * profile, so only trust it while the node still carries a rate limit.
+ */
+uint32_t
+ice_txq_rate_limit_kbps(struct ice_pf *pf, uint16_t queue_idx)
+{
+ struct ice_hw *hw = ICE_PF_TO_HW(pf);
+ struct ice_sched_node *node;
+ struct ice_q_ctx *q_ctx;
+
+ q_ctx = ice_get_lan_q_ctx(hw, pf->main_vsi->idx, 0, queue_idx);
+ if (q_ctx == NULL)
+ return 0;
+
+ node = ice_sched_find_node_by_teid(hw->port_info->root, q_ctx->q_teid);
+ if (node == NULL ||
+ rte_le_to_cpu_16(node->info.data.eir_bw.bw_profile_idx) ==
+ ICE_SCHED_DFLT_RL_PROF_ID)
+ return 0;
+
+ return q_ctx->bw_t_info.eir_bw.bw;
+}
+
+static int
+ice_get_queue_rate_limit(struct rte_eth_dev *dev, uint16_t queue_idx,
+ uint32_t *tx_rate)
+{
+ struct ice_pf *pf = ICE_DEV_PRIVATE_TO_PF(dev->data->dev_private);
+
+ *tx_rate = ice_txq_rate_limit_kbps(pf, queue_idx) / 1000;
+
+ return 0;
+}
+
static void
__vsi_queues_bind_intr(struct ice_vsi *vsi, uint16_t msix_vect,
int base_queue, int nb_queue)
diff --git a/drivers/net/intel/ice/ice_ethdev.h b/drivers/net/intel/ice/ice_ethdev.h
index 3cfd7afbae..e39757e842 100644
--- a/drivers/net/intel/ice/ice_ethdev.h
+++ b/drivers/net/intel/ice/ice_ethdev.h
@@ -832,4 +832,6 @@ int rte_pmd_ice_dump_txsched(uint16_t port, bool detail, FILE *stream);
int
ice_tm_setup_txq_node(struct ice_pf *pf, struct ice_hw *hw, uint16_t qid, uint32_t node_teid);
+uint32_t ice_txq_rate_limit_kbps(struct ice_pf *pf, uint16_t queue_idx);
+
#endif /* _ICE_ETHDEV_H_ */
diff --git a/drivers/net/intel/ice/ice_tm.c b/drivers/net/intel/ice/ice_tm.c
index d93704dd3f..9566a36098 100644
--- a/drivers/net/intel/ice/ice_tm.c
+++ b/drivers/net/intel/ice/ice_tm.c
@@ -903,8 +903,22 @@ ice_hierarchy_commit(struct rte_eth_dev *dev,
int clear_on_fail,
struct rte_tm_error *error)
{
+ struct ice_pf *pf = ICE_DEV_PRIVATE_TO_PF(dev->data->dev_private);
bool restart = false;
+ /*
+ * A commit reapplies the bandwidth of every node, which would silently
+ * discard any rate set through rte_eth_set_queue_rate_limit().
+ */
+ for (uint16_t i = 0; i < dev->data->nb_tx_queues; i++) {
+ if (ice_txq_rate_limit_kbps(pf, i) != 0) {
+ error->type = RTE_TM_ERROR_TYPE_UNSPECIFIED;
+ error->message =
+ "queue rate limit already set via rte_eth_set_queue_rate_limit";
+ return -EBUSY;
+ }
+ }
+
/* commit should only be done to topology before start
* If port is already started, stop it and then restart when done.
*/
--
2.34.1
^ permalink raw reply related [flat|nested] 7+ messages in thread
* Re: [PATCH v2] net/ice: add per-queue Tx rate limit support
2026-09-17 6:38 ` [PATCH v2] " Anurag Mandal
@ 2026-09-18 10:14 ` Bruce Richardson
0 siblings, 0 replies; 7+ messages in thread
From: Bruce Richardson @ 2026-09-18 10:14 UTC (permalink / raw)
To: Anurag Mandal; +Cc: dev, anatoly.burakov
On Thu, Sep 17, 2026 at 06:38:23AM +0000, Anurag Mandal wrote:
> The Tx rate can be limited per queue with
> ethdev operation ``rte_eth_set_queue_rate_limit()``
> and can be read through ``rte_eth_get_queue_rate_limit()``.
>
> This feature uses the hardware packet pacing
> mechanism to enforce a data rate on individual
> Tx queues without tearing down the queue or
> bouncing the port.
>
> The rate is specified in Mbps.
>
> ice_set_queue_rate_limit() applies the requested rate
> as the EIR (maximum bandwidth) limit of the queue
> scheduler node using ice_cfg_q_bw_lmt(),
> converting the Mbps value taken by the API to the Kbps
> expected by the scheduler.
> A rate of 0 removes the limit and restores the default
> bandwidth via ice_cfg_q_bw_dflt_lmt().
>
> ice_get_queue_rate_limit() reads back the value cached
> in the queue context by the scheduler on a successful
> set, and reports 0 when the queue runs unlimited.
>
> This interface and the Traffic Management API are
> mutually exclusive.
>
> Signed-off-by: Anurag Mandal <anurag.mandal@intel.com>
> ---
> V2: Addressed Bruce Richardson's feedback
> - The per-queue Tx rate limit and Traffic Management APIs are mutually exclusive.
>
Acked-by: Bruce Richardson <bruce.richardson@intel.com>
Patch applied to dpdk-next-net-intel.
Thanks,
/Bruce
^ permalink raw reply [flat|nested] 7+ messages in thread
end of thread, other threads:[~2026-09-18 10:14 UTC | newest]
Thread overview: 7+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-10 10:04 [PATCH] net/ice: add per-queue Tx rate limit support Anurag Mandal
2026-09-10 10:29 ` Bruce Richardson
2026-09-15 10:19 ` Mandal, Anurag
2026-09-15 17:13 ` Bruce Richardson
2026-09-17 6:37 ` Mandal, Anurag
2026-09-17 6:38 ` [PATCH v2] " Anurag Mandal
2026-09-18 10:14 ` Bruce Richardson
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox