Linux-HyperV List
 help / color / mirror / Atom feed
* [PATCH net-next] net: mana: Add support to process throttled EQEs
@ 2026-10-08  8:33 Sahil Chandna
  2026-10-08  8:39 ` netdev-bot+sinfo
  2026-10-09  8:33 ` sashiko-bot
  0 siblings, 2 replies; 4+ messages in thread
From: Sahil Chandna @ 2026-10-08  8:33 UTC (permalink / raw)
  To: haiyangz, wei.liu, decui, andrew+netdev, davem, edumazet, kuba,
	pabeni, kotaranov, horms, ernis, gargaditya, mawasthi, leitao,
	linux-hyperv, netdev, linux-kernel, linux-rdma

When an event queue nears full, the hardware coalesces the per-CQ
GDMA_EQE_COMPLETION notifications into a single throttle EQE of type 4
that carries no cq_id. The driver does not handle this event type, so
it falls through to the default case and is dropped. Any CQ whose
completion notification was replaced by the throttle EQE is then never
scheduled again and its queue stalls.
Add support to process the CQs which belong to a throttled EQE of type 4.

Signed-off-by: Sahil Chandna <sahilchandna@linux.microsoft.com>
---
 .../net/ethernet/microsoft/mana/gdma_main.c   | 43 ++++++++++++++++++-
 .../ethernet/microsoft/mana/mana_ethtool.c    |  7 +++
 include/net/mana/gdma.h                       | 12 +++++-
 include/net/mana/mana.h                       |  1 +
 4 files changed, 61 insertions(+), 2 deletions(-)

diff --git a/drivers/net/ethernet/microsoft/mana/gdma_main.c b/drivers/net/ethernet/microsoft/mana/gdma_main.c
index ae0ed700b3b9..b8e5ad2c6d31 100644
--- a/drivers/net/ethernet/microsoft/mana/gdma_main.c
+++ b/drivers/net/ethernet/microsoft/mana/gdma_main.c
@@ -747,6 +747,18 @@ void mana_gd_free_service_wq(struct gdma_context *gc)
 	gc->service_wq = NULL;
 }

+static void mana_gd_schedule_eq_cqs(struct gdma_queue *eq)
+{
+	struct gdma_queue *cq;
+	u8 i;
+
+	for (i = 0; i < GDMA_EQ_MAX_CHILD_CQ; i++) {
+		cq = rcu_dereference(eq->eq.child_cq[i]);
+		if (cq && cq->cq.callback)
+			cq->cq.callback(cq->cq.context, cq);
+	}
+}
+
 static void mana_gd_process_eqe(struct gdma_queue *eq)
 {
 	u32 head = eq->head % (eq->queue_size / GDMA_EQE_SIZE);
@@ -777,6 +789,11 @@ static void mana_gd_process_eqe(struct gdma_queue *eq)

 		break;

+	case GDMA_EQE_THROTTLE:
+		eq->eq.throttle_count++;
+		mana_gd_schedule_eq_cqs(eq);
+		break;
+
 	case GDMA_EQE_TEST_EVENT:
 		gc->test_event_eq_id = eq->id;
 		complete(&gc->eq_test_event);
@@ -1063,17 +1080,41 @@ static void mana_gd_create_cq(const struct gdma_queue_spec *spec,
 			      struct gdma_queue *queue)
 {
 	u32 log2_num_entries = ilog2(spec->queue_size / GDMA_CQE_SIZE);
+	struct gdma_queue *parent;
+	u8 i;

 	queue->head |= INITIALIZED_OWNER_BIT(log2_num_entries);
-	queue->cq.parent = spec->cq.parent_eq;
+	parent = spec->cq.parent_eq;
+	queue->cq.parent = parent;
 	queue->cq.context = spec->cq.context;
 	queue->cq.callback = spec->cq.callback;
+
+	if (!parent)
+		return;
+
+	/* For throttled EQE store the child CQ */
+	for (i = 0; i < GDMA_EQ_MAX_CHILD_CQ; i++)
+		if (!rcu_access_pointer(parent->eq.child_cq[i])) {
+			rcu_assign_pointer(parent->eq.child_cq[i], queue);
+			return;
+		}
 }

 static void mana_gd_destroy_cq(struct gdma_context *gc,
 			       struct gdma_queue *queue)
 {
+	struct gdma_queue *parent = queue->cq.parent;
 	u32 id = queue->id;
+	u8 i;
+
+	if (parent) {
+		for (i = 0; i < GDMA_EQ_MAX_CHILD_CQ; i++) {
+			if (rcu_access_pointer(parent->eq.child_cq[i]) == queue)
+				RCU_INIT_POINTER(parent->eq.child_cq[i], NULL);
+		}
+
+		synchronize_rcu();
+	}

 	if (id >= gc->max_num_cqs)
 		return;
diff --git a/drivers/net/ethernet/microsoft/mana/mana_ethtool.c b/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
index ece7ff9cc409..1004c3334532 100644
--- a/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
+++ b/drivers/net/ethernet/microsoft/mana/mana_ethtool.c
@@ -22,6 +22,8 @@ static const struct mana_stats_desc mana_eth_stats[] = {
 				       tx_linear_pkt_cnt)},
 	{"rx_cqe_unknown_type", offsetof(struct mana_ethtool_stats,
 					rx_cqe_unknown_type)},
+	{"eq_throttle_events", offsetof(struct mana_ethtool_stats,
+					eq_throttle_events)},
 };

 static const struct mana_stats_desc mana_hc_stats[] = {
@@ -261,6 +263,11 @@ static void mana_get_ethtool_stats(struct net_device *ndev,
 	 */
 	mana_query_phy_stats(apc);

+	apc->eth_stats.eq_throttle_events = 0;
+	for (q = 0; q < num_queues; q++)
+		apc->eth_stats.eq_throttle_events +=
+			apc->eqs[q].eq->eq.throttle_count;
+
 	for (q = 0; q < ARRAY_SIZE(mana_eth_stats); q++)
 		data[i++] = *(u64 *)(eth_stats + mana_eth_stats[q].offset);

diff --git a/include/net/mana/gdma.h b/include/net/mana/gdma.h
index c610fc1067e0..03fe9395604d 100644
--- a/include/net/mana/gdma.h
+++ b/include/net/mana/gdma.h
@@ -58,6 +58,7 @@ enum gdma_work_request_flags {

 enum gdma_eqe_type {
 	GDMA_EQE_COMPLETION		= 3,
+	GDMA_EQE_THROTTLE		= 4,
 	GDMA_EQE_TEST_EVENT		= 64,
 	GDMA_EQE_HWC_INIT_EQ_ID_DB	= 129,
 	GDMA_EQE_HWC_INIT_DATA		= 130,
@@ -286,6 +287,7 @@ struct gdma_dev {
 #define GDMA_EQE_SIZE 16
 #define GDMA_MAX_SQE_SIZE 512
 #define GDMA_MAX_RQE_SIZE 256
+#define GDMA_EQ_MAX_CHILD_CQ  2

 #define GDMA_COMP_DATA_SIZE 0x3C

@@ -366,6 +368,10 @@ struct gdma_queue {
 			unsigned int irq;

 			u32 log2_throttle_limit;
+
+			u64 throttle_count;
+
+			struct gdma_queue __rcu *child_cq[GDMA_EQ_MAX_CHILD_CQ];
 		} eq;

 		struct {
@@ -752,6 +758,9 @@ enum {
 /* Driver supports non-contiguous queue buffers */
 #define GDMA_DRV_CAP_FLAG_1_NON_CONTIGUOUS_BUFFERS BIT(30)

+/* Driver supports handling throttled EQEs */
+#define GDMA_DRV_CAP_FLAG_1_THROTTLED_EVENT_QUEUE BIT_ULL(32)
+
 /* Capabilities in the PCI-only group below rely on dynamic MSI-X allocation
  * and on the servicing and reset paths reached through
  * mana_schedule_serv_work(). Transports that provide neither leave
@@ -776,7 +785,8 @@ enum {
 	 GDMA_DRV_CAP_FLAG_1_HANDLE_STALL_SQ_RECOVERY | \
 	 GDMA_DRV_CAP_FLAG_1_EQ_MSI_UNSHARE_MULTI_VPORT | \
 	 GDMA_DRV_CAP_FLAG_1_DYN_INTERRUPT_MODERATION | \
-	 GDMA_DRV_CAP_FLAG_1_NON_CONTIGUOUS_BUFFERS)
+	 GDMA_DRV_CAP_FLAG_1_NON_CONTIGUOUS_BUFFERS | \
+	 GDMA_DRV_CAP_FLAG_1_THROTTLED_EVENT_QUEUE)

 #define GDMA_DRV_CAP_FLAGS2 0

diff --git a/include/net/mana/mana.h b/include/net/mana/mana.h
index 83b7eff4646e..46aa843737b8 100644
--- a/include/net/mana/mana.h
+++ b/include/net/mana/mana.h
@@ -439,6 +439,7 @@ struct mana_ethtool_stats {
 	u64 tx_cqe_unknown_type;
 	u64 tx_linear_pkt_cnt;
 	u64 rx_cqe_unknown_type;
+	u64 eq_throttle_events;
 };

 struct mana_ethtool_hc_stats {
--
2.53.0


^ permalink raw reply related	[flat|nested] 4+ messages in thread

* Re: [PATCH net-next] net: mana: Add support to process throttled EQEs
  2026-10-08  8:33 [PATCH net-next] net: mana: Add support to process throttled EQEs Sahil Chandna
@ 2026-10-08  8:39 ` netdev-bot+sinfo
  2026-10-09 14:02   ` Sahil Chandna
  2026-10-09  8:33 ` sashiko-bot
  1 sibling, 1 reply; 4+ messages in thread
From: netdev-bot+sinfo @ 2026-10-08  8:39 UTC (permalink / raw)
  To: Sahil Chandna
  Cc: haiyangz, wei.liu, decui, andrew+netdev, davem, edumazet, kuba,
	pabeni, kotaranov, horms, ernis, gargaditya, mawasthi, leitao,
	linux-hyperv, netdev, linux-kernel, linux-rdma

Hi!

This is an automated message. This series looks like a fix, but its
commit messages seem to be missing some information:

 - How the issue was discovered, e.g. hit in production, hit during
   development, syzbot report, manual code inspection, LLM or static
   analysis tool scan.

 - Whether the issue was actually triggered, or is only theoretical
   (e.g. found by code inspection). If it was triggered please include
   the symptoms, like the stack trace or error messages.

 - What hardware the change was tested on. For driver fixes please
   mention the device (and if relevant firmware version) used for
   testing, or say that the change was not tested on real hardware.

Please do not repost the series just to address the above. Instead,
reply to this email with the missing information, so that reviewers
can take it into account. If the series needs another revision for
other reasons, please include the information in the commit messages
then.

The evaluation is done by an LLM so it may be wrong, if you think
that is the case please reply and explain.

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: [PATCH net-next] net: mana: Add support to process throttled EQEs
  2026-10-08  8:33 [PATCH net-next] net: mana: Add support to process throttled EQEs Sahil Chandna
  2026-10-08  8:39 ` netdev-bot+sinfo
@ 2026-10-09  8:33 ` sashiko-bot
  1 sibling, 0 replies; 4+ messages in thread
From: sashiko-bot @ 2026-10-09  8:33 UTC (permalink / raw)
  To: Sahil Chandna; +Cc: linux-rdma, linux-hyperv

Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] mana: RDMA queue stalls due to dropped throttle events
--

commit 49b4827fa426b6ae0b8759a5984d01f44b8db5e1
Author: Sahil Chandna <sahilchandna@linux.microsoft.com>

net: mana: Add support to process throttled EQEs

This commit adds support for processing throttled EQEs of type 4 when an
event queue nears full capacity. The driver advertises the
GDMA_DRV_CAP_FLAG_1_THROTTLED_EVENT_QUEUE capability and attempts to schedule
the CQs associated with a throttled EQE. CQs are tracked in a new child_cq
array initialized when creating a CQ.

> diff --git a/drivers/net/ethernet/microsoft/mana/gdma_main.c b/drivers/net/ethernet/microsoft/mana/gdma_main.c
> index ae0ed700b3b93..b8e5ad2c6d312 100644
> --- a/drivers/net/ethernet/microsoft/mana/gdma_main.c
> +++ b/drivers/net/ethernet/microsoft/mana/gdma_main.c
> @@ -747,6 +747,18 @@ void mana_gd_free_service_wq(struct gdma_context *gc)
>  	gc->service_wq = NULL;
>  }
>  
> +static void mana_gd_schedule_eq_cqs(struct gdma_queue *eq)
> +{
> +	struct gdma_queue *cq;
> +	u8 i;
> +
> +	for (i = 0; i < GDMA_EQ_MAX_CHILD_CQ; i++) {
> +		cq = rcu_dereference(eq->eq.child_cq[i]);
> +		if (cq && cq->cq.callback)
> +			cq->cq.callback(cq->cq.context, cq);

[Severity: High]
When a GDMA_EQE_THROTTLE event is processed, this function attempts to notify
the CQs. Will RDMA CQs receive these notifications?

RDMA CQs do not appear to be added to child_cq, which might cause their
notifications to be entirely dropped and result in RDMA queue stalls.

> +	}
> +}
> +

[ ... ]

> @@ -1063,17 +1080,41 @@ static void mana_gd_create_cq(const struct gdma_queue_spec *spec,
>  			      struct gdma_queue *queue)
>  {
>  	u32 log2_num_entries = ilog2(spec->queue_size / GDMA_CQE_SIZE);
> +	struct gdma_queue *parent;
> +	u8 i;
>  
>  	queue->head |= INITIALIZED_OWNER_BIT(log2_num_entries);
> -	queue->cq.parent = spec->cq.parent_eq;
> +	parent = spec->cq.parent_eq;
> +	queue->cq.parent = parent;
>  	queue->cq.context = spec->cq.context;
>  	queue->cq.callback = spec->cq.callback;
> +
> +	if (!parent)
> +		return;
> +
> +	/* For throttled EQE store the child CQ */
> +	for (i = 0; i < GDMA_EQ_MAX_CHILD_CQ; i++)
> +		if (!rcu_access_pointer(parent->eq.child_cq[i])) {
> +			rcu_assign_pointer(parent->eq.child_cq[i], queue);
> +			return;
> +		}

[Severity: High]
Does this logic handle RDMA CQs correctly?

RDMA CQs are created with a NULL parent_eq (via mana_ib_create_kernel_queue),
which causes this function to return early without populating the child_cq
array.

Additionally, if an EQ has more than GDMA_EQ_MAX_CHILD_CQ (which is 2) CQs
attached, which is common in RDMA workloads, this loop will silently return
without adding the additional CQs. Can this lead to lost completions?

>  }

[ ... ]

> diff --git a/include/net/mana/gdma.h b/include/net/mana/gdma.h
> index c610fc1067e0d..03fe9395604d6 100644
> --- a/include/net/mana/gdma.h
> +++ b/include/net/mana/gdma.h
> @@ -776,7 +785,8 @@ enum {
>  	 GDMA_DRV_CAP_FLAG_1_HANDLE_STALL_SQ_RECOVERY | \
>  	 GDMA_DRV_CAP_FLAG_1_EQ_MSI_UNSHARE_MULTI_VPORT | \
>  	 GDMA_DRV_CAP_FLAG_1_DYN_INTERRUPT_MODERATION | \
> -	 GDMA_DRV_CAP_FLAG_1_NON_CONTIGUOUS_BUFFERS)
> +	 GDMA_DRV_CAP_FLAG_1_NON_CONTIGUOUS_BUFFERS | \
> +	 GDMA_DRV_CAP_FLAG_1_THROTTLED_EVENT_QUEUE)

[Severity: High]
Since this patch globally advertises throttling support, the hardware might
send throttle events on RDMA EQs when they are heavily loaded.

Given that RDMA EQs are initialized with a log2_throttle_limit in
mana_ib_create_eqs(), will enabling this capability cause RDMA workloads to
stall if their throttle events are dropped due to the missing child_cq mapping
in mana_gd_create_cq()?

-- 
Sashiko AI review · https://sashiko.dev/#/patchset/20261008083315.306076-1-sahilchandna@linux.microsoft.com?part=1

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: [PATCH net-next] net: mana: Add support to process throttled EQEs
  2026-10-08  8:39 ` netdev-bot+sinfo
@ 2026-10-09 14:02   ` Sahil Chandna
  0 siblings, 0 replies; 4+ messages in thread
From: Sahil Chandna @ 2026-10-09 14:02 UTC (permalink / raw)
  To: netdev-bot+sinfo
  Cc: haiyangz, wei.liu, decui, andrew+netdev, davem, edumazet, kuba,
	pabeni, kotaranov, horms, ernis, gargaditya, mawasthi, leitao,
	linux-hyperv, netdev, linux-kernel, linux-rdma

On 08-10-2026 14:09, netdev-bot+sinfo@kernel.org wrote:
> Hi!
> 
> This is an automated message. This series looks like a fix, but its
> commit messages seem to be missing some information:
> 
>  - How the issue was discovered, e.g. hit in production, hit during
>    development, syzbot report, manual code inspection, LLM or static
>    analysis tool scan.
> 
This issue was discovered internally>  - Whether the issue was actually
triggered, or is only theoretical
>    (e.g. found by code inspection). If it was triggered please include
>    the symptoms, like the stack trace or error messages.
> 
Logs:
2026-10-03T05:29:01 hv_netvsc eth0: VF slot 0 removed
2026-10-03T05:29:23 mana 7870:00:00.0 ens1: NETDEV WATCHDOG: CPU: 45:
                    transmit queue 1 timed out 15133 ms
2026-10-03T05:31:05 mana 7870:00:00.0: HWC: Request timed out: 60000 ms
2026-10-03T05:31:05 mana 7870:00:00.0: Gf stats wk handler: gf stats
query timed out.
2026-10-03T05:31:21 ens1: NETDEV WATCHDOG: CPU: 20: transmit queue 0
timed out 15859 ms
2026-10-03T05:31:37 ens1: NETDEV WATCHDOG: CPU: 15: transmit queue 0
timed out 31731 ms
2026-10-03T05:31:53 ens1: NETDEV WATCHDOG: CPU: 21: transmit queue 0
timed out 47603 ms
2026-10-03T05:32:08 ens1: NETDEV WATCHDOG: CPU: 42: transmit queue 0
timed out 62451 ms
>  - What hardware the change was tested on. For driver fixes please
>    mention the device (and if relevant firmware version) used for
>    testing, or say that the change was not tested on real hardware.
> 
This was tested on azure arm64 VMs while running Virtual Client
(https://github.com/microsoft/VirtualClient) with busy poll enabled.>
Please do not repost the series just to address the above. Instead,
> reply to this email with the missing information, so that reviewers
> can take it into account. If the series needs another revision for
> other reasons, please include the information in the commit messages
> then.
> 
> The evaluation is done by an LLM so it may be wrong, if you think
> that is the case please reply and explain.
This is a feature being enabled to handle a new EQE type 4 hence chose
to send it to net-next

^ permalink raw reply	[flat|nested] 4+ messages in thread

end of thread, other threads:[~2026-10-09 14:02 UTC | newest]

Thread overview: 4+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-10-08  8:33 [PATCH net-next] net: mana: Add support to process throttled EQEs Sahil Chandna
2026-10-08  8:39 ` netdev-bot+sinfo
2026-10-09 14:02   ` Sahil Chandna
2026-10-09  8:33 ` sashiko-bot

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox