* [PATCH v3] net: airoha: npu: use coherent DMA for mailbox messages
@ 2026-08-06 14:28 Daniel Pawlik
2026-08-06 15:39 ` Lorenzo Bianconi
0 siblings, 1 reply; 2+ messages in thread
From: Daniel Pawlik @ 2026-08-06 14:28 UTC (permalink / raw)
To: netdev; +Cc: lorenzo, win847, linux-arm-kernel, linux-mediatek, Daniel Pawlik
Commit 6f884eb87a79 ("net: airoha: Fix DMA direction for NPU mailbox
buffer") switched airoha_npu_send_msg() to DMA_BIDIRECTIONAL so
non-coherent CPUs invalidate caches before reading NPU GET responses.
On EN7581 + MT7996 that change regresses probe: the mailbox completes
successfully, but WLAN_FUNC_GET_WAIT_NPU_VERSION still reads as 0.0 and
mt76 never binds NPU offload. Healthy boards report 0.1111.
airoha_npu_send_msg() is also used from PPE foe_commit under
spin_lock_bh(), so per-message dma_alloc_coherent(GFP_ATOMIC) is a poor
fit. Allocate one device-managed coherent bounce buffer per NPU core at
probe and reuse it under the existing per-core mailbox lock. That also
keeps the buffer valid if the NPU completes a write after a mailbox
timeout.
Pass an optional reply buffer into airoha_npu_send_msg() so GET callers
(e.g. airoha_npu_wlan_msg_get) copy the response payload once instead of
bouncing through the request allocation.
Verified on Quantum Fiber / Gemtek W1700K (EN7581 + MT7996).
Fixes: 6f884eb87a79 ("net: airoha: Fix DMA direction for NPU mailbox buffer")
Link: https://patchwork.kernel.org/project/linux-mediatek/patch/20260805070851.2885888-1-pawlik.dan@gmail.com/
Link: https://patchwork.kernel.org/project/linux-mediatek/patch/20260806072601.1815487-1-pawlik.dan@gmail.com/
Assisted-by: Cursor:composer-2
Signed-off-by: Daniel Pawlik <pawlik.dan@gmail.com>
---
v3:
- move coherent mailbox buffer into struct airoha_npu_core (buf/addr)
- allocate one buffer per core at probe
- add optional reply buffer to airoha_npu_send_msg() to avoid the extra
copy in airoha_npu_wlan_msg_get()
- document AIROHA_NPU_MBOX_SIZE vs ppe_mbox_data / WLAN payloads
v2:
- allocate one coherent mailbox buffer at probe and reuse it under the
per-core lock instead of per-message dma_alloc_coherent(GFP_ATOMIC)
drivers/net/ethernet/airoha/airoha_npu.c | 67 ++++++++++++++++-------
include/linux/soc/airoha/airoha_offload.h | 3 +
2 files changed, 51 insertions(+), 19 deletions(-)
diff --git a/drivers/net/ethernet/airoha/airoha_npu.c b/drivers/net/ethernet/airoha/airoha_npu.c
index b679bed952de..e1e0ddb92f03 100644
--- a/drivers/net/ethernet/airoha/airoha_npu.c
+++ b/drivers/net/ethernet/airoha/airoha_npu.c
@@ -23,6 +23,12 @@
#define NPU_EN7581_FIRMWARE_RV32_MAX_SIZE 0x200000
#define NPU_EN7581_FIRMWARE_DATA_MAX_SIZE 0x10000
#define NPU_DUMP_SIZE 512
+/*
+ * Mailbox DMA payload size. Covers sizeof(struct ppe_mbox_data) (28) and
+ * WLAN TLV messages (header + payload); largest in-tree WLAN payload today
+ * is 16 bytes (INODE_TXRX_REG_ADDR). Keep headroom for future commands.
+ */
+#define AIROHA_NPU_MBOX_SIZE 256
#define REG_NPU_LOCAL_SRAM 0x0
@@ -161,21 +167,32 @@ struct wlan_mbox_data {
};
static int airoha_npu_send_msg(struct airoha_npu *npu, int func_id,
- void *p, int size)
+ const void *req, int size,
+ void *rsp, int rsp_off, int rsp_len)
{
u16 core = 0; /* FIXME */
+ struct airoha_npu_core *c = &npu->cores[core];
u32 val, offset = core << 4;
- dma_addr_t dma_addr;
int ret;
- dma_addr = dma_map_single(npu->dev, p, size, DMA_BIDIRECTIONAL);
- ret = dma_mapping_error(npu->dev, dma_addr);
- if (ret)
- return ret;
+ if (size > AIROHA_NPU_MBOX_SIZE ||
+ (rsp && (rsp_off < 0 || rsp_len < 0 ||
+ rsp_off + rsp_len > size)))
+ return -EINVAL;
- spin_lock_bh(&npu->cores[core].lock);
+ /*
+ * Mailbox payloads are bidirectional (CPU request, NPU response).
+ * On EN7581+MT7996, streaming DMA_BIDIRECTIONAL against the
+ * caller kzalloc() buffer can leave WLAN_FUNC_GET_WAIT_NPU_VERSION
+ * reading as 0.0 despite MBOX success. Reuse a probe-time coherent
+ * bounce buffer under the per-core lock (also used from PPE
+ * foe_commit under atomic context).
+ */
+ spin_lock_bh(&c->lock);
+
+ memcpy(c->buf, req, size);
- regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(0) + offset, dma_addr);
+ regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(0) + offset, c->addr);
regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(1) + offset, size);
regmap_read(npu->regmap, REG_CR_MBQ0_CTRL(2) + offset, &val);
regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(2) + offset, val + 1);
@@ -189,9 +206,10 @@ static int airoha_npu_send_msg(struct airoha_npu *npu, int func_id,
if (!ret && FIELD_GET(MBOX_MSG_STATUS, val) != NPU_MBOX_SUCCESS)
ret = -EINVAL;
- spin_unlock_bh(&npu->cores[core].lock);
+ if (!ret && rsp)
+ memcpy(rsp, c->buf + rsp_off, rsp_len);
- dma_unmap_single(npu->dev, dma_addr, size, DMA_BIDIRECTIONAL);
+ spin_unlock_bh(&c->lock);
return ret;
}
@@ -343,7 +361,7 @@ static int airoha_npu_ppe_init(struct airoha_npu *npu)
ppe_data->init_info.wan_mode = QDMA_WAN_ETHER;
err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
- sizeof(*ppe_data));
+ sizeof(*ppe_data), NULL, 0, 0);
kfree(ppe_data);
return err;
@@ -362,7 +380,7 @@ static int airoha_npu_ppe_deinit(struct airoha_npu *npu)
ppe_data->func_id = PPE_FUNC_SET_WAIT_HWNAT_DEINIT;
err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
- sizeof(*ppe_data));
+ sizeof(*ppe_data), NULL, 0, 0);
kfree(ppe_data);
return err;
@@ -386,7 +404,7 @@ static int airoha_npu_ppe_flush_sram_entries(struct airoha_npu *npu,
ppe_data->set_info.size = sram_num_entries;
err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
- sizeof(*ppe_data));
+ sizeof(*ppe_data), NULL, 0, 0);
kfree(ppe_data);
return err;
@@ -411,7 +429,7 @@ static int airoha_npu_foe_commit_entry(struct airoha_npu *npu,
: PPE_SRAM_SET_ENTRY;
err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
- sizeof(*ppe_data));
+ sizeof(*ppe_data), NULL, 0, 0);
if (err)
goto out;
@@ -420,7 +438,7 @@ static int airoha_npu_foe_commit_entry(struct airoha_npu *npu,
ppe_data->set_info.size = sizeof(u32);
err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
- sizeof(*ppe_data));
+ sizeof(*ppe_data), NULL, 0, 0);
out:
kfree(ppe_data);
@@ -443,6 +461,7 @@ static int airoha_npu_ppe_stats_setup(struct airoha_npu *npu,
ppe_data->stats_info.foe_stats_addr = foe_stats_addr;
err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
+ sizeof(*ppe_data), ppe_data, 0,
sizeof(*ppe_data));
if (err)
goto out;
@@ -475,7 +494,8 @@ static int airoha_npu_wlan_msg_send(struct airoha_npu *npu, int ifindex,
wlan_data->func_id = func_id;
memcpy(wlan_data->d, data, data_len);
- err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len);
+ err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len,
+ NULL, 0, 0);
kfree(wlan_data);
return err;
@@ -497,9 +517,9 @@ static int airoha_npu_wlan_msg_get(struct airoha_npu *npu, int ifindex,
wlan_data->func_type = NPU_OP_GET;
wlan_data->func_id = func_id;
- err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len);
- if (!err)
- memcpy(data, wlan_data->d, data_len);
+ err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len,
+ data, offsetof(struct wlan_mbox_data, d),
+ data_len);
kfree(wlan_data);
return err;
@@ -770,6 +790,15 @@ static int airoha_npu_probe(struct platform_device *pdev)
if (err)
return err;
+ for (i = 0; i < ARRAY_SIZE(npu->cores); i++) {
+ struct airoha_npu_core *core = &npu->cores[i];
+
+ core->buf = dmam_alloc_coherent(dev, AIROHA_NPU_MBOX_SIZE,
+ &core->addr, GFP_KERNEL);
+ if (!core->buf)
+ return -ENOMEM;
+ }
+
err = airoha_npu_run_firmware(dev, base, &res);
if (err)
return dev_err_probe(dev, err, "failed to run npu firmware\n");
diff --git a/include/linux/soc/airoha/airoha_offload.h b/include/linux/soc/airoha/airoha_offload.h
index 7589fccfeef6..16a988ad7467 100644
--- a/include/linux/soc/airoha/airoha_offload.h
+++ b/include/linux/soc/airoha/airoha_offload.h
@@ -173,6 +173,9 @@ struct airoha_npu {
/* protect concurrent npu memory accesses */
spinlock_t lock;
struct work_struct wdt_work;
+ /* Coherent bounce buffer for mailbox cmd/rsp */
+ void *buf;
+ dma_addr_t addr;
} cores[NPU_NUM_CORES];
int irqs[NPU_NUM_IRQ];
--
2.55.0
^ permalink raw reply related [flat|nested] 2+ messages in thread
* Re: [PATCH v3] net: airoha: npu: use coherent DMA for mailbox messages
2026-08-06 14:28 [PATCH v3] net: airoha: npu: use coherent DMA for mailbox messages Daniel Pawlik
@ 2026-08-06 15:39 ` Lorenzo Bianconi
0 siblings, 0 replies; 2+ messages in thread
From: Lorenzo Bianconi @ 2026-08-06 15:39 UTC (permalink / raw)
To: Daniel Pawlik; +Cc: netdev, win847, linux-arm-kernel, linux-mediatek
[-- Attachment #1: Type: text/plain, Size: 9203 bytes --]
On Aug 06, Daniel Pawlik wrote:
> Commit 6f884eb87a79 ("net: airoha: Fix DMA direction for NPU mailbox
> buffer") switched airoha_npu_send_msg() to DMA_BIDIRECTIONAL so
> non-coherent CPUs invalidate caches before reading NPU GET responses.
>
> On EN7581 + MT7996 that change regresses probe: the mailbox completes
> successfully, but WLAN_FUNC_GET_WAIT_NPU_VERSION still reads as 0.0 and
> mt76 never binds NPU offload. Healthy boards report 0.1111.
>
> airoha_npu_send_msg() is also used from PPE foe_commit under
> spin_lock_bh(), so per-message dma_alloc_coherent(GFP_ATOMIC) is a poor
> fit. Allocate one device-managed coherent bounce buffer per NPU core at
> probe and reuse it under the existing per-core mailbox lock. That also
> keeps the buffer valid if the NPU completes a write after a mailbox
> timeout.
>
> Pass an optional reply buffer into airoha_npu_send_msg() so GET callers
> (e.g. airoha_npu_wlan_msg_get) copy the response payload once instead of
> bouncing through the request allocation.
>
> Verified on Quantum Fiber / Gemtek W1700K (EN7581 + MT7996).
>
> Fixes: 6f884eb87a79 ("net: airoha: Fix DMA direction for NPU mailbox buffer")
> Link: https://patchwork.kernel.org/project/linux-mediatek/patch/20260805070851.2885888-1-pawlik.dan@gmail.com/
> Link: https://patchwork.kernel.org/project/linux-mediatek/patch/20260806072601.1815487-1-pawlik.dan@gmail.com/
> Assisted-by: Cursor:composer-2
> Signed-off-by: Daniel Pawlik <pawlik.dan@gmail.com>
> ---
> v3:
> - move coherent mailbox buffer into struct airoha_npu_core (buf/addr)
> - allocate one buffer per core at probe
> - add optional reply buffer to airoha_npu_send_msg() to avoid the extra
> copy in airoha_npu_wlan_msg_get()
> - document AIROHA_NPU_MBOX_SIZE vs ppe_mbox_data / WLAN payloads
>
> v2:
> - allocate one coherent mailbox buffer at probe and reuse it under the
> per-core lock instead of per-message dma_alloc_coherent(GFP_ATOMIC)
>
> drivers/net/ethernet/airoha/airoha_npu.c | 67 ++++++++++++++++-------
> include/linux/soc/airoha/airoha_offload.h | 3 +
> 2 files changed, 51 insertions(+), 19 deletions(-)
>
> diff --git a/drivers/net/ethernet/airoha/airoha_npu.c b/drivers/net/ethernet/airoha/airoha_npu.c
> index b679bed952de..e1e0ddb92f03 100644
> --- a/drivers/net/ethernet/airoha/airoha_npu.c
> +++ b/drivers/net/ethernet/airoha/airoha_npu.c
> @@ -23,6 +23,12 @@
> #define NPU_EN7581_FIRMWARE_RV32_MAX_SIZE 0x200000
> #define NPU_EN7581_FIRMWARE_DATA_MAX_SIZE 0x10000
> #define NPU_DUMP_SIZE 512
> +/*
> + * Mailbox DMA payload size. Covers sizeof(struct ppe_mbox_data) (28) and
> + * WLAN TLV messages (header + payload); largest in-tree WLAN payload today
> + * is 16 bytes (INODE_TXRX_REG_ADDR). Keep headroom for future commands.
> + */
> +#define AIROHA_NPU_MBOX_SIZE 256
>
> #define REG_NPU_LOCAL_SRAM 0x0
>
> @@ -161,21 +167,32 @@ struct wlan_mbox_data {
> };
>
> static int airoha_npu_send_msg(struct airoha_npu *npu, int func_id,
> - void *p, int size)
> + const void *req, int size,
> + void *rsp, int rsp_off, int rsp_len)
I guess you do not need rsp_off, just pass the data pointer (wlan_data->d for
airoha_npu_wlan_msg_get()). Moreover you can define something like:
static int __airoha_npu_send_msg(..., void *reply, int reply_len)
{
...
}
static inline int airoha_npu_send_msg() { return __airoha_npu_send_msg( ..., NULL, 0); }
In this way the change will be less intrusive.
> {
> u16 core = 0; /* FIXME */
> + struct airoha_npu_core *c = &npu->cores[core];
please respect RCT if possible.
Regards,
Lorenzo
> u32 val, offset = core << 4;
> - dma_addr_t dma_addr;
> int ret;
>
> - dma_addr = dma_map_single(npu->dev, p, size, DMA_BIDIRECTIONAL);
> - ret = dma_mapping_error(npu->dev, dma_addr);
> - if (ret)
> - return ret;
> + if (size > AIROHA_NPU_MBOX_SIZE ||
> + (rsp && (rsp_off < 0 || rsp_len < 0 ||
> + rsp_off + rsp_len > size)))
> + return -EINVAL;
>
> - spin_lock_bh(&npu->cores[core].lock);
> + /*
> + * Mailbox payloads are bidirectional (CPU request, NPU response).
> + * On EN7581+MT7996, streaming DMA_BIDIRECTIONAL against the
> + * caller kzalloc() buffer can leave WLAN_FUNC_GET_WAIT_NPU_VERSION
> + * reading as 0.0 despite MBOX success. Reuse a probe-time coherent
> + * bounce buffer under the per-core lock (also used from PPE
> + * foe_commit under atomic context).
> + */
> + spin_lock_bh(&c->lock);
> +
> + memcpy(c->buf, req, size);
>
> - regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(0) + offset, dma_addr);
> + regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(0) + offset, c->addr);
> regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(1) + offset, size);
> regmap_read(npu->regmap, REG_CR_MBQ0_CTRL(2) + offset, &val);
> regmap_write(npu->regmap, REG_CR_MBQ0_CTRL(2) + offset, val + 1);
> @@ -189,9 +206,10 @@ static int airoha_npu_send_msg(struct airoha_npu *npu, int func_id,
> if (!ret && FIELD_GET(MBOX_MSG_STATUS, val) != NPU_MBOX_SUCCESS)
> ret = -EINVAL;
>
> - spin_unlock_bh(&npu->cores[core].lock);
> + if (!ret && rsp)
> + memcpy(rsp, c->buf + rsp_off, rsp_len);
>
> - dma_unmap_single(npu->dev, dma_addr, size, DMA_BIDIRECTIONAL);
> + spin_unlock_bh(&c->lock);
>
> return ret;
> }
> @@ -343,7 +361,7 @@ static int airoha_npu_ppe_init(struct airoha_npu *npu)
> ppe_data->init_info.wan_mode = QDMA_WAN_ETHER;
>
> err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
> - sizeof(*ppe_data));
> + sizeof(*ppe_data), NULL, 0, 0);
> kfree(ppe_data);
>
> return err;
> @@ -362,7 +380,7 @@ static int airoha_npu_ppe_deinit(struct airoha_npu *npu)
> ppe_data->func_id = PPE_FUNC_SET_WAIT_HWNAT_DEINIT;
>
> err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
> - sizeof(*ppe_data));
> + sizeof(*ppe_data), NULL, 0, 0);
> kfree(ppe_data);
>
> return err;
> @@ -386,7 +404,7 @@ static int airoha_npu_ppe_flush_sram_entries(struct airoha_npu *npu,
> ppe_data->set_info.size = sram_num_entries;
>
> err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
> - sizeof(*ppe_data));
> + sizeof(*ppe_data), NULL, 0, 0);
> kfree(ppe_data);
>
> return err;
> @@ -411,7 +429,7 @@ static int airoha_npu_foe_commit_entry(struct airoha_npu *npu,
> : PPE_SRAM_SET_ENTRY;
>
> err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
> - sizeof(*ppe_data));
> + sizeof(*ppe_data), NULL, 0, 0);
> if (err)
> goto out;
>
> @@ -420,7 +438,7 @@ static int airoha_npu_foe_commit_entry(struct airoha_npu *npu,
> ppe_data->set_info.size = sizeof(u32);
>
> err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
> - sizeof(*ppe_data));
> + sizeof(*ppe_data), NULL, 0, 0);
> out:
> kfree(ppe_data);
>
> @@ -443,6 +461,7 @@ static int airoha_npu_ppe_stats_setup(struct airoha_npu *npu,
> ppe_data->stats_info.foe_stats_addr = foe_stats_addr;
>
> err = airoha_npu_send_msg(npu, NPU_FUNC_PPE, ppe_data,
> + sizeof(*ppe_data), ppe_data, 0,
> sizeof(*ppe_data));
> if (err)
> goto out;
> @@ -475,7 +494,8 @@ static int airoha_npu_wlan_msg_send(struct airoha_npu *npu, int ifindex,
> wlan_data->func_id = func_id;
> memcpy(wlan_data->d, data, data_len);
>
> - err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len);
> + err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len,
> + NULL, 0, 0);
> kfree(wlan_data);
>
> return err;
> @@ -497,9 +517,9 @@ static int airoha_npu_wlan_msg_get(struct airoha_npu *npu, int ifindex,
> wlan_data->func_type = NPU_OP_GET;
> wlan_data->func_id = func_id;
>
> - err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len);
> - if (!err)
> - memcpy(data, wlan_data->d, data_len);
> + err = airoha_npu_send_msg(npu, NPU_FUNC_WIFI, wlan_data, len,
> + data, offsetof(struct wlan_mbox_data, d),
> + data_len);
> kfree(wlan_data);
>
> return err;
> @@ -770,6 +790,15 @@ static int airoha_npu_probe(struct platform_device *pdev)
> if (err)
> return err;
>
> + for (i = 0; i < ARRAY_SIZE(npu->cores); i++) {
> + struct airoha_npu_core *core = &npu->cores[i];
> +
> + core->buf = dmam_alloc_coherent(dev, AIROHA_NPU_MBOX_SIZE,
> + &core->addr, GFP_KERNEL);
> + if (!core->buf)
> + return -ENOMEM;
> + }
> +
> err = airoha_npu_run_firmware(dev, base, &res);
> if (err)
> return dev_err_probe(dev, err, "failed to run npu firmware\n");
> diff --git a/include/linux/soc/airoha/airoha_offload.h b/include/linux/soc/airoha/airoha_offload.h
> index 7589fccfeef6..16a988ad7467 100644
> --- a/include/linux/soc/airoha/airoha_offload.h
> +++ b/include/linux/soc/airoha/airoha_offload.h
> @@ -173,6 +173,9 @@ struct airoha_npu {
> /* protect concurrent npu memory accesses */
> spinlock_t lock;
> struct work_struct wdt_work;
> + /* Coherent bounce buffer for mailbox cmd/rsp */
> + void *buf;
> + dma_addr_t addr;
> } cores[NPU_NUM_CORES];
>
> int irqs[NPU_NUM_IRQ];
> --
> 2.55.0
>
[-- Attachment #2: signature.asc --]
[-- Type: application/pgp-signature, Size: 228 bytes --]
^ permalink raw reply [flat|nested] 2+ messages in thread
end of thread, other threads:[~2026-08-06 15:40 UTC | newest]
Thread overview: 2+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-06 14:28 [PATCH v3] net: airoha: npu: use coherent DMA for mailbox messages Daniel Pawlik
2026-08-06 15:39 ` Lorenzo Bianconi
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox