DMA Engine development
 help / color / mirror / Atom feed
From: Linus Walleij <linusw@kernel.org>
To: Vinod Koul <vkoul@kernel.org>, Frank Li <Frank.Li@kernel.org>
Cc: dmaengine@vger.kernel.org, phone-devel@vger.kernel.org,
	 Linus Walleij <linusw@kernel.org>,
	Frank Li <Frank.li@oss.nxp.com>
Subject: [PATCH v6 02/23] dmaengine: ste_dma40: Fix cyclic transfer residue
Date: Thu, 24 Sep 2026 10:35:14 +0200	[thread overview]
Message-ID: <20260924-dma40-fixes-v6-2-fdb6755020a2@kernel.org> (raw)
In-Reply-To: <20260924-dma40-fixes-v6-0-fdb6755020a2@kernel.org>

DMA40 reads residue from the element count of the currently active LLI.
For a cyclic transfer this reports at most one period, not the bytes
remaining until the cyclic buffer wraps.

Once DMA40 advertises burst granularity, DMAengine PCM uses this residue
directly. For a four-period PCM buffer it consequently reports the
hardware pointer near three periods after every period interrupt. ALSA
eventually stops playback with -EIO although DMA period callbacks
continue.

Calculate cyclic residue from the current memory-side hardware pointer
instead. Read the destination pointer for capture and the source pointer
for playback. Sample the split logical channel pointer coherently and
retain the last valid residue during relink transitions.

Make at most three residue sampling attempts: one initial read plus two
retries. The retries let transient split-register updates or LLI relink
windows settle, while the limit prevents unbounded polling if hardware does
not yield a valid pointer.

This avoids counting terminal-count interrupts, which races with hardware
advancing to the next LLI and cannot account for coalesced interrupt
status. Also reject cyclic periods that expand into multiple LLIs because
logical cyclic LLIs each request a terminal-count interrupt and would
generate more than one callback per period.

Reject invalid cyclic geometries before dividing or constructing the
scatterlist as well.

Reported-by: Frank Li <Frank.li@oss.nxp.com>
Closes: https://lore.kernel.org/dmaengine/aq2wIPJW6viUxyy9@SMW015318/
Fixes: 15c606686541 ("dmaengine: ste_dma40: indicate granularity on channels")
Assisted-by: LLM
Signed-off-by: Linus Walleij <linusw@kernel.org>
---
 drivers/dma/ste_dma40.c | 106 +++++++++++++++++++++++++++++++++++++++++++++---
 1 file changed, 101 insertions(+), 5 deletions(-)

diff --git a/drivers/dma/ste_dma40.c b/drivers/dma/ste_dma40.c
index e4d689c9eba8..eab9c09b4bfe 100644
--- a/drivers/dma/ste_dma40.c
+++ b/drivers/dma/ste_dma40.c
@@ -65,6 +65,9 @@ struct stedma40_platform_data {
 /* Maximum iterations taken before giving up suspending a channel */
 #define D40_SUSPEND_MAX_IT 500
 
+/* Maximum attempts to sample a stable cyclic residue position */
+#define D40_RESIDUE_MAX_ATTEMPTS 3
+
 /* Milliseconds */
 #define DMA40_AUTOSUSPEND_DELAY	100
 
@@ -378,6 +381,9 @@ struct d40_lli_pool {
  * @lli_len: Number of llis of current descriptor.
  * @lli_current: Number of transferred llis.
  * @lcla_alloc: Number of LCLA entries allocated.
+ * @cyclic_dma_addr: Start address of the cyclic buffer.
+ * @cyclic_buf_len: Length of the cyclic buffer.
+ * @cyclic_residue: Last valid cyclic residue sample.
  * @txd: DMA engine struct. Used for among other things for communication
  * during a transfer.
  * @node: List entry.
@@ -396,6 +402,9 @@ struct d40_desc {
 	int				 lli_len;
 	int				 lli_current;
 	int				 lcla_alloc;
+	dma_addr_t			 cyclic_dma_addr;
+	size_t				 cyclic_buf_len;
+	size_t				 cyclic_residue;
 
 	struct dma_async_tx_descriptor	 txd;
 	struct list_head		 node;
@@ -1420,6 +1429,64 @@ static u32 d40_residue(struct d40_chan *d40c)
 	return num_elt * d40c->dma_cfg.dst_info.data_width;
 }
 
+static bool d40_current_addr(struct d40_chan *d40c, dma_addr_t *addr)
+{
+	bool dst = d40c->dma_cfg.dir == DMA_DEV_TO_MEM;
+	void __iomem *high_reg;
+	void __iomem *low_reg;
+	u32 low;
+	u32 high;
+	u32 check;
+	int i;
+
+	if (chan_is_physical(d40c)) {
+		*addr = readl(chan_base(d40c) +
+			      (dst ? D40_CHAN_REG_SDPTR : D40_CHAN_REG_SSPTR));
+		return true;
+	}
+
+	if (dst) {
+		low_reg = &d40c->lcpa->lcsp2;
+		high_reg = &d40c->lcpa->lcsp3;
+	} else {
+		low_reg = &d40c->lcpa->lcsp0;
+		high_reg = &d40c->lcpa->lcsp1;
+	}
+
+	for (i = 0; i < D40_RESIDUE_MAX_ATTEMPTS; i++) {
+		high = readl(high_reg) & D40_MEM_LCSP1_SPTR_MASK;
+		low = readl(low_reg) & D40_MEM_LCSP0_SPTR_MASK;
+		check = readl(high_reg) & D40_MEM_LCSP1_SPTR_MASK;
+		if (high == check) {
+			*addr = low | high;
+			return true;
+		}
+	}
+
+	return false;
+}
+
+static bool d40_cyclic_offset(struct d40_chan *d40c, struct d40_desc *d40d,
+			      size_t *offset)
+{
+	dma_addr_t current_addr;
+	dma_addr_t current_offset;
+	int i;
+
+	for (i = 0; i < D40_RESIDUE_MAX_ATTEMPTS; i++) {
+		if (!d40_current_addr(d40c, &current_addr))
+			continue;
+
+		current_offset = current_addr - d40d->cyclic_dma_addr;
+		if (current_offset <= d40d->cyclic_buf_len) {
+			*offset = current_offset;
+			return true;
+		}
+	}
+
+	return false;
+}
+
 static bool d40_tx_is_linked(struct d40_chan *d40c)
 {
 	bool is_link;
@@ -2108,15 +2175,26 @@ static bool d40_is_paused(struct d40_chan *d40c)
 
 }
 
-static u32 stedma40_residue(struct dma_chan *chan)
+static u32 stedma40_residue(struct dma_chan *chan, dma_cookie_t cookie)
 {
 	struct d40_chan *d40c =
 		container_of(chan, struct d40_chan, chan);
+	struct d40_desc *d40d;
+	size_t offset;
 	u32 bytes_left;
 	unsigned long flags;
 
 	spin_lock_irqsave(&d40c->lock, flags);
-	bytes_left = d40_residue(d40c);
+	d40d = d40_first_active_get(d40c);
+	if (d40d && d40d->txd.cookie == cookie && d40d->cyclic &&
+	    d40d->cyclic_buf_len) {
+		if (d40_cyclic_offset(d40c, d40d, &offset))
+			d40d->cyclic_residue = d40d->cyclic_buf_len - offset;
+		bytes_left = d40d->cyclic_residue;
+	} else {
+		bytes_left = d40_residue(d40c);
+	}
+
 	spin_unlock_irqrestore(&d40c->lock, flags);
 
 	return bytes_left;
@@ -2246,8 +2324,13 @@ d40_prep_sg(struct dma_chan *dchan, struct scatterlist *sg_src,
 	if (desc == NULL)
 		goto unlock;
 
-	if (sg_next(&sg_src[sg_len - 1]) == sg_src)
+	if (sg_next(&sg_src[sg_len - 1]) == sg_src) {
 		desc->cyclic = true;
+		if (desc->lli_len != sg_len) {
+			chan_err(chan, "Cyclic periods must fit in one LLI\n");
+			goto free_desc;
+		}
+	}
 
 	src_dev_addr = 0;
 	dst_dev_addr = 0;
@@ -2524,11 +2607,18 @@ dma40_prep_dma_cyclic(struct dma_chan *chan, dma_addr_t dma_addr,
 		     size_t buf_len, size_t period_len,
 		     enum dma_transfer_direction direction, unsigned long flags)
 {
-	unsigned int periods = buf_len / period_len;
+	unsigned int periods;
 	struct dma_async_tx_descriptor *txd;
+	struct d40_desc *desc;
 	struct scatterlist *sg;
+	dma_addr_t buf_addr = dma_addr;
 	int i;
 
+	if (!buf_len || !period_len || buf_len % period_len)
+		return NULL;
+
+	periods = buf_len / period_len;
+
 	sg = kzalloc_objs(struct scatterlist, periods + 1, GFP_NOWAIT);
 	if (!sg)
 		return NULL;
@@ -2543,6 +2633,12 @@ dma40_prep_dma_cyclic(struct dma_chan *chan, dma_addr_t dma_addr,
 
 	txd = d40_prep_sg(chan, sg, sg, periods, direction,
 			  DMA_PREP_INTERRUPT);
+	if (txd) {
+		desc = container_of(txd, struct d40_desc, txd);
+		desc->cyclic_dma_addr = buf_addr;
+		desc->cyclic_buf_len = buf_len;
+		desc->cyclic_residue = buf_len;
+	}
 
 	kfree(sg);
 
@@ -2563,7 +2659,7 @@ static enum dma_status d40_tx_status(struct dma_chan *chan,
 
 	ret = dma_cookie_status(chan, cookie, txstate);
 	if (ret != DMA_COMPLETE && txstate)
-		dma_set_residue(txstate, stedma40_residue(chan));
+		dma_set_residue(txstate, stedma40_residue(chan, cookie));
 
 	if (d40_is_paused(d40c))
 		ret = DMA_PAUSED;

-- 
2.55.0


  parent reply	other threads:[~2026-09-24  8:35 UTC|newest]

Thread overview: 45+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24  8:35 [PATCH v6 00/23] dmaengine: ste_dma40: Fix numerous accumulated bugs Linus Walleij
2026-09-24  8:35 ` [PATCH v6 01/23] dmaengine: ste_dma40: Fix physical cyclic capability Linus Walleij
2026-09-24  8:35 ` Linus Walleij [this message]
2026-09-24 14:37   ` [PATCH v6 02/23] dmaengine: ste_dma40: Fix cyclic transfer residue Frank Li
2026-09-24  8:35 ` [PATCH v6 03/23] dmaengine: ste_dma40: Recover coalesced cyclic callbacks Linus Walleij
2026-09-24 14:48   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 04/23] dmaengine: ste_dma40: Fix failed start cleanup Linus Walleij
2026-09-24  8:35 ` [PATCH v6 05/23] dmaengine: ste_dma40: Fix probe runtime PM disable Linus Walleij
2026-09-24 14:50   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 06/23] dmaengine: ste_dma40: Check runtime PM in IRQ Linus Walleij
2026-09-24  8:35 ` [PATCH v6 07/23] dmaengine: ste_dma40: Handle runtime PM resume errors Linus Walleij
2026-09-24 14:54   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 08/23] dmaengine: ste_dma40: Return IRQ_NONE when no interrupt is pending Linus Walleij
2026-09-24 14:55   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 09/23] dmaengine: ste_dma40: Init hardware before registration Linus Walleij
2026-09-24 14:58   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 10/23] dmaengine: ste_dma40: Fix probe IRQ leak Linus Walleij
2026-09-24 15:02   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 11/23] dmaengine: ste_dma40: Fix DMA registration unwind Linus Walleij
2026-09-24  9:10   ` sashiko-bot
2026-09-24 15:11   ` Frank Li
2026-09-27  8:41     ` Linus Walleij
2026-09-24  8:35 ` [PATCH v6 12/23] dmaengine: ste_dma40: Fix LCLA allocation order Linus Walleij
2026-09-24 15:16   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 13/23] dmaengine: ste_dma40: Fix probe LCLA free Linus Walleij
2026-09-24 15:34   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 14/23] dmaengine: ste_dma40: Put the LCPA SRAM node Linus Walleij
2026-09-24 15:43   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 15/23] dmaengine: ste_dma40: Fix memcpy channel parsing Linus Walleij
2026-09-24 15:49   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 16/23] dmaengine: ste_dma40: Validate disabled channel indexes Linus Walleij
2026-09-24 15:53   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 17/23] dmaengine: ste_dma40: Validate DMA specifier length Linus Walleij
2026-09-24 15:54   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 18/23] dmaengine: ste_dma40: Reject direction changes after allocation Linus Walleij
2026-09-24 15:57   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 19/23] dmaengine: ste_dma40: Fix logical channel bounds check Linus Walleij
2026-09-24  8:35 ` [PATCH v6 20/23] dmaengine: ste_dma40: Fix event group bounds Linus Walleij
2026-09-24  9:28   ` sashiko-bot
2026-09-24  8:35 ` [PATCH v6 21/23] dmaengine: ste_dma40: Search all blocks for fixed logical channels Linus Walleij
2026-09-24 16:08   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 22/23] dmaengine: ste_dma40: Validate fixed physical channel indexes Linus Walleij
2026-09-24 16:09   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 23/23] dmaengine: ste_dma40: Validate memcpy configuration Linus Walleij
2026-09-24 16:11   ` Frank Li

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260924-dma40-fixes-v6-2-fdb6755020a2@kernel.org \
    --to=linusw@kernel.org \
    --cc=Frank.Li@kernel.org \
    --cc=Frank.li@oss.nxp.com \
    --cc=dmaengine@vger.kernel.org \
    --cc=phone-devel@vger.kernel.org \
    --cc=vkoul@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox