DMA Engine development
 help / color / mirror / Atom feed
From: Frank Li <Frank.li@oss.nxp.com>
To: Linus Walleij <linusw@kernel.org>
Cc: Vinod Koul <vkoul@kernel.org>, Frank Li <Frank.Li@kernel.org>,
	dmaengine@vger.kernel.org, phone-devel@vger.kernel.org
Subject: Re: [PATCH v6 02/23] dmaengine: ste_dma40: Fix cyclic transfer residue
Date: Thu, 24 Sep 2026 09:37:18 -0500	[thread overview]
Message-ID: <arU1nsEFPn43Madb@SMW015318> (raw)
In-Reply-To: <20260924-dma40-fixes-v6-2-fdb6755020a2@kernel.org>

On Thu, Sep 24, 2026 at 10:35:14AM +0200, Linus Walleij wrote:
> DMA40 reads residue from the element count of the currently active LLI.
> For a cyclic transfer this reports at most one period, not the bytes
> remaining until the cyclic buffer wraps.
>
> Once DMA40 advertises burst granularity, DMAengine PCM uses this residue
> directly. For a four-period PCM buffer it consequently reports the
> hardware pointer near three periods after every period interrupt. ALSA
> eventually stops playback with -EIO although DMA period callbacks
> continue.
>
> Calculate cyclic residue from the current memory-side hardware pointer
> instead. Read the destination pointer for capture and the source pointer
> for playback. Sample the split logical channel pointer coherently and
> retain the last valid residue during relink transitions.
>
> Make at most three residue sampling attempts: one initial read plus two
> retries. The retries let transient split-register updates or LLI relink
> windows settle, while the limit prevents unbounded polling if hardware does
> not yield a valid pointer.
>
> This avoids counting terminal-count interrupts, which races with hardware
> advancing to the next LLI and cannot account for coalesced interrupt
> status. Also reject cyclic periods that expand into multiple LLIs because
> logical cyclic LLIs each request a terminal-count interrupt and would
> generate more than one callback per period.
>
> Reject invalid cyclic geometries before dividing or constructing the
> scatterlist as well.
>
> Reported-by: Frank Li <Frank.li@oss.nxp.com>
> Closes: https://lore.kernel.org/dmaengine/aq2wIPJW6viUxyy9@SMW015318/
> Fixes: 15c606686541 ("dmaengine: ste_dma40: indicate granularity on channels")
> Assisted-by: LLM
> Signed-off-by: Linus Walleij <linusw@kernel.org>
> ---

Reviewed-by: Frank Li <Frank.Li@nxp.com>

>  drivers/dma/ste_dma40.c | 106 +++++++++++++++++++++++++++++++++++++++++++++---
>  1 file changed, 101 insertions(+), 5 deletions(-)
>
> diff --git a/drivers/dma/ste_dma40.c b/drivers/dma/ste_dma40.c
> index e4d689c9eba8..eab9c09b4bfe 100644
> --- a/drivers/dma/ste_dma40.c
> +++ b/drivers/dma/ste_dma40.c
> @@ -65,6 +65,9 @@ struct stedma40_platform_data {
>  /* Maximum iterations taken before giving up suspending a channel */
>  #define D40_SUSPEND_MAX_IT 500
>
> +/* Maximum attempts to sample a stable cyclic residue position */
> +#define D40_RESIDUE_MAX_ATTEMPTS 3
> +
>  /* Milliseconds */
>  #define DMA40_AUTOSUSPEND_DELAY	100
>
> @@ -378,6 +381,9 @@ struct d40_lli_pool {
>   * @lli_len: Number of llis of current descriptor.
>   * @lli_current: Number of transferred llis.
>   * @lcla_alloc: Number of LCLA entries allocated.
> + * @cyclic_dma_addr: Start address of the cyclic buffer.
> + * @cyclic_buf_len: Length of the cyclic buffer.
> + * @cyclic_residue: Last valid cyclic residue sample.
>   * @txd: DMA engine struct. Used for among other things for communication
>   * during a transfer.
>   * @node: List entry.
> @@ -396,6 +402,9 @@ struct d40_desc {
>  	int				 lli_len;
>  	int				 lli_current;
>  	int				 lcla_alloc;
> +	dma_addr_t			 cyclic_dma_addr;
> +	size_t				 cyclic_buf_len;
> +	size_t				 cyclic_residue;
>
>  	struct dma_async_tx_descriptor	 txd;
>  	struct list_head		 node;
> @@ -1420,6 +1429,64 @@ static u32 d40_residue(struct d40_chan *d40c)
>  	return num_elt * d40c->dma_cfg.dst_info.data_width;
>  }
>
> +static bool d40_current_addr(struct d40_chan *d40c, dma_addr_t *addr)
> +{
> +	bool dst = d40c->dma_cfg.dir == DMA_DEV_TO_MEM;
> +	void __iomem *high_reg;
> +	void __iomem *low_reg;
> +	u32 low;
> +	u32 high;
> +	u32 check;
> +	int i;
> +
> +	if (chan_is_physical(d40c)) {
> +		*addr = readl(chan_base(d40c) +
> +			      (dst ? D40_CHAN_REG_SDPTR : D40_CHAN_REG_SSPTR));
> +		return true;
> +	}
> +
> +	if (dst) {
> +		low_reg = &d40c->lcpa->lcsp2;
> +		high_reg = &d40c->lcpa->lcsp3;
> +	} else {
> +		low_reg = &d40c->lcpa->lcsp0;
> +		high_reg = &d40c->lcpa->lcsp1;
> +	}
> +
> +	for (i = 0; i < D40_RESIDUE_MAX_ATTEMPTS; i++) {
> +		high = readl(high_reg) & D40_MEM_LCSP1_SPTR_MASK;
> +		low = readl(low_reg) & D40_MEM_LCSP0_SPTR_MASK;
> +		check = readl(high_reg) & D40_MEM_LCSP1_SPTR_MASK;
> +		if (high == check) {
> +			*addr = low | high;
> +			return true;
> +		}
> +	}
> +
> +	return false;
> +}
> +
> +static bool d40_cyclic_offset(struct d40_chan *d40c, struct d40_desc *d40d,
> +			      size_t *offset)
> +{
> +	dma_addr_t current_addr;
> +	dma_addr_t current_offset;
> +	int i;
> +
> +	for (i = 0; i < D40_RESIDUE_MAX_ATTEMPTS; i++) {
> +		if (!d40_current_addr(d40c, &current_addr))
> +			continue;
> +
> +		current_offset = current_addr - d40d->cyclic_dma_addr;
> +		if (current_offset <= d40d->cyclic_buf_len) {
> +			*offset = current_offset;
> +			return true;
> +		}
> +	}
> +
> +	return false;
> +}
> +
>  static bool d40_tx_is_linked(struct d40_chan *d40c)
>  {
>  	bool is_link;
> @@ -2108,15 +2175,26 @@ static bool d40_is_paused(struct d40_chan *d40c)
>
>  }
>
> -static u32 stedma40_residue(struct dma_chan *chan)
> +static u32 stedma40_residue(struct dma_chan *chan, dma_cookie_t cookie)
>  {
>  	struct d40_chan *d40c =
>  		container_of(chan, struct d40_chan, chan);
> +	struct d40_desc *d40d;
> +	size_t offset;
>  	u32 bytes_left;
>  	unsigned long flags;
>
>  	spin_lock_irqsave(&d40c->lock, flags);
> -	bytes_left = d40_residue(d40c);
> +	d40d = d40_first_active_get(d40c);
> +	if (d40d && d40d->txd.cookie == cookie && d40d->cyclic &&
> +	    d40d->cyclic_buf_len) {
> +		if (d40_cyclic_offset(d40c, d40d, &offset))
> +			d40d->cyclic_residue = d40d->cyclic_buf_len - offset;
> +		bytes_left = d40d->cyclic_residue;
> +	} else {
> +		bytes_left = d40_residue(d40c);
> +	}
> +
>  	spin_unlock_irqrestore(&d40c->lock, flags);
>
>  	return bytes_left;
> @@ -2246,8 +2324,13 @@ d40_prep_sg(struct dma_chan *dchan, struct scatterlist *sg_src,
>  	if (desc == NULL)
>  		goto unlock;
>
> -	if (sg_next(&sg_src[sg_len - 1]) == sg_src)
> +	if (sg_next(&sg_src[sg_len - 1]) == sg_src) {
>  		desc->cyclic = true;
> +		if (desc->lli_len != sg_len) {
> +			chan_err(chan, "Cyclic periods must fit in one LLI\n");
> +			goto free_desc;
> +		}
> +	}
>
>  	src_dev_addr = 0;
>  	dst_dev_addr = 0;
> @@ -2524,11 +2607,18 @@ dma40_prep_dma_cyclic(struct dma_chan *chan, dma_addr_t dma_addr,
>  		     size_t buf_len, size_t period_len,
>  		     enum dma_transfer_direction direction, unsigned long flags)
>  {
> -	unsigned int periods = buf_len / period_len;
> +	unsigned int periods;
>  	struct dma_async_tx_descriptor *txd;
> +	struct d40_desc *desc;
>  	struct scatterlist *sg;
> +	dma_addr_t buf_addr = dma_addr;
>  	int i;
>
> +	if (!buf_len || !period_len || buf_len % period_len)
> +		return NULL;
> +
> +	periods = buf_len / period_len;
> +
>  	sg = kzalloc_objs(struct scatterlist, periods + 1, GFP_NOWAIT);
>  	if (!sg)
>  		return NULL;
> @@ -2543,6 +2633,12 @@ dma40_prep_dma_cyclic(struct dma_chan *chan, dma_addr_t dma_addr,
>
>  	txd = d40_prep_sg(chan, sg, sg, periods, direction,
>  			  DMA_PREP_INTERRUPT);
> +	if (txd) {
> +		desc = container_of(txd, struct d40_desc, txd);
> +		desc->cyclic_dma_addr = buf_addr;
> +		desc->cyclic_buf_len = buf_len;
> +		desc->cyclic_residue = buf_len;
> +	}
>
>  	kfree(sg);
>
> @@ -2563,7 +2659,7 @@ static enum dma_status d40_tx_status(struct dma_chan *chan,
>
>  	ret = dma_cookie_status(chan, cookie, txstate);
>  	if (ret != DMA_COMPLETE && txstate)
> -		dma_set_residue(txstate, stedma40_residue(chan));
> +		dma_set_residue(txstate, stedma40_residue(chan, cookie));
>
>  	if (d40_is_paused(d40c))
>  		ret = DMA_PAUSED;
>
> --
> 2.55.0
>

  reply	other threads:[~2026-09-24 14:37 UTC|newest]

Thread overview: 45+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24  8:35 [PATCH v6 00/23] dmaengine: ste_dma40: Fix numerous accumulated bugs Linus Walleij
2026-09-24  8:35 ` [PATCH v6 01/23] dmaengine: ste_dma40: Fix physical cyclic capability Linus Walleij
2026-09-24  8:35 ` [PATCH v6 02/23] dmaengine: ste_dma40: Fix cyclic transfer residue Linus Walleij
2026-09-24 14:37   ` Frank Li [this message]
2026-09-24  8:35 ` [PATCH v6 03/23] dmaengine: ste_dma40: Recover coalesced cyclic callbacks Linus Walleij
2026-09-24 14:48   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 04/23] dmaengine: ste_dma40: Fix failed start cleanup Linus Walleij
2026-09-24  8:35 ` [PATCH v6 05/23] dmaengine: ste_dma40: Fix probe runtime PM disable Linus Walleij
2026-09-24 14:50   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 06/23] dmaengine: ste_dma40: Check runtime PM in IRQ Linus Walleij
2026-09-24  8:35 ` [PATCH v6 07/23] dmaengine: ste_dma40: Handle runtime PM resume errors Linus Walleij
2026-09-24 14:54   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 08/23] dmaengine: ste_dma40: Return IRQ_NONE when no interrupt is pending Linus Walleij
2026-09-24 14:55   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 09/23] dmaengine: ste_dma40: Init hardware before registration Linus Walleij
2026-09-24 14:58   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 10/23] dmaengine: ste_dma40: Fix probe IRQ leak Linus Walleij
2026-09-24 15:02   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 11/23] dmaengine: ste_dma40: Fix DMA registration unwind Linus Walleij
2026-09-24  9:10   ` sashiko-bot
2026-09-24 15:11   ` Frank Li
2026-09-27  8:41     ` Linus Walleij
2026-09-24  8:35 ` [PATCH v6 12/23] dmaengine: ste_dma40: Fix LCLA allocation order Linus Walleij
2026-09-24 15:16   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 13/23] dmaengine: ste_dma40: Fix probe LCLA free Linus Walleij
2026-09-24 15:34   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 14/23] dmaengine: ste_dma40: Put the LCPA SRAM node Linus Walleij
2026-09-24 15:43   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 15/23] dmaengine: ste_dma40: Fix memcpy channel parsing Linus Walleij
2026-09-24 15:49   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 16/23] dmaengine: ste_dma40: Validate disabled channel indexes Linus Walleij
2026-09-24 15:53   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 17/23] dmaengine: ste_dma40: Validate DMA specifier length Linus Walleij
2026-09-24 15:54   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 18/23] dmaengine: ste_dma40: Reject direction changes after allocation Linus Walleij
2026-09-24 15:57   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 19/23] dmaengine: ste_dma40: Fix logical channel bounds check Linus Walleij
2026-09-24  8:35 ` [PATCH v6 20/23] dmaengine: ste_dma40: Fix event group bounds Linus Walleij
2026-09-24  9:28   ` sashiko-bot
2026-09-24  8:35 ` [PATCH v6 21/23] dmaengine: ste_dma40: Search all blocks for fixed logical channels Linus Walleij
2026-09-24 16:08   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 22/23] dmaengine: ste_dma40: Validate fixed physical channel indexes Linus Walleij
2026-09-24 16:09   ` Frank Li
2026-09-24  8:35 ` [PATCH v6 23/23] dmaengine: ste_dma40: Validate memcpy configuration Linus Walleij
2026-09-24 16:11   ` Frank Li

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=arU1nsEFPn43Madb@SMW015318 \
    --to=frank.li@oss.nxp.com \
    --cc=Frank.Li@kernel.org \
    --cc=dmaengine@vger.kernel.org \
    --cc=linusw@kernel.org \
    --cc=phone-devel@vger.kernel.org \
    --cc=vkoul@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox