All of lore.kernel.org
 help / color / mirror / Atom feed
From: Max Chou <max.chou@sifive.com>
To: Richard Henderson <richard.henderson@linaro.org>
Cc: qemu-devel@nongnu.org, frank.chang@sifive.com, qemu-riscv@nongnu.org
Subject: Re: [PATCH 05/23] target/riscv: Add evl argument to vext_ldst_elem_fn_host
Date: Wed, 26 Aug 2026 02:30:55 +0800	[thread overview]
Message-ID: <ao3Yufx1A7yoMoDw@sifive.com> (raw)
In-Reply-To: <20260815194549.1377505-6-richard.henderson@linaro.org>

On 2026-08-15 12:45, Richard Henderson wrote:
> This merges vext_continuous_ldst_host into the ldst_host function.
> We can then handle the byte little-endian optimization at compile-time.
> 
> Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
> ---
>  target/riscv/tcg/vector_helper.c | 83 +++++++++++++++-----------------
>  1 file changed, 40 insertions(+), 43 deletions(-)
> 
> diff --git a/target/riscv/tcg/vector_helper.c b/target/riscv/tcg/vector_helper.c
> index 661f1ab2f6..3786a17231 100644
> --- a/target/riscv/tcg/vector_helper.c
> +++ b/target/riscv/tcg/vector_helper.c
> @@ -214,7 +214,8 @@ static inline MemOpIdx vext_make_memop_idx(CPURISCVState *env, size_t size)
>  /* elements operations for load and store */
>  typedef void vext_ldst_elem_fn_tlb(CPURISCVState *env, abi_ptr addr,
>                                     uint32_t idx, void *vd, uintptr_t retaddr);
> -typedef void vext_ldst_elem_fn_host(void *vd, uint32_t idx, void *host);
> +typedef void vext_ldst_elem_fn_host(void *vd, void *host,
> +                                    uint32_t idx, uint32_t evl);
>  
>  #define GEN_VEXT_TLB_LD_ELEM(NAME, ETYPE, H, LDSUF)         \
>  static inline QEMU_ALWAYS_INLINE                            \
> @@ -226,12 +227,15 @@ void NAME##_tlb(CPURISCVState *env, abi_ptr addr,           \
>      *cur = cpu_##LDSUF##_mmu(env, addr, oi, retaddr);       \
>  }                                                           \
>  
> -#define GEN_VEXT_HOST_LD_ELEM(NAME, ETYPE, H, LDSUF)        \
> -static inline QEMU_ALWAYS_INLINE                            \
> -void NAME##_host(void *vd, uint32_t idx, void *host)        \
> -{                                                           \
> -    ETYPE *cur = ((ETYPE *)vd + H(idx));                    \
> -    *cur = (ETYPE)LDSUF##_p(host);                          \
> +#define GEN_VEXT_HOST_LD_ELEM(NAME, ETYPE, H, LDSUF)                    \
> +static inline QEMU_ALWAYS_INLINE                                        \
> +void NAME##_host(void *vd, void *host, uint32_t idx, uint32_t evl)      \
> +{                                                                       \
> +    do {                                                                \
> +        ETYPE *cur = (ETYPE *)vd + H(idx);                              \
> +        *cur = LDSUF##_p(host);                                         \
> +        host += sizeof(ETYPE);                                          \
> +    } while (++idx < evl);                                              \
>  }
>  
>  GEN_VEXT_TLB_LD_ELEM(lde_b, uint8_t,  H1, ldb)
> @@ -239,7 +243,16 @@ GEN_VEXT_TLB_LD_ELEM(lde_h, uint16_t, H2, ldw)
>  GEN_VEXT_TLB_LD_ELEM(lde_w, uint32_t, H4, ldl)
>  GEN_VEXT_TLB_LD_ELEM(lde_d, uint64_t, H8, ldq)
>  
> +#if HOST_BIG_ENDIAN
>  GEN_VEXT_HOST_LD_ELEM(lde_b, uint8_t,  H1, ldub)
> +#else
> +static inline QEMU_ALWAYS_INLINE
> +void lde_b_host(void *vd, void *host, uint32_t idx, uint32_t evl)
> +{
> +    memcpy(vd + idx, host, evl - idx);
> +}
> +#endif
> +
>  GEN_VEXT_HOST_LD_ELEM(lde_h, uint16_t, H2, lduw_le)
>  GEN_VEXT_HOST_LD_ELEM(lde_w, uint32_t, H4, ldl_le)
>  GEN_VEXT_HOST_LD_ELEM(lde_d, uint64_t, H8, ldq_le)
> @@ -254,12 +267,15 @@ void NAME##_tlb(CPURISCVState *env, abi_ptr addr,           \
>      cpu_##STSUF##_mmu(env, addr, data, oi, retaddr);        \
>  }                                                           \
>  
> -#define GEN_VEXT_HOST_ST_ELEM(NAME, ETYPE, H, STSUF)        \
> -static inline QEMU_ALWAYS_INLINE                            \
> -void NAME##_host(void *vd, uint32_t idx, void *host)        \
> -{                                                           \
> -    ETYPE data = *((ETYPE *)vd + H(idx));                   \
> -    STSUF##_p(host, data);                                  \
> +#define GEN_VEXT_HOST_ST_ELEM(NAME, ETYPE, H, STSUF)                    \
> +static inline QEMU_ALWAYS_INLINE                                        \
> +void NAME##_host(void *vd, void *host, uint32_t idx, uint32_t evl)      \
> +{                                                                       \
> +    do {                                                                \
> +        ETYPE data = *((ETYPE *)vd + H(idx));                           \
> +        STSUF##_p(host, data);                                          \
> +        host += sizeof(ETYPE);                                          \
> +    } while (++idx < evl);                                              \
>  }
>  
>  GEN_VEXT_TLB_ST_ELEM(ste_b, uint8_t,  H1, stb)
> @@ -267,7 +283,16 @@ GEN_VEXT_TLB_ST_ELEM(ste_h, uint16_t, H2, stw)
>  GEN_VEXT_TLB_ST_ELEM(ste_w, uint32_t, H4, stl)
>  GEN_VEXT_TLB_ST_ELEM(ste_d, uint64_t, H8, stq)
>  
> +#if HOST_BIG_ENDIAN
>  GEN_VEXT_HOST_ST_ELEM(ste_b, uint8_t,  H1, stb)
> +#else
> +static inline QEMU_ALWAYS_INLINE
> +void ste_b_host(void *vd, void *host, uint32_t idx, uint32_t evl)
> +{
> +    memcpy(host, vd + idx, evl - idx);
> +}
> +#endif
> +
>  GEN_VEXT_HOST_ST_ELEM(ste_h, uint16_t, H2, stw_le)
>  GEN_VEXT_HOST_ST_ELEM(ste_w, uint32_t, H4, stl_le)
>  GEN_VEXT_HOST_ST_ELEM(ste_d, uint64_t, H8, stq_le)
> @@ -284,33 +309,6 @@ vext_continuous_ldst_tlb(CPURISCVState *env, vext_ldst_elem_fn_tlb *ldst_tlb,
>      }
>  }
>  
> -static inline QEMU_ALWAYS_INLINE void
> -vext_continuous_ldst_host(CPURISCVState *env, vext_ldst_elem_fn_host *ldst_host,
> -                        void *vd, uint32_t evl, uint32_t reg_start, void *host,
> -                        uint32_t esz, bool is_load)
> -{
> -    if (HOST_BIG_ENDIAN) {
> -        for (; reg_start < evl; reg_start++, host += esz) {
> -            ldst_host(vd, reg_start, host);
> -        }
> -    } else {
> -        if (esz == 1) {
> -            uint32_t byte_offset = reg_start * esz;
> -            uint32_t size = (evl - reg_start) * esz;
> -
> -            if (is_load) {
> -                memcpy(vd + byte_offset, host, size);
> -            } else {
> -                memcpy(host, vd + byte_offset, size);
> -            }
> -        } else {
> -            for (; reg_start < evl; reg_start++, host += esz) {
> -                ldst_host(vd, reg_start, host);
> -            }
> -        }
> -    }
> -}
> -
>  static void vext_set_tail_elems_1s(uint32_t vl, void *vd, uint32_t nf,
>                                     uint32_t esz, uint32_t max_elems)
>  {
> @@ -334,7 +332,7 @@ static void vext_ldst_nf_host(void *vd, void *host, uint32_t i, uint32_t nf,
>                                vext_ldst_elem_fn_host *ldst_host)
>  {
>      for (uint32_t k = 0; k < nf; k++, host += esz) {
> -        ldst_host(vd, i + k * max_elems, host);
> +        ldst_host(vd + k * max_elems, host, i, i + 1);

The type of ldst_host vd parameter is void *, so vd + k * max_elems
advances in bytes, not the target element width.

I think here should be

> +        ldst_host(vd, host, i + k * max_elems, i + k * max_elems + 1);


rnax


  parent reply	other threads:[~2026-08-25 18:31 UTC|newest]

Thread overview: 33+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-15 19:45 [PATCH 00/23] target/riscv: Reorg vector load/store Richard Henderson
2026-08-15 19:45 ` [PATCH 01/23] target/riscv: Split out vext_set_nf_elems_1s Richard Henderson
2026-08-15 19:45 ` [PATCH 02/23] target/riscv: Hoist vma check out of vext_set_tail_elems_1s Richard Henderson
2026-08-16 15:25   ` Richard Henderson
2026-08-15 19:45 ` [PATCH 03/23] target/riscv: Split out vext_ldst_nf_tlb Richard Henderson
2026-08-15 19:45 ` [PATCH 04/23] target/riscv: Split out vext_ldst_nf_host Richard Henderson
2026-08-15 19:45 ` [PATCH 05/23] target/riscv: Add evl argument to vext_ldst_elem_fn_host Richard Henderson
2026-08-16 15:55   ` Richard Henderson
2026-08-25 18:30   ` Max Chou [this message]
2026-08-15 19:45 ` [PATCH 06/23] target/riscv: Drop is_load parameter from vext_continuous_ldst_tlb Richard Henderson
2026-08-15 19:45 ` [PATCH 07/23] target/riscv: Remove vext_continuous_ldst_tlb Richard Henderson
2026-08-15 19:45 ` [PATCH 08/23] target/riscv: Move misalignment check out of vext_page_ldst_us Richard Henderson
2026-08-15 19:45 ` [PATCH 09/23] target/riscv: Split out vext_page_ldst_us_{host,tlb} Richard Henderson
2026-08-15 19:45 ` [PATCH 10/23] target/riscv: Rewrite vext_ldst_us Richard Henderson
2026-08-25 19:00   ` Max Chou
2026-08-15 19:45 ` [PATCH 11/23] target/riscv: Rewrite vext_ldff Richard Henderson
2026-08-25 17:58   ` Max Chou
2026-08-25 20:15     ` Richard Henderson
2026-08-26 18:40       ` Max Chou
2026-08-26 21:33         ` Richard Henderson
2026-08-27 12:48           ` Max Chou
2026-08-15 19:45 ` [PATCH 12/23] target/riscv: Split out vext_ldst_us_desc Richard Henderson
2026-08-15 19:45 ` [PATCH 13/23] target/riscv: Use vext_ldst_us in vext_ldst_whole Richard Henderson
2026-08-15 19:45 ` [PATCH 14/23] target/riscv: Mark VSTART_CHECK_EARLY_EXIT unlikely Richard Henderson
2026-08-15 19:45 ` [PATCH 15/23] target/riscv: Remove unused VDATA,WD Richard Henderson
2026-08-15 19:45 ` [PATCH 16/23] target/riscv: Add MEM_IDX, BSWAP, ALIGN to VDATA Richard Henderson
2026-08-15 19:45 ` [PATCH 17/23] target/riscv: Pass MemOpIdx to vext_ldst_us Richard Henderson
2026-08-15 19:45 ` [PATCH 18/23] target/riscv: Build MemOpIdx to vext_ldff Richard Henderson
2026-08-15 19:45 ` [PATCH 19/23] target/riscv: Pass MemOpIdx to vext_ldst_elem_fn_tlb Richard Henderson
2026-08-15 19:45 ` [PATCH 20/23] target/riscv: Use FLATTEN rather than ALWAYS_INLINE for vector ldst Richard Henderson
2026-08-15 19:45 ` [PATCH 21/23] target/riscv: Drop v0 argument from gen_helper_ldst_us Richard Henderson
2026-08-15 19:45 ` [PATCH 22/23] target/riscv: Drop v0 argument from gen_helper_ldst_stride Richard Henderson
2026-08-15 19:45 ` [PATCH 23/23] target/riscv: Drop v0 argument from gen_helper_ldst_index Richard Henderson

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=ao3Yufx1A7yoMoDw@sifive.com \
    --to=max.chou@sifive.com \
    --cc=frank.chang@sifive.com \
    --cc=qemu-devel@nongnu.org \
    --cc=qemu-riscv@nongnu.org \
    --cc=richard.henderson@linaro.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.