From: Pierrick Bouvier <pierrick.bouvier@oss.qualcomm.com>
To: Brian Cain <brian.cain@oss.qualcomm.com>, qemu-devel@nongnu.org
Cc: "Alex Bennée" <alex.bennee@linaro.org>, sid.manning@oss.qualcomm.com
Subject: Re: [PATCH 10/17] tests/tcg/hexagon: add HVX vwhist tests
Date: Mon, 14 Sep 2026 14:21:08 -0700 [thread overview]
Message-ID: <92ff5291-7e27-4e2c-af01-09744c97dfe1@oss.qualcomm.com> (raw)
In-Reply-To: <20260905180949.1852673-11-brian.cain@oss.qualcomm.com>
On 9/5/2026 11:09 AM, Brian Cain wrote:
> Test all eight vwhist instruction variants and the four Q-masked forms.
>
> Signed-off-by: Brian Cain <brian.cain@oss.qualcomm.com>
> ---
> tests/tcg/hexagon/test_vwhist.c | 361 ++++++++++++++++++++++++++++++++
> tests/tcg/hexagon/meson.build | 1 +
> 2 files changed, 362 insertions(+)
> create mode 100644 tests/tcg/hexagon/test_vwhist.c
>
> diff --git a/tests/tcg/hexagon/test_vwhist.c b/tests/tcg/hexagon/test_vwhist.c
> new file mode 100644
> index 00000000000..5e143f79a43
> --- /dev/null
> +++ b/tests/tcg/hexagon/test_vwhist.c
> @@ -0,0 +1,361 @@
> +/*
> + * Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries.
> + * SPDX-License-Identifier: GPL-2.0-or-later
> + */
> +
> +/*
> + * Test HVX weighted histogram instructions (vwhist128/vwhist256 variants).
> + *
> + * Each halfword in the input vector (loaded via Vx.tmp) contains a bucket
> + * index (low byte) and a weight (high byte). The instruction accumulates
> + * the weight into the appropriate element of the destination vector.
> + */
> +
> +#include <stdio.h>
> +#include <stdint.h>
> +#include <string.h>
> +
> +int err;
> +
> +#define MAX_VEC_SIZE_BYTES 128
> +
> +typedef union {
> + uint32_t uw[MAX_VEC_SIZE_BYTES / 4];
> + uint16_t uh[MAX_VEC_SIZE_BYTES / 2];
> + uint8_t ub[MAX_VEC_SIZE_BYTES];
> +} MMVector;
> +
> +/*
> + * Build input vector for vwhist256: each halfword is (weight << 8 | bucket).
> + * We put bucket=0, weight=1 in every slot so that v0.uh[0] gets incremented
> + * by 1 for each of the 64 halfwords -> v0.uh[0] should equal 64.
> + */
> +static MMVector input __attribute__((aligned(MAX_VEC_SIZE_BYTES)));
> +static MMVector zero_vec __attribute__((aligned(MAX_VEC_SIZE_BYTES)));
> +static MMVector result __attribute__((aligned(MAX_VEC_SIZE_BYTES)));
> +
> +static void check_uint32(int line, int idx, uint32_t val, uint32_t expect)
> +{
> + if (val != expect) {
> + printf("ERROR at line %d: [%d] 0x%08x != 0x%08x\n",
> + line, idx, val, expect);
> + err++;
> + }
> +}
> +
> +static void check_uint16(int line, int idx, uint16_t val, uint16_t expect)
> +{
> + if (val != expect) {
> + printf("ERROR at line %d: [%d] 0x%04x != 0x%04x\n",
> + line, idx, val, expect);
> + err++;
> + }
> +}
> +
Seems like duplication of check* in hex_test.h.
check16 needs to be added there.
> +/*
> + * Test vwhist256: 256-bin histogram with 16-bit accumulation.
> + * Each halfword of the input: low byte = bucket, high byte = weight.
> + * bucket bits [7:3] -> vindex (which vector register), [2:0] + lane -> element.
> + * All entries have bucket=0 weight=1, so v0.uh[0..7] get incremented.
> + */
> +static void test_vwhist256(void)
> +{
> + int i;
> +
> + /* Set all input halfwords to bucket=0, weight=1 */
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0x0100; /* weight=1, bucket=0 */
> + }
> +
> + asm volatile(
> + /* Zero out v0 (the target for bucket 0) */
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist256\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result)
> + : "v0", "v12", "memory");
> +
> + /*
> + * 64 input halfwords all targeting bucket 0.
> + * elindex = (i & ~7) | (bucket & 7) = i & ~7 (since bucket=0).
> + * So elements 0,8,16,24,32,40,48,56 each get 8 increments.
> + */
> + for (i = 0; i < 64; i++) {
> + uint16_t expected = ((i & 7) == 0) ? 8 : 0;
> + check_uint16(__LINE__, i, result.uh[i], expected);
> + }
> +}
> +
> +/*
> + * Test vwhist256:sat -- saturating variant.
> + * Use weight=0xFF to test that saturation to 0xFFFF works.
> + */
> +static void test_vwhist256_sat(void)
> +{
> + int i;
> +
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0xFF00; /* weight=0xFF, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist256:sat\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result)
> + : "v0", "v12", "memory");
> +
> + /*
> + * Same distribution as vwhist256: elements 0,8,16,...,56.
> + * Each gets 8 * 0xFF = 2040 = 0x7F8 (within uint16 range).
> + */
> + for (i = 0; i < 64; i++) {
> + uint16_t expected = ((i & 7) == 0) ? 8 * 0xFF : 0;
> + check_uint16(__LINE__, i, result.uh[i], expected);
> + }
> +}
> +
> +/*
> + * Test vwhist128: 128-bin histogram with 32-bit accumulation.
> + * bucket bits [7:3] -> vindex, [2:1] + lane -> element (word granularity).
> + */
> +static void test_vwhist128(void)
> +{
> + int i;
> +
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0x0100; /* weight=1, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist128\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result)
> + : "v0", "v12", "memory");
> +
> + /*
> + * 128-bin: 64 halfwords all targeting bucket 0.
> + * bucket 0 -> vindex=0, elindex based on (i>>1)&(~3) | (bucket>>1)&3.
> + * With bucket=0, elindex = (i>>1)&(~3).
> + * For i=0,1: elindex=0; i=2,3: elindex=0; ... up to i=6,7: elindex=0
> + * i ranges 0..63. (i>>1) ranges 0..31.
> + * (i>>1)&(~3) = 0,0,0,0,4,4,4,4,8,...
> + * So elements 0,4,8,12,16,20,24,28 each get 8 increments.
> + */
> + for (i = 0; i < 32; i++) {
> + uint32_t expected = ((i & 3) == 0) ? 8 : 0;
> + check_uint32(__LINE__, i, result.uw[i], expected);
> + }
> +}
> +
> +/*
> + * Test vwhist128(#0) -- masked variant.
> + * Only processes elements where (bucket & 1) == mode.
> + */
> +static void test_vwhist128m(void)
> +{
> + int i;
> +
> + /* All bucket=0, weight=1. bucket&1==0, so mode=0 matches all. */
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0x0100; /* weight=1, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist128(#0)\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result)
> + : "v0", "v12", "memory");
> +
> + /* Same distribution as vwhist128 since all buckets have bit0=0 */
> + for (i = 0; i < 32; i++) {
> + uint32_t expected = ((i & 3) == 0) ? 8 : 0;
> + check_uint32(__LINE__, i, result.uw[i], expected);
> + }
> +
> + /* Now test with mode=1: bucket=0 has bit0=0, so nothing should match */
> + memset(&result, 0, sizeof(result));
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist128(#1)\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result)
> + : "v0", "v12", "memory");
> +
> + for (i = 0; i < 32; i++) {
> + check_uint32(__LINE__, i, result.uw[i], 0);
> + }
> +}
> +
> +static MMVector ones_vec __attribute__((aligned(MAX_VEC_SIZE_BYTES)));
> +
> +/*
> + * Test vwhist256(Qv4) -- Q-masked vwhist256.
> + * Set Q0 to all-ones so all elements pass the mask -> same as vwhist256.
> + */
> +static void test_vwhist256q(void)
> +{
> + int i;
> +
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0x0100; /* weight=1, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "v1 = vmem(%[ones] + #0)\n\t"
> + "q0 = vcmp.eq(v1.b, v1.b)\n\t" /* all-ones Q0 */
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist256(q0)\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result), [ones] "r"(&ones_vec)
> + : "v0", "v1", "v12", "q0", "memory");
> +
> + for (i = 0; i < 64; i++) {
> + uint16_t expected = ((i & 7) == 0) ? 8 : 0;
> + check_uint16(__LINE__, i, result.uh[i], expected);
> + }
> +}
> +
> +/*
> + * Test vwhist256(Qv4):sat -- Q-masked saturating vwhist256.
> + */
> +static void test_vwhist256q_sat(void)
> +{
> + int i;
> +
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0xFF00; /* weight=0xFF, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "v1 = vmem(%[ones] + #0)\n\t"
> + "q0 = vcmp.eq(v1.b, v1.b)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist256(q0):sat\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result), [ones] "r"(&ones_vec)
> + : "v0", "v1", "v12", "q0", "memory");
> +
> + for (i = 0; i < 64; i++) {
> + uint16_t expected = ((i & 7) == 0) ? 8 * 0xFF : 0;
> + check_uint16(__LINE__, i, result.uh[i], expected);
> + }
> +}
> +
> +/*
> + * Test vwhist128(Qv4) -- Q-masked vwhist128.
> + */
> +static void test_vwhist128q(void)
> +{
> + int i;
> +
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0x0100; /* weight=1, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "v1 = vmem(%[ones] + #0)\n\t"
> + "q0 = vcmp.eq(v1.b, v1.b)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist128(q0)\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result), [ones] "r"(&ones_vec)
> + : "v0", "v1", "v12", "q0", "memory");
> +
> + for (i = 0; i < 32; i++) {
> + uint32_t expected = ((i & 3) == 0) ? 8 : 0;
> + check_uint32(__LINE__, i, result.uw[i], expected);
> + }
> +}
> +
> +/*
> + * Test vwhist128(Qv4,#0) -- Q-masked mode vwhist128.
> + */
> +static void test_vwhist128qm(void)
> +{
> + int i;
> +
> + for (i = 0; i < MAX_VEC_SIZE_BYTES / 2; i++) {
> + input.uh[i] = 0x0100; /* weight=1, bucket=0 */
> + }
> +
> + asm volatile(
> + "v0 = vmem(%[zero] + #0)\n\t"
> + "v1 = vmem(%[ones] + #0)\n\t"
> + "q0 = vcmp.eq(v1.b, v1.b)\n\t"
> + "{\n\t"
> + " v12.tmp = vmem(%[inp] + #0)\n\t"
> + " vwhist128(q0,#0)\n\t"
> + "}\n\t"
> + "vmem(%[out] + #0) = v0\n\t"
> + :
> + : [inp] "r"(&input), [zero] "r"(&zero_vec),
> + [out] "r"(&result), [ones] "r"(&ones_vec)
> + : "v0", "v1", "v12", "q0", "memory");
> +
> + /* bucket=0, bit0=0, mode=0 matches -> same as vwhist128 */
> + for (i = 0; i < 32; i++) {
> + uint32_t expected = ((i & 3) == 0) ? 8 : 0;
> + check_uint32(__LINE__, i, result.uw[i], expected);
> + }
> +}
> +
> +int main(void)
> +{
> + memset(&zero_vec, 0, sizeof(zero_vec));
> + memset(&ones_vec, 0xff, sizeof(ones_vec));
> +
> + test_vwhist256();
> + test_vwhist256_sat();
> + test_vwhist128();
> + test_vwhist128m();
> + test_vwhist256q();
> + test_vwhist256q_sat();
> + test_vwhist128q();
> + test_vwhist128qm();
> +
> + puts(err ? "FAIL" : "PASS");
> + return err ? 1 : 0;
> +}
> diff --git a/tests/tcg/hexagon/meson.build b/tests/tcg/hexagon/meson.build
> index fcf0bd38f26..4c70875034d 100644
> --- a/tests/tcg/hexagon/meson.build
> +++ b/tests/tcg/hexagon/meson.build
> @@ -94,6 +94,7 @@ tests += {
> 'test_vminh.S': {'cflags': asmflags},
> 'test_vpmpyh.S': {'cflags': asmflags},
> 'test_vspliceb.S': {'cflags': asmflags},
> + 'test_vwhist.c': {'cflags': [cflags, '-mhvx']},
> 'unaligned_pc.c': {'cflags': cflags},
> 'unaligned_data.c': {'cflags': cflags},
> 'usr.c': {
next prev parent reply other threads:[~2026-09-14 21:22 UTC|newest]
Thread overview: 35+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-05 18:09 [PATCH 00/17] hexagon: TCG test coverage + ubsan fixes Brian Cain
2026-09-05 18:09 ` [PATCH 01/17] tests/tcg/hexagon: fix undefined signed left shift in circ.c Brian Cain
2026-09-08 15:54 ` Sid Manning
2026-09-05 18:09 ` [PATCH 02/17] tests/tcg/hexagon: fix undefined signed left shift in brev.c Brian Cain
2026-09-09 14:43 ` Sid Manning
2026-09-05 18:09 ` [PATCH 03/17] tests/tcg/hexagon: fix undefined signed integer overflow in hvx_misc Brian Cain
2026-09-09 14:50 ` Sid Manning
2026-09-05 18:09 ` [PATCH 04/17] tests/tcg/hexagon: fix undefined integer overflow in v69_hvx.c Brian Cain
2026-09-09 14:51 ` Sid Manning
2026-09-05 18:09 ` [PATCH 05/17] tests/tcg/multiarch: suppress UBSan for float-to-int conversions Brian Cain
2026-09-14 19:43 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 06/17] tests/tcg/hexagon: add FP classification, conversion, and fixup tests Brian Cain
2026-09-14 19:43 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 07/17] tests/tcg/hexagon: add bit interleave and convergent rounding tests Brian Cain
2026-09-10 17:54 ` Sid Manning
2026-09-05 18:09 ` [PATCH 08/17] tests/tcg/hexagon: add cond call, tstbit-jump, and endloop01 tests Brian Cain
2026-09-14 19:44 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 09/17] tests/tcg/hexagon: add dczeroa cache line zero test Brian Cain
2026-09-10 18:14 ` Sid Manning
2026-09-05 18:09 ` [PATCH 10/17] tests/tcg/hexagon: add HVX vwhist tests Brian Cain
2026-09-14 21:21 ` Pierrick Bouvier [this message]
2026-09-05 18:09 ` [PATCH 11/17] tests/tcg/hexagon: add sfrecipa edge case tests Brian Cain
2026-09-14 21:22 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 12/17] tests/tcg/hexagon: add dfmpyhh Brian Cain
2026-09-14 21:22 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 13/17] tests/tcg/hexagon: add unconditional store-immediate tests Brian Cain
2026-09-14 21:23 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 14/17] tests/tcg/hexagon: add predicate register transfer test Brian Cain
2026-09-14 21:23 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 15/17] tests/tcg/hexagon: add rounding conv, sffma:lib, and sfinvsqrta Brian Cain
2026-09-14 21:23 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 16/17] tests/tcg/hexagon: add scalar+HVX and masked store tests Brian Cain
2026-09-14 21:24 ` Pierrick Bouvier
2026-09-05 18:09 ` [PATCH 17/17] tests/tcg/hexagon: add decbin MPS test case Brian Cain
2026-09-14 21:24 ` Pierrick Bouvier
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=92ff5291-7e27-4e2c-af01-09744c97dfe1@oss.qualcomm.com \
--to=pierrick.bouvier@oss.qualcomm.com \
--cc=alex.bennee@linaro.org \
--cc=brian.cain@oss.qualcomm.com \
--cc=qemu-devel@nongnu.org \
--cc=sid.manning@oss.qualcomm.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.