qemu-devel.nongnu.org archive mirror
 help / color / mirror / Atom feed
From: Richard Henderson <richard.henderson@linaro.org>
To: qemu-devel@nongnu.org
Cc: berrange@redhat.com, ardb@kernel.org
Subject: [PATCH v2 11/18] target/s390x: Use clmul_32* routines
Date: Fri, 18 Aug 2023 18:02:11 -0700	[thread overview]
Message-ID: <20230819010218.192706-12-richard.henderson@linaro.org> (raw)
In-Reply-To: <20230819010218.192706-1-richard.henderson@linaro.org>

Use generic routines for 32-bit carry-less multiply.
Remove our local version of galois_multiply32.

Signed-off-by: Richard Henderson <richard.henderson@linaro.org>
---
 target/s390x/tcg/vec_int_helper.c | 75 +++++++++----------------------
 1 file changed, 22 insertions(+), 53 deletions(-)

diff --git a/target/s390x/tcg/vec_int_helper.c b/target/s390x/tcg/vec_int_helper.c
index 11477556e5..ba284b5379 100644
--- a/target/s390x/tcg/vec_int_helper.c
+++ b/target/s390x/tcg/vec_int_helper.c
@@ -165,22 +165,6 @@ DEF_VCTZ(8)
 DEF_VCTZ(16)
 
 /* like binary multiplication, but XOR instead of addition */
-#define DEF_GALOIS_MULTIPLY(BITS, TBITS)                                       \
-static uint##TBITS##_t galois_multiply##BITS(uint##TBITS##_t a,                \
-                                             uint##TBITS##_t b)                \
-{                                                                              \
-    uint##TBITS##_t res = 0;                                                   \
-                                                                               \
-    while (b) {                                                                \
-        if (b & 0x1) {                                                         \
-            res = res ^ a;                                                     \
-        }                                                                      \
-        a = a << 1;                                                            \
-        b = b >> 1;                                                            \
-    }                                                                          \
-    return res;                                                                \
-}
-DEF_GALOIS_MULTIPLY(32, 64)
 
 static S390Vector galois_multiply64(uint64_t a, uint64_t b)
 {
@@ -254,24 +238,29 @@ void HELPER(gvec_vgfma16)(void *v1, const void *v2, const void *v3,
     q1[1] = do_gfma16(q2[1], q3[1], q4[1]);
 }
 
-#define DEF_VGFM(BITS, TBITS)                                                  \
-void HELPER(gvec_vgfm##BITS)(void *v1, const void *v2, const void *v3,         \
-                             uint32_t desc)                                    \
-{                                                                              \
-    int i;                                                                     \
-                                                                               \
-    for (i = 0; i < (128 / TBITS); i++) {                                      \
-        uint##BITS##_t a = s390_vec_read_element##BITS(v2, i * 2);             \
-        uint##BITS##_t b = s390_vec_read_element##BITS(v3, i * 2);             \
-        uint##TBITS##_t d = galois_multiply##BITS(a, b);                       \
-                                                                               \
-        a = s390_vec_read_element##BITS(v2, i * 2 + 1);                        \
-        b = s390_vec_read_element##BITS(v3, i * 2 + 1);                        \
-        d = d ^ galois_multiply32(a, b);                                       \
-        s390_vec_write_element##TBITS(v1, i, d);                               \
-    }                                                                          \
+static inline uint64_t do_gfma32(uint64_t n, uint64_t m, uint64_t a)
+{
+    return clmul_32(n, m) ^ clmul_32(n >> 32, m >> 32) ^ a;
+}
+
+void HELPER(gvec_vgfm32)(void *v1, const void *v2, const void *v3, uint32_t d)
+{
+    uint64_t *q1 = v1;
+    const uint64_t *q2 = v2, *q3 = v3;
+
+    q1[0] = do_gfma32(q2[0], q3[0], 0);
+    q1[1] = do_gfma32(q2[1], q3[1], 0);
+}
+
+void HELPER(gvec_vgfma32)(void *v1, const void *v2, const void *v3,
+                         const void *v4, uint32_t d)
+{
+    uint64_t *q1 = v1;
+    const uint64_t *q2 = v2, *q3 = v3, *q4 = v4;
+
+    q1[0] = do_gfma32(q2[0], q3[0], q4[0]);
+    q1[1] = do_gfma32(q2[1], q3[1], q4[1]);
 }
-DEF_VGFM(32, 64)
 
 void HELPER(gvec_vgfm64)(void *v1, const void *v2, const void *v3,
                          uint32_t desc)
@@ -288,26 +277,6 @@ void HELPER(gvec_vgfm64)(void *v1, const void *v2, const void *v3,
     s390_vec_xor(v1, &tmp1, &tmp2);
 }
 
-#define DEF_VGFMA(BITS, TBITS)                                                 \
-void HELPER(gvec_vgfma##BITS)(void *v1, const void *v2, const void *v3,        \
-                              const void *v4, uint32_t desc)                   \
-{                                                                              \
-    int i;                                                                     \
-                                                                               \
-    for (i = 0; i < (128 / TBITS); i++) {                                      \
-        uint##BITS##_t a = s390_vec_read_element##BITS(v2, i * 2);             \
-        uint##BITS##_t b = s390_vec_read_element##BITS(v3, i * 2);             \
-        uint##TBITS##_t d = galois_multiply##BITS(a, b);                       \
-                                                                               \
-        a = s390_vec_read_element##BITS(v2, i * 2 + 1);                        \
-        b = s390_vec_read_element##BITS(v3, i * 2 + 1);                        \
-        d = d ^ galois_multiply32(a, b);                                       \
-        d = d ^ s390_vec_read_element##TBITS(v4, i);                           \
-        s390_vec_write_element##TBITS(v1, i, d);                               \
-    }                                                                          \
-}
-DEF_VGFMA(32, 64)
-
 void HELPER(gvec_vgfma64)(void *v1, const void *v2, const void *v3,
                           const void *v4, uint32_t desc)
 {
-- 
2.34.1



  parent reply	other threads:[~2023-08-19  1:05 UTC|newest]

Thread overview: 34+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2023-08-19  1:02 [PATCH v2 00/18] crypto: Provide clmul.h and host accel Richard Henderson
2023-08-19  1:02 ` [PATCH v2 01/18] crypto: Add generic 8-bit carry-less multiply routines Richard Henderson
2023-09-10 12:19   ` Ard Biesheuvel
2023-08-19  1:02 ` [PATCH v2 02/18] target/arm: Use clmul_8* routines Richard Henderson
2023-08-21  7:49   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 03/18] target/s390x: " Richard Henderson
2023-08-21 12:45   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 04/18] target/ppc: " Richard Henderson
2023-08-19  1:02 ` [PATCH v2 05/18] crypto: Add generic 16-bit carry-less multiply routines Richard Henderson
2023-09-10 12:22   ` Ard Biesheuvel
2023-08-19  1:02 ` [PATCH v2 06/18] target/arm: Use clmul_16* routines Richard Henderson
2023-08-21  7:51   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 07/18] target/s390x: " Richard Henderson
2023-08-21 12:44   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 08/18] target/ppc: " Richard Henderson
2023-08-19  1:02 ` [PATCH v2 09/18] crypto: Add generic 32-bit carry-less multiply routines Richard Henderson
2023-09-10 12:23   ` Ard Biesheuvel
2023-08-19  1:02 ` [PATCH v2 10/18] target/arm: Use clmul_32* routines Richard Henderson
2023-08-21  7:53   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` Richard Henderson [this message]
2023-08-21 12:39   ` [PATCH v2 11/18] target/s390x: " Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 12/18] target/ppc: " Richard Henderson
2023-08-19  1:02 ` [PATCH v2 13/18] crypto: Add generic 64-bit carry-less multiply routine Richard Henderson
2023-08-19  1:02 ` [PATCH v2 14/18] target/arm: Use clmul_64 Richard Henderson
2023-08-21 12:27   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 15/18] target/s390x: " Richard Henderson
2023-08-21 12:33   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 16/18] target/ppc: " Richard Henderson
2023-08-21 12:34   ` Philippe Mathieu-Daudé
2023-08-19  1:02 ` [PATCH v2 17/18] host/include/i386: Implement clmul.h Richard Henderson
2023-08-19  1:02 ` [PATCH v2 18/18] host/include/aarch64: " Richard Henderson
2023-08-21 14:57 ` [PATCH v2 00/18] crypto: Provide clmul.h and host accel Ard Biesheuvel
2023-08-21 15:14   ` Richard Henderson
2023-08-21 15:44     ` Ard Biesheuvel

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20230819010218.192706-12-richard.henderson@linaro.org \
    --to=richard.henderson@linaro.org \
    --cc=ardb@kernel.org \
    --cc=berrange@redhat.com \
    --cc=qemu-devel@nongnu.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox;
as well as URLs for NNTP newsgroup(s).