All of lore.kernel.org
 help / color / mirror / Atom feed
From: Dapeng Mi <dapeng1.mi@linux.intel.com>
To: Peter Zijlstra <peterz@infradead.org>,
	Ingo Molnar <mingo@redhat.com>,
	Arnaldo Carvalho de Melo <acme@kernel.org>,
	Namhyung Kim <namhyung@kernel.org>,
	Thomas Gleixner <tglx@linutronix.de>,
	Dave Hansen <dave.hansen@linux.intel.com>,
	Ian Rogers <irogers@google.com>,
	Adrian Hunter <adrian.hunter@intel.com>,
	Jiri Olsa <jolsa@kernel.org>,
	Alexander Shishkin <alexander.shishkin@linux.intel.com>,
	Andi Kleen <ak@linux.intel.com>,
	Eranian Stephane <eranian@google.com>
Cc: Mark Rutland <mark.rutland@arm.com>,
	broonie@kernel.org, Ravi Bangoria <ravi.bangoria@amd.com>,
	linux-kernel@vger.kernel.org, linux-perf-users@vger.kernel.org,
	Zide Chen <zide.chen@intel.com>,
	Falcon Thomas <thomas.falcon@intel.com>,
	Dapeng Mi <dapeng1.mi@intel.com>,
	Xudong Hao <xudong.hao@intel.com>,
	Dapeng Mi <dapeng1.mi@linux.intel.com>,
	Kan Liang <kan.liang@linux.intel.com>
Subject: [Patch v10 15/23] perf/x86: Support ZMM sampling using sample_simd_vec_reg_* fields
Date: Tue, 21 Jul 2026 14:24:58 +0800	[thread overview]
Message-ID: <20260721062506.3745816-16-dapeng1.mi@linux.intel.com> (raw)
In-Reply-To: <20260721062506.3745816-1-dapeng1.mi@linux.intel.com>

Support sampling of ZMM registers via the sample_simd_vec_reg_* fields.

Each ZMM register consists of 8 u64 words. Current x86 hardware supports
up to 32 ZMM registers. For ZMM registers from ZMM0 to ZMM15, they are
assembled from three parts: XMM (the lower 2 u64 words),
YMMH (the middle 2 u64 words), and ZMMH (the upper 4 u64 words). The
perf_simd_reg_value() function is responsible for assembling these three
parts into a complete ZMM register for output to userspace.

For ZMM registers ZMM16 to ZMM31, each register can be read as a whole
and directly outputted to userspace.

Additionally, sample_simd_vec_reg_qwords should be set to 8 to indicate
ZMM sampling.

ZMM sampling will be enabled in a subsequent patch that sets
PERF_PMU_CAP_SIMD_REGS.

Co-developed-by: Kan Liang <kan.liang@linux.intel.com>
Signed-off-by: Kan Liang <kan.liang@linux.intel.com>
Signed-off-by: Dapeng Mi <dapeng1.mi@linux.intel.com>
---
 arch/x86/events/core.c                | 22 ++++++++++-
 arch/x86/events/perf_event.h          | 54 +++++++++++++++++++++++++++
 arch/x86/include/asm/perf_event.h     |  8 ++++
 arch/x86/include/uapi/asm/perf_regs.h |  8 +++-
 arch/x86/kernel/perf_regs.c           | 19 +++++++++-
 5 files changed, 106 insertions(+), 5 deletions(-)

diff --git a/arch/x86/events/core.c b/arch/x86/events/core.c
index 3039a311e733..b0e3f1b9fa19 100644
--- a/arch/x86/events/core.c
+++ b/arch/x86/events/core.c
@@ -642,8 +642,10 @@ static int pebs_simd_regs_validate(struct perf_event *event)
 	if (event_needs_xmm(event) &&
 	    x86_pmu.arch_pebs && !(caps & ARCH_PEBS_VECR_XMM))
 		return -EINVAL;
-	/* PEBS does not support YMM registers sampling yet. */
-	if (event_needs_ymm(event))
+	/* PEBS does not support YMM/ZMM registers sampling yet. */
+	if (event_needs_ymm(event) ||
+	    event_needs_low16_zmm(event) ||
+	    event_needs_high16_zmm(event))
 		return -EINVAL;
 
 	return 0;
@@ -662,6 +664,12 @@ static int event_simd_regs_validate(struct perf_event *event)
 	if (event_needs_ymm(event) &&
 	   !(x86_pmu.ext_regs_mask & XFEATURE_MASK_YMM))
 		return -EINVAL;
+	if (event_needs_low16_zmm(event) &&
+	    !(x86_pmu.ext_regs_mask & XFEATURE_MASK_ZMM_Hi256))
+		return -EINVAL;
+	if (event_needs_high16_zmm(event) &&
+	    !(x86_pmu.ext_regs_mask & XFEATURE_MASK_Hi16_ZMM))
+		return -EINVAL;
 
 	return 0;
 }
@@ -1826,6 +1834,8 @@ void x86_pmu_clear_perf_regs(struct pt_regs *regs)
 	perf_regs->abi = PERF_SAMPLE_REGS_ABI_NONE;
 	perf_regs->xmm_regs = NULL;
 	perf_regs->ymmh_regs = NULL;
+	perf_regs->zmmh_regs = NULL;
+	perf_regs->h16zmm_regs = NULL;
 }
 
 static void update_perf_regs(struct x86_perf_regs *perf_regs,
@@ -1843,6 +1853,10 @@ static void update_perf_regs(struct x86_perf_regs *perf_regs,
 		perf_regs->xmm_space = xsave->i387.xmm_space;
 	if (mask & XFEATURE_MASK_YMM)
 		perf_regs->ymmh = get_xsave_addr(xsave, XFEATURE_YMM);
+	if (mask & XFEATURE_MASK_ZMM_Hi256)
+		perf_regs->zmmh = get_xsave_addr(xsave, XFEATURE_ZMM_Hi256);
+	if (mask & XFEATURE_MASK_Hi16_ZMM)
+		perf_regs->h16zmm = get_xsave_addr(xsave, XFEATURE_Hi16_ZMM);
 }
 
 /*
@@ -2013,6 +2027,10 @@ static u64 get_simd_sample_mask(struct perf_event *event, u64 sample_type)
 		mask |= XFEATURE_MASK_SSE;
 	if (__event_needs_ymm(event, sample_type))
 		mask |= XFEATURE_MASK_YMM;
+	if (__event_needs_low16_zmm(event, sample_type))
+		mask |= XFEATURE_MASK_ZMM_Hi256;
+	if (__event_needs_high16_zmm(event, sample_type))
+		mask |= XFEATURE_MASK_Hi16_ZMM;
 
 	return mask;
 }
diff --git a/arch/x86/events/perf_event.h b/arch/x86/events/perf_event.h
index 01de7799f907..f59551200f18 100644
--- a/arch/x86/events/perf_event.h
+++ b/arch/x86/events/perf_event.h
@@ -209,6 +209,60 @@ static inline bool event_needs_ymm(struct perf_event *event)
 			PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
 }
 
+static inline bool __event_needs_low16_zmm(struct perf_event *event,
+					   u64 sample_type)
+{
+	if (!event->attr.sample_simd_regs_enabled)
+		return false;
+	if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_ZMM_QWORDS)
+		return false;
+
+	if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+	    (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+	    (event->attr.sample_simd_vec_reg_user > 0))
+		return true;
+
+	if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+	    (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+	    (event->attr.sample_simd_vec_reg_intr > 0))
+		return true;
+
+	return false;
+}
+
+static inline bool event_needs_low16_zmm(struct perf_event *event)
+{
+	return __event_needs_low16_zmm(event,
+			PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
+static inline bool __event_needs_high16_zmm(struct perf_event *event,
+					    u64 sample_type)
+{
+	if (!event->attr.sample_simd_regs_enabled)
+		return false;
+	if (event->attr.sample_simd_vec_reg_qwords < PERF_X86_ZMM_QWORDS)
+		return false;
+
+	if ((sample_type & PERF_SAMPLE_REGS_USER) &&
+	    (event->attr.sample_type & PERF_SAMPLE_REGS_USER) &&
+	    (fls64(event->attr.sample_simd_vec_reg_user) > PERF_X86_H16ZMM_BASE))
+		return true;
+
+	if ((sample_type & PERF_SAMPLE_REGS_INTR) &&
+	    (event->attr.sample_type & PERF_SAMPLE_REGS_INTR) &&
+	    (fls64(event->attr.sample_simd_vec_reg_intr) > PERF_X86_H16ZMM_BASE))
+		return true;
+
+	return false;
+}
+
+static inline bool event_needs_high16_zmm(struct perf_event *event)
+{
+	return __event_needs_high16_zmm(event,
+			PERF_SAMPLE_REGS_INTR | PERF_SAMPLE_REGS_USER);
+}
+
 struct amd_nb {
 	int nb_id;  /* NorthBridge id */
 	int refcnt; /* reference count */
diff --git a/arch/x86/include/asm/perf_event.h b/arch/x86/include/asm/perf_event.h
index da77845e1f02..75394c4e8bc3 100644
--- a/arch/x86/include/asm/perf_event.h
+++ b/arch/x86/include/asm/perf_event.h
@@ -737,6 +737,14 @@ struct x86_perf_regs {
 		u64	*ymmh_regs;
 		struct ymmh_struct *ymmh;
 	};
+	union {
+		u64	*zmmh_regs;
+		struct avx_512_zmm_uppers_state *zmmh;
+	};
+	union {
+		u64	*h16zmm_regs;
+		struct avx_512_hi16_state *h16zmm;
+	};
 };
 
 extern unsigned long perf_arch_instruction_pointer(struct pt_regs *regs);
diff --git a/arch/x86/include/uapi/asm/perf_regs.h b/arch/x86/include/uapi/asm/perf_regs.h
index d544f6d79871..b88d0b6822fd 100644
--- a/arch/x86/include/uapi/asm/perf_regs.h
+++ b/arch/x86/include/uapi/asm/perf_regs.h
@@ -60,16 +60,20 @@ enum perf_event_x86_regs {
 enum {
 	PERF_X86_SIMD_XMM_REGS      = 16,
 	PERF_X86_SIMD_YMM_REGS      = 16,
-	PERF_X86_SIMD_VEC_REGS_MAX  = PERF_X86_SIMD_YMM_REGS,
+	PERF_X86_SIMD_ZMM_REGS      = 32,
+	PERF_X86_SIMD_VEC_REGS_MAX  = PERF_X86_SIMD_ZMM_REGS,
 };
 
 #define PERF_X86_SIMD_VEC_MASK	__GENMASK_ULL(PERF_X86_SIMD_VEC_REGS_MAX - 1, 0)
 
+#define PERF_X86_H16ZMM_BASE		16
+
 enum {
 	/* 1 qword = 8 bytes */
 	PERF_X86_XMM_QWORDS      = 2,
 	PERF_X86_YMM_QWORDS      = 4,
-	PERF_X86_SIMD_QWORDS_MAX = PERF_X86_YMM_QWORDS,
+	PERF_X86_ZMM_QWORDS      = 8,
+	PERF_X86_SIMD_QWORDS_MAX = PERF_X86_ZMM_QWORDS,
 };
 
 #endif /* _ASM_X86_PERF_REGS_H */
diff --git a/arch/x86/kernel/perf_regs.c b/arch/x86/kernel/perf_regs.c
index 0076974498ee..93370d465786 100644
--- a/arch/x86/kernel/perf_regs.c
+++ b/arch/x86/kernel/perf_regs.c
@@ -78,6 +78,7 @@ u64 perf_reg_value(struct pt_regs *regs, int idx)
 }
 
 #define PERF_X86_YMMH_QWORDS	(PERF_X86_YMM_QWORDS / 2)
+#define PERF_X86_ZMMH_QWORDS	(PERF_X86_ZMM_QWORDS / 2)
 
 u64 perf_simd_reg_value(struct pt_regs *regs, int idx,
 			u16 qwords_idx, bool pred)
@@ -95,6 +96,13 @@ u64 perf_simd_reg_value(struct pt_regs *regs, int idx,
 			 qwords_idx >= PERF_X86_SIMD_QWORDS_MAX))
 		return 0;
 
+	if (idx >= PERF_X86_H16ZMM_BASE) {
+		if (!perf_regs->h16zmm_regs)
+			return 0;
+		return perf_regs->h16zmm_regs[(idx - PERF_X86_H16ZMM_BASE) *
+					PERF_X86_ZMM_QWORDS + qwords_idx];
+	}
+
 	if (qwords_idx < PERF_X86_XMM_QWORDS) {
 		if (!perf_regs->xmm_regs)
 			return 0;
@@ -105,6 +113,11 @@ u64 perf_simd_reg_value(struct pt_regs *regs, int idx,
 			return 0;
 		return perf_regs->ymmh_regs[idx * PERF_X86_YMMH_QWORDS +
 					    qwords_idx - PERF_X86_XMM_QWORDS];
+	} else if (qwords_idx < PERF_X86_ZMM_QWORDS) {
+		if (!perf_regs->zmmh_regs)
+			return 0;
+		return perf_regs->zmmh_regs[idx * PERF_X86_ZMMH_QWORDS +
+					    qwords_idx - PERF_X86_YMM_QWORDS];
 	}
 
 	return 0;
@@ -123,7 +136,8 @@ int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
 
 	if (vec_qwords) {
 		if (vec_qwords != PERF_X86_XMM_QWORDS &&
-		    vec_qwords != PERF_X86_YMM_QWORDS)
+		    vec_qwords != PERF_X86_YMM_QWORDS &&
+		    vec_qwords != PERF_X86_ZMM_QWORDS)
 			return -EINVAL;
 		if (vec_mask & ~PERF_X86_SIMD_VEC_MASK)
 			return -EINVAL;
@@ -135,6 +149,9 @@ int perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
 		if (vec_qwords == PERF_X86_YMM_QWORDS && mask &&
 		    !bitmap_full(&mask, PERF_X86_SIMD_YMM_REGS))
 			return -EINVAL;
+		if (vec_qwords == PERF_X86_ZMM_QWORDS && mask &&
+		    !bitmap_full(&mask, PERF_X86_SIMD_ZMM_REGS))
+			return -EINVAL;
 	}
 
 	/* PRED registers are not supported yet. */
-- 
2.34.1


  parent reply	other threads:[~2026-07-21  6:33 UTC|newest]

Thread overview: 26+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-07-21  6:24 [Patch v10 00/23] Support SIMD/eGPRs/SSP registers sampling for perf Dapeng Mi
2026-07-21  6:24 ` [Patch v10 01/23] perf/x86: Move hybrid PMU initialization before x86_pmu_starting_cpu() Dapeng Mi
2026-07-21  6:24 ` [Patch v10 02/23] perf/x86/intel: Enable large PEBS sampling for XMMs Dapeng Mi
2026-07-21  6:24 ` [Patch v10 03/23] perf/x86/intel: Convert x86_perf_regs to per-cpu variables Dapeng Mi
2026-07-21  6:24 ` [Patch v10 04/23] perf: Eliminate duplicate arch-specific function definitions Dapeng Mi
2026-07-21  6:24 ` [Patch v10 05/23] perf/x86: Use x86_perf_regs in NMI handlers Dapeng Mi
2026-07-21  6:24 ` [Patch v10 06/23] x86/fpu/xstate: Add xsaves_nmi() helper Dapeng Mi
2026-07-21  6:24 ` [Patch v10 07/23] x86/fpu: Add update_fpu_state_and_flag() helper Dapeng Mi
2026-07-21  6:24 ` [Patch v10 08/23] perf: Move and enhance has_extended_regs() for arch-specific use Dapeng Mi
2026-07-21  6:24 ` [Patch v10 09/23] perf/x86/intel: Centralize PERF_PMU_CAP_EXTENDED_REGS updates Dapeng Mi
2026-07-21  6:24 ` [Patch v10 10/23] perf/x86: Enable XMM register sampling for non-PEBS events Dapeng Mi
2026-07-21  6:24 ` [Patch v10 11/23] perf/x86: Enable XMM register sampling for REGS_USER case Dapeng Mi
2026-07-21  6:24 ` [Patch v10 12/23] perf: Add sampling support for SIMD registers Dapeng Mi
2026-07-21  6:24 ` [Patch v10 13/23] perf/x86: Support XMM sampling using sample_simd_vec_reg_* fields Dapeng Mi
2026-07-21  6:24 ` [Patch v10 14/23] perf/x86: Support YMM " Dapeng Mi
2026-07-21  6:24 ` Dapeng Mi [this message]
2026-07-21  6:24 ` [Patch v10 16/23] perf/x86: Support OPMASK sampling using sample_simd_pred_reg_* fields Dapeng Mi
2026-07-21  6:25 ` [Patch v10 17/23] perf: Enhance perf_reg_validate() with simd_enabled argument Dapeng Mi
2026-07-21  6:25 ` [Patch v10 18/23] perf/x86: Support eGPRs sampling using sample_regs_* fields Dapeng Mi
2026-07-21  6:25 ` [Patch v10 19/23] perf/x86: Support SSP " Dapeng Mi
2026-07-21  6:25 ` [Patch v10 20/23] perf/x86/intel: Support arch-PEBS based SIMD/eGPRs sampling Dapeng Mi
2026-07-21  6:25 ` [Patch v10 21/23] perf/x86/intel: Advertise PERF_PMU_CAP_SIMD_REGS capability Dapeng Mi
2026-07-21  6:25 ` [Patch v10 22/23] perf/x86: Activate back-to-back NMI detection for arch-PEBS induced NMIs Dapeng Mi
2026-07-21  6:25 ` [Patch v10 23/23] perf/x86/intel: Add sanity check for PEBS record/fragment size Dapeng Mi
2026-07-21 17:42 ` [Patch v10 00/23] Support SIMD/eGPRs/SSP registers sampling for perf Ian Rogers
2026-07-22  0:17   ` Mi, Dapeng

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260721062506.3745816-16-dapeng1.mi@linux.intel.com \
    --to=dapeng1.mi@linux.intel.com \
    --cc=acme@kernel.org \
    --cc=adrian.hunter@intel.com \
    --cc=ak@linux.intel.com \
    --cc=alexander.shishkin@linux.intel.com \
    --cc=broonie@kernel.org \
    --cc=dapeng1.mi@intel.com \
    --cc=dave.hansen@linux.intel.com \
    --cc=eranian@google.com \
    --cc=irogers@google.com \
    --cc=jolsa@kernel.org \
    --cc=kan.liang@linux.intel.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-perf-users@vger.kernel.org \
    --cc=mark.rutland@arm.com \
    --cc=mingo@redhat.com \
    --cc=namhyung@kernel.org \
    --cc=peterz@infradead.org \
    --cc=ravi.bangoria@amd.com \
    --cc=tglx@linutronix.de \
    --cc=thomas.falcon@intel.com \
    --cc=xudong.hao@intel.com \
    --cc=zide.chen@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.