Linux Documentation
 help / color / mirror / Atom feed
* [RFC PATCH v2 04/13] HWBP: Add modify_wide_hw_breakpoint_local() API
From: Jinchao Wang @ 2026-07-17 13:03 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

From: "Masami Hiramatsu (Google)" <mhiramat@kernel.org>

Add modify_wide_hw_breakpoint_local() arch-wide interface which allows
hwbp users to update watch address on-line. This is available if the
arch supports CONFIG_HAVE_REINSTALL_HW_BREAKPOINT.
Note that this allows to change the type only for compatible types,
because it does not release and reserve the hwbp slot based on type.
For instance, you can not change HW_BREAKPOINT_W to HW_BREAKPOINT_X.

Assisted-by: Antigravity:gemini-3.5-flash
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 include/linux/hw_breakpoint.h |  6 +++++
 kernel/events/hw_breakpoint.c | 43 +++++++++++++++++++++++++++++++++++
 2 files changed, 49 insertions(+)

diff --git a/include/linux/hw_breakpoint.h b/include/linux/hw_breakpoint.h
index db199d653dd1..6754ffbee9ed 100644
--- a/include/linux/hw_breakpoint.h
+++ b/include/linux/hw_breakpoint.h
@@ -81,6 +81,9 @@ register_wide_hw_breakpoint(struct perf_event_attr *attr,
 			    perf_overflow_handler_t triggered,
 			    void *context);
 
+extern int modify_wide_hw_breakpoint_local(struct perf_event *bp,
+					   struct perf_event_attr *attr);
+
 extern int register_perf_hw_breakpoint(struct perf_event *bp);
 extern void unregister_hw_breakpoint(struct perf_event *bp);
 extern void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events);
@@ -124,6 +127,9 @@ register_wide_hw_breakpoint(struct perf_event_attr *attr,
 			    perf_overflow_handler_t triggered,
 			    void *context)		{ return NULL; }
 static inline int
+modify_wide_hw_breakpoint_local(struct perf_event *bp,
+				struct perf_event_attr *attr) { return -EOPNOTSUPP; }
+static inline int
 register_perf_hw_breakpoint(struct perf_event *bp)	{ return -ENOSYS; }
 static inline void unregister_hw_breakpoint(struct perf_event *bp)	{ }
 static inline void
diff --git a/kernel/events/hw_breakpoint.c b/kernel/events/hw_breakpoint.c
index 789add0c185a..f4709c892d67 100644
--- a/kernel/events/hw_breakpoint.c
+++ b/kernel/events/hw_breakpoint.c
@@ -888,6 +888,49 @@ void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events)
 }
 EXPORT_SYMBOL_GPL(unregister_wide_hw_breakpoint);
 
+/**
+ * modify_wide_hw_breakpoint_local - update breakpoint config for local CPU
+ * @bp: the hwbp perf event for this CPU
+ * @attr: the new attribute for @bp
+ *
+ * This does not release and reserve the slot of a HWBP; it just reuses the
+ * current slot on local CPU. So the users must update the other CPUs by
+ * themselves.
+ * Also, since this does not release/reserve the slot, this can not change the
+ * type to incompatible type of the HWBP.
+ * Return err if attr is invalid or the CPU fails to update debug register
+ * for new @attr.
+ */
+#ifdef CONFIG_HAVE_REINSTALL_HW_BREAKPOINT
+int modify_wide_hw_breakpoint_local(struct perf_event *bp,
+				    struct perf_event_attr *attr)
+{
+	struct arch_hw_breakpoint info;
+	int ret;
+
+	if (find_slot_idx(bp->attr.bp_type) != find_slot_idx(attr->bp_type))
+		return -EINVAL;
+
+	ret = hw_breakpoint_arch_parse(bp, attr, &info);
+	if (ret)
+		return ret;
+
+	*counter_arch_bp(bp) = info;
+	bp->attr.bp_addr = attr->bp_addr;
+	bp->attr.bp_type = attr->bp_type;
+	bp->attr.bp_len = attr->bp_len;
+
+	return arch_reinstall_hw_breakpoint(bp);
+}
+#else
+int modify_wide_hw_breakpoint_local(struct perf_event *bp,
+				    struct perf_event_attr *attr)
+{
+	return -EOPNOTSUPP;
+}
+#endif
+EXPORT_SYMBOL_GPL(modify_wide_hw_breakpoint_local);
+
 /**
  * hw_breakpoint_is_used - check if breakpoints are currently used
  *
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 05/13] mm/kwatch: add watch expression parser and dereference engine
From: Jinchao Wang @ 2026-07-17 13:04 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

KWatch watches a memory address that is only known once the target
function runs, e.g. "argument 1, plus 8, dereferenced once". Add the
two halves of that mechanism:

- kwatch_deref_parse() turns a textual watch expression
  {base}[+-off][->[+-]off]... into a kwatch_config: a base anchor
  (arg1..arg6, stack, an absolute address or - for built-in KWatch -
  a symbol name) plus a static offset chain.

- kwatch_deref_resolve() replays the chain at probe time against
  pt_regs. Every pointer load goes through get_kernel_nofault() and
  the final address must be a kernel address.

Also add the internal kwatch.h header shared by the rest of the
series. Nothing is built yet; the Kconfig entry comes with the
control plane.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 mm/kwatch/Makefile |   3 +
 mm/kwatch/deref.c  | 174 +++++++++++++++++++++++++++++++++++++++++++++
 mm/kwatch/kwatch.h | 101 ++++++++++++++++++++++++++
 3 files changed, 278 insertions(+)
 create mode 100644 mm/kwatch/Makefile
 create mode 100644 mm/kwatch/deref.c
 create mode 100644 mm/kwatch/kwatch.h

diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
new file mode 100644
index 000000000000..69c21ae62123
--- /dev/null
+++ b/mm/kwatch/Makefile
@@ -0,0 +1,3 @@
+obj-$(CONFIG_KWATCH) += kwatch.o
+
+kwatch-y := deref.o
diff --git a/mm/kwatch/deref.c b/mm/kwatch/deref.c
new file mode 100644
index 000000000000..a93c76139e7c
--- /dev/null
+++ b/mm/kwatch/deref.c
@@ -0,0 +1,174 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/ptrace.h>
+#include <linux/sched.h>
+#include <linux/uaccess.h>
+#include <linux/kallsyms.h>
+#include <linux/string.h>
+#include <linux/slab.h>
+
+#include "kwatch.h"
+
+int kwatch_deref_resolve(const struct kwatch_config *cfg, struct pt_regs *regs,
+			 unsigned long *out_addr, u16 *out_len)
+{
+	unsigned long addr = 0;
+	int i;
+
+	/* 1. Resolve the Base Anchor */
+	if (cfg->base == KWATCH_BASE_STACK) {
+		addr = kernel_stack_pointer(regs);
+		if (unlikely(!addr))
+			return -EINVAL;
+	} else if (cfg->base >= KWATCH_BASE_ARG1 &&
+		   cfg->base <= KWATCH_BASE_ARG6) {
+		int arg_idx = cfg->base - KWATCH_BASE_ARG1;
+
+		addr = regs_get_kernel_argument(regs, arg_idx);
+	} else if (cfg->base == KWATCH_BASE_ABS_ADDR ||
+		   cfg->base == KWATCH_BASE_GLOBAL_SYM) {
+		/* Zero-latency load of the static symbol location */
+		addr = cfg->sym_addr;
+	} else {
+		return -EINVAL;
+	}
+
+	/* 2. The Pointer-Chasing FSM */
+	for (i = 0; i < cfg->offset_count; i++) {
+		addr += cfg->offsets[i];
+
+		if (i < cfg->offset_count - 1) {
+			unsigned long next_addr;
+
+			/* Dynamically read the pointer contents at runtime */
+			if (get_kernel_nofault(next_addr, (unsigned long *)addr))
+				return -EFAULT;
+
+			addr = next_addr;
+		}
+	}
+
+	/* Enforce strict Kernel-Space boundary */
+	if (unlikely(addr < TASK_SIZE_MAX))
+		return -EINVAL;
+
+	*out_addr = addr;
+	*out_len = cfg->watch_len;
+	return 0;
+}
+
+int kwatch_deref_parse(struct kwatch_config *cfg, const char *watch_expr)
+{
+	char *p, *sep, *dup_expr;
+	char type = '\0';
+	bool is_deref = false;
+	int ret = 0;
+
+	dup_expr = kstrdup(watch_expr, GFP_KERNEL);
+	if (!dup_expr)
+		return -ENOMEM;
+
+	cfg->offset_count = 1;
+	cfg->offsets[0] = 0;
+
+	/* 1. Isolate and Resolve Base Anchor */
+	p = dup_expr;
+	sep = NULL;
+	while (*p) {
+		if (*p == '+') {
+			sep = p;
+			type = '+';
+			break;
+		}
+		if (*p == '-') {
+			sep = p;
+			type = '-';
+			if (p[1] == '>')
+				is_deref = true;
+			break;
+		}
+		p++;
+	}
+
+	if (type)
+		*sep = '\0';
+
+	if (!strcmp(dup_expr, "stack")) {
+		cfg->base = KWATCH_BASE_STACK;
+	} else if (!strncmp(dup_expr, "arg", 3) && strlen(dup_expr) == 4) {
+		int arg_num;
+
+		if (kstrtoint(dup_expr + 3, 10, &arg_num) || arg_num < 1 ||
+		    arg_num > 6) {
+			ret = -EINVAL;
+			goto out;
+		}
+		cfg->base = KWATCH_BASE_ARG1 + (arg_num - 1);
+	} else if (kstrtoul(dup_expr, 0, &cfg->sym_addr) == 0) {
+		cfg->base = KWATCH_BASE_ABS_ADDR;
+	} else {
+#if IS_BUILTIN(CONFIG_KWATCH)
+		cfg->sym_addr = kallsyms_lookup_name(dup_expr);
+		if (!cfg->sym_addr) {
+			pr_err("Failed to resolve symbol name: %s\n", dup_expr);
+			ret = -EINVAL;
+			goto out;
+		}
+		cfg->base = KWATCH_BASE_GLOBAL_SYM;
+#else
+		pr_err("cannot resolve symbol %s when built as a module, use a hex address\n",
+		       dup_expr);
+		ret = -EINVAL;
+		goto out;
+#endif
+	}
+
+	if (!type)
+		goto out;
+
+	/* 2. Resolve Base Offset (if + or - exists) */
+	if (!is_deref) {
+		char *next;
+
+		*sep = type; /* Restore the '+' or '-' for kstrtol */
+		next = strstr(sep, "->");
+		if (next)
+			*next = '\0';
+
+		if (kstrtol(sep, 0, &cfg->offsets[0])) {
+			ret = -EINVAL;
+			goto out;
+		}
+
+		p = next ? next + 2 : NULL;
+	} else {
+		/* Jump directly to the first dereference after '->' */
+		p = sep + 2;
+	}
+
+	/* 3. Resolve Dereference Chain */
+	while (p) {
+		char *next;
+
+		if (cfg->offset_count >= MAX_DEREF_CHAIN) {
+			ret = -E2BIG;
+			goto out;
+		}
+
+		next = strstr(p, "->");
+		if (next)
+			*next = '\0';
+
+		if (kstrtol(p, 0, &cfg->offsets[cfg->offset_count++])) {
+			ret = -EINVAL;
+			goto out;
+		}
+
+		p = next ? next + 2 : NULL;
+	}
+
+out:
+	kfree(dup_expr);
+	return ret;
+}
diff --git a/mm/kwatch/kwatch.h b/mm/kwatch/kwatch.h
new file mode 100644
index 000000000000..dbe0fd0e6a0d
--- /dev/null
+++ b/mm/kwatch/kwatch.h
@@ -0,0 +1,101 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef _MM_KWATCH_H
+#define _MM_KWATCH_H
+
+#include <linux/fprobe.h>
+#include <linux/kprobes.h>
+#include <linux/perf_event.h>
+#include <linux/sched.h>
+#include <linux/types.h>
+#include <linux/compiler.h>
+#include <linux/atomic.h>
+
+#define MAX_CONFIG_STR_LEN 512
+#define MAX_DEREF_CHAIN 4
+
+struct kwatch_watchpoint;
+
+struct kwatch_tsk_ctx {
+	struct task_struct *task;
+	struct kwatch_watchpoint *wp;
+	u16 depth;
+	u32 epoch;
+};
+
+struct kwatch_watchpoint {
+	struct perf_event *__percpu *event;
+	call_single_data_t __percpu *csd_arm;
+	call_single_data_t __percpu *csd_disarm;
+	struct perf_event_attr attr;
+	atomic_t in_use; // multi-consumer safe get/put
+	struct list_head list; // for cpu online and offline
+
+	struct task_struct *arm_tsk;
+	atomic_t pending_ipis;
+	atomic_t refcount;
+	bool teardown;
+};
+
+enum kwatch_base_type {
+	KWATCH_BASE_STACK,
+	KWATCH_BASE_ABS_ADDR,
+	KWATCH_BASE_GLOBAL_SYM,
+	KWATCH_BASE_ARG1,
+	KWATCH_BASE_ARG2,
+	KWATCH_BASE_ARG3,
+	KWATCH_BASE_ARG4,
+	KWATCH_BASE_ARG5,
+	KWATCH_BASE_ARG6,
+};
+
+struct kwatch_config {
+	u16 max_watch;
+	char func_name[KSYM_NAME_LEN];
+	u16 func_offset;
+	u16 depth;
+	u16 duration;
+	u16 watch_len;
+
+	/* Unified Deref Engine State */
+	enum kwatch_base_type base;
+	char watch_expr[MAX_CONFIG_STR_LEN];
+	unsigned long sym_addr;
+	long offsets[MAX_DEREF_CHAIN];
+	u8 offset_count;
+	u16 max_concurrency;
+};
+
+int kwatch_hwbp_prealloc(u16 max_watch);
+void kwatch_hwbp_free(void);
+int kwatch_hwbp_get(struct kwatch_watchpoint **out_wp);
+void kwatch_hwbp_arm(struct kwatch_watchpoint *wp, unsigned long addr, u16 len);
+int kwatch_hwbp_put(struct kwatch_watchpoint *wp);
+
+int kwatch_probe_start(struct kwatch_config *cfg);
+void kwatch_probe_stop(void);
+void kwatch_probe_mute(bool mute);
+bool kwatch_probe_validate_hit(struct pt_regs *regs, struct task_struct *arm_tsk);
+unsigned long kwatch_probe_nmi_rejected(void);
+unsigned long kwatch_hwbp_arm_ipi_suppressed(void);
+
+int kwatch_tsk_ctx_prealloc(u16 max_concurrency);
+struct kwatch_tsk_ctx *kwatch_tsk_ctx_get(bool can_alloc);
+void kwatch_tsk_ctx_put(void);
+void kwatch_tsk_ctx_release(struct kwatch_tsk_ctx *ctx);
+void kwatch_tsk_ctx_reset(struct kwatch_tsk_ctx *ctx, u32 new_epoch);
+void kwatch_tsk_ctx_release_wps(void);
+void kwatch_tsk_ctx_free(void);
+
+void kwatch_global_anchor(unsigned long duration_sec);
+int kwatch_anchor_start(u16 duration);
+void kwatch_anchor_stop(void);
+void kwatch_anchor_cancel_work(void);
+bool kwatch_anchor_has_expired(void);
+void kwatch_anchor_clear_expired(void);
+void kwatch_auto_stop(void);
+
+int kwatch_deref_resolve(const struct kwatch_config *cfg, struct pt_regs *regs,
+			 unsigned long *out_addr, u16 *out_len);
+int kwatch_deref_parse(struct kwatch_config *cfg, const char *watch_expr);
+
+#endif /* _MM_KWATCH_H */
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 06/13] mm/kwatch: add lockless per-task context pool
From: Jinchao Wang @ 2026-07-17 13:04 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

A task that enters the watched function needs somewhere to keep its
window state (nesting depth, owned watchpoint, config epoch). The
lookup runs in kprobe and NMI-like contexts, so it must not allocate
or take locks.

Use a preallocated open-addressing array hashed by task_struct
pointer. Slots are claimed with cmpxchg() and released with
smp_store_release(); lookup is a read-only probe sequence. The pool
size (max_concurrency) bounds how many tasks can be inside watch
windows concurrently; excess tasks are simply not tracked.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 mm/kwatch/Makefile   |   2 +-
 mm/kwatch/task_ctx.c | 125 +++++++++++++++++++++++++++++++++++++++++++
 2 files changed, 126 insertions(+), 1 deletion(-)
 create mode 100644 mm/kwatch/task_ctx.c

diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index 69c21ae62123..cc6574df0d68 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
 obj-$(CONFIG_KWATCH) += kwatch.o
 
-kwatch-y := deref.o
+kwatch-y := deref.o task_ctx.o
diff --git a/mm/kwatch/task_ctx.c b/mm/kwatch/task_ctx.c
new file mode 100644
index 000000000000..64383a4429e7
--- /dev/null
+++ b/mm/kwatch/task_ctx.c
@@ -0,0 +1,125 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/slab.h>
+#include <linux/hash.h>
+#include <linux/sched.h>
+#include <linux/log2.h>
+#include "kwatch.h"
+
+static u16 kwatch_ctx_pool_size;
+static u16 kwatch_ctx_pool_mask;
+
+static struct kwatch_tsk_ctx *kwatch_ctx_pool;
+
+/* Pool size is a u16 and indexes a power-of-two hash table, so bound the
+ * request away from both the roundup_pow_of_two() u16 overflow (>32768 wraps
+ * to 0) and the degenerate size-1 case (ilog2(1) == 0 breaks hash_ptr()).
+ */
+#define KWATCH_CTX_POOL_MIN	256
+#define KWATCH_CTX_POOL_MAX	32768
+
+int kwatch_tsk_ctx_prealloc(u16 max_concurrency)
+{
+	if (max_concurrency < KWATCH_CTX_POOL_MIN)
+		max_concurrency = KWATCH_CTX_POOL_MIN;
+	else if (max_concurrency > KWATCH_CTX_POOL_MAX)
+		max_concurrency = KWATCH_CTX_POOL_MAX;
+
+	/*
+	 * Set the size/mask only when actually allocating, so they can never
+	 * drift out of sync with the live pool if prealloc is ever called
+	 * again without a matching free.
+	 */
+	if (unlikely(!kwatch_ctx_pool)) {
+		kwatch_ctx_pool_size = roundup_pow_of_two(max_concurrency);
+		kwatch_ctx_pool_mask = kwatch_ctx_pool_size - 1;
+
+		kwatch_ctx_pool = kcalloc(kwatch_ctx_pool_size,
+					  sizeof(struct kwatch_tsk_ctx),
+					  GFP_KERNEL);
+		if (!kwatch_ctx_pool)
+			return -ENOMEM;
+	}
+	return 0;
+}
+
+struct kwatch_tsk_ctx *kwatch_tsk_ctx_get(bool can_alloc)
+{
+	int start_idx, i, idx;
+	struct task_struct *t;
+
+	if (unlikely(!kwatch_ctx_pool))
+		return NULL;
+
+	start_idx = hash_ptr(current, ilog2(kwatch_ctx_pool_size));
+
+	for (i = 0; i < kwatch_ctx_pool_size; i++) {
+		idx = (start_idx + i) & kwatch_ctx_pool_mask;
+		t = READ_ONCE(kwatch_ctx_pool[idx].task);
+		if (t == current)
+			return &kwatch_ctx_pool[idx];
+	}
+
+	if (!can_alloc)
+		return NULL;
+
+	for (i = 0; i < kwatch_ctx_pool_size; i++) {
+		idx = (start_idx + i) & kwatch_ctx_pool_mask;
+		t = READ_ONCE(kwatch_ctx_pool[idx].task);
+		if (!t) {
+			if (!cmpxchg(&kwatch_ctx_pool[idx].task, NULL, current))
+				return &kwatch_ctx_pool[idx];
+		}
+	}
+
+	return NULL;
+}
+
+void kwatch_tsk_ctx_reset(struct kwatch_tsk_ctx *ctx, u32 new_epoch)
+{
+	struct kwatch_watchpoint *wp = xchg(&ctx->wp, NULL);
+
+	if (wp)
+		kwatch_hwbp_put(wp);
+	ctx->depth = 0;
+	ctx->epoch = new_epoch;
+}
+
+/* Release a slot we hold a pointer to: disarm its wp and free the slot. */
+void kwatch_tsk_ctx_release(struct kwatch_tsk_ctx *ctx)
+{
+	kwatch_tsk_ctx_reset(ctx, 0);
+
+	/* Pairs with READ_ONCE() in kwatch_tsk_ctx_get() */
+	smp_store_release(&ctx->task, NULL);
+}
+
+void kwatch_tsk_ctx_put(void)
+{
+	struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+
+	if (unlikely(!ctx))
+		return;
+
+	kwatch_tsk_ctx_release(ctx);
+}
+
+void kwatch_tsk_ctx_release_wps(void)
+{
+	int i;
+
+	if (!kwatch_ctx_pool)
+		return;
+
+	for (i = 0; i < kwatch_ctx_pool_size; i++) {
+		struct kwatch_watchpoint *wp = xchg(&kwatch_ctx_pool[i].wp,
+						    NULL);
+		if (wp)
+			kwatch_hwbp_put(wp);
+	}
+}
+
+void kwatch_tsk_ctx_free(void)
+{
+	kfree(kwatch_ctx_pool);
+	kwatch_ctx_pool = NULL;
+}
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 07/13] stacktrace: export stack_trace_save_regs()
From: Jinchao Wang @ 2026-07-17 13:04 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

The other stack_trace_save_*() flavours are exported, but the regs
variant is not, so no module can capture a stack trace for a given
pt_regs. KWatch, which may be built as a module, uses it to record
who wrote to a watched address from the hardware breakpoint handler.
Export it like its siblings.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 kernel/stacktrace.c | 2 ++
 1 file changed, 2 insertions(+)

diff --git a/kernel/stacktrace.c b/kernel/stacktrace.c
index afb3c116da91..d853c40f916b 100644
--- a/kernel/stacktrace.c
+++ b/kernel/stacktrace.c
@@ -175,6 +175,7 @@ unsigned int stack_trace_save_regs(struct pt_regs *regs, unsigned long *store,
 	arch_stack_walk(consume_entry, &c, current, regs);
 	return c.len;
 }
+EXPORT_SYMBOL_GPL(stack_trace_save_regs);
 
 #ifdef CONFIG_HAVE_RELIABLE_STACKTRACE
 /**
@@ -325,6 +326,7 @@ unsigned int stack_trace_save_regs(struct pt_regs *regs, unsigned long *store,
 	save_stack_trace_regs(regs, &trace);
 	return trace.nr_entries;
 }
+EXPORT_SYMBOL_GPL(stack_trace_save_regs);
 
 #ifdef CONFIG_HAVE_RELIABLE_STACKTRACE
 /**
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 08/13] mm/kwatch: add hardware breakpoint backend
From: Jinchao Wang @ 2026-07-17 13:05 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

Manage a preallocated pool of wide (per-CPU) perf hardware
breakpoints. All breakpoints are registered up front against a dummy
address; arming a watchpoint only re-points an already-registered
event, so the arm path can run from a kprobe handler.

- kwatch_hwbp_get()/put() claim and release pool entries with
  per-slot cmpxchg, safe for concurrent consumers on any CPU.
- kwatch_hwbp_arm() updates the local CPU synchronously via
  modify_wide_hw_breakpoint_local() and broadcasts asynchronous IPIs
  to the other CPUs. Arm-side IPIs are rate-limited per CPU; disarm
  IPIs are refcounted so an entry is only recycled once every CPU
  has dropped it.
- Hits are reported through the kwatch:kwatch_hit tracepoint with a
  short stack trace: the ftrace ring buffer is usable from NMI-like
  context and survives a subsequent crash, unlike printk.
- A CPU hotplug callback creates/destroys the per-CPU events as CPUs
  come and go.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 include/trace/events/kwatch.h |  68 ++++++
 mm/kwatch/Makefile            |   2 +-
 mm/kwatch/hwbp.c              | 388 ++++++++++++++++++++++++++++++++++
 3 files changed, 457 insertions(+), 1 deletion(-)
 create mode 100644 include/trace/events/kwatch.h
 create mode 100644 mm/kwatch/hwbp.c

diff --git a/include/trace/events/kwatch.h b/include/trace/events/kwatch.h
new file mode 100644
index 000000000000..8a2ec6811ad4
--- /dev/null
+++ b/include/trace/events/kwatch.h
@@ -0,0 +1,68 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#undef TRACE_SYSTEM
+#define TRACE_SYSTEM kwatch
+
+#if !defined(_TRACE_KWATCH_H) || defined(TRACE_HEADER_MULTI_READ)
+#define _TRACE_KWATCH_H
+
+#include <linux/tracepoint.h>
+#include <linux/ptrace.h>
+#include <linux/math64.h>
+
+#define KWATCH_STACK_DEPTH 8
+
+struct trace_seq;
+const char *kwatch_trace_print_stack(struct trace_seq *p,
+				     const unsigned long *stack,
+				     unsigned int nr);
+
+TRACE_EVENT(kwatch_hit,
+	TP_PROTO(unsigned long ip, unsigned long sp, unsigned long addr,
+		 u64 time_ns,
+		 unsigned long *stack_entries, unsigned int stack_nr),
+	TP_ARGS(ip, sp, addr, time_ns, stack_entries, stack_nr),
+
+	TP_STRUCT__entry(
+		/*
+		 * time_ns first: u64 leading the entry avoids a 4-byte hole
+		 * after the unsigned-long fields on 32-bit. stack_nr trails
+		 * the fixed fields for the same reason; the stack is a
+		 * dynamic array sized to what was actually captured, so a
+		 * short trace neither wastes space nor leaks uninitialized
+		 * tail slots.
+		 */
+		__field(u64, time_ns)
+		__field(unsigned long, ip)
+		__field(unsigned long, sp)
+		__field(unsigned long, addr)
+		__dynamic_array(unsigned long, stack,
+				min_t(unsigned int, stack_nr, KWATCH_STACK_DEPTH))
+		__field(unsigned int, stack_nr)
+	),
+
+	TP_fast_assign(
+		unsigned long *stack = __get_dynamic_array(stack);
+		unsigned int i;
+
+		__entry->time_ns = time_ns;
+		__entry->ip = ip;
+		__entry->sp = sp;
+		__entry->addr = addr;
+		__entry->stack_nr = min_t(unsigned int, stack_nr,
+					  KWATCH_STACK_DEPTH);
+		for (i = 0; i < __entry->stack_nr; i++)
+			stack[i] = stack_entries[i];
+	),
+
+	TP_printk("KWatch HIT: time=%llu.%06u ip=%pS addr=0x%lx%s",
+		  div_u64(__entry->time_ns, 1000000000ULL),
+		  (unsigned int)(div_u64(__entry->time_ns, 1000ULL) % 1000000ULL),
+		  (void *)__entry->ip, __entry->addr,
+		  kwatch_trace_print_stack(p, __get_dynamic_array(stack),
+					   __entry->stack_nr))
+);
+
+#endif /* _TRACE_KWATCH_H */
+
+/* This part must be outside protection */
+#include <trace/define_trace.h>
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index cc6574df0d68..b2bc3003c89b 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
 obj-$(CONFIG_KWATCH) += kwatch.o
 
-kwatch-y := deref.o task_ctx.o
+kwatch-y := deref.o task_ctx.o hwbp.o
diff --git a/mm/kwatch/hwbp.c b/mm/kwatch/hwbp.c
new file mode 100644
index 000000000000..d1e93754cce8
--- /dev/null
+++ b/mm/kwatch/hwbp.c
@@ -0,0 +1,388 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/cpuhotplug.h>
+#include <linux/ftrace.h>
+#include <linux/hw_breakpoint.h>
+#include <linux/sched/clock.h>
+#include <linux/irqflags.h>
+#include <linux/kallsyms.h>
+#include <linux/mutex.h>
+#include <linux/printk.h>
+#include <linux/slab.h>
+#include <linux/stacktrace.h>
+#include <linux/trace_seq.h>
+#include <linux/workqueue.h>
+
+#include "kwatch.h"
+
+/* Minimum spacing between cross-CPU arm broadcasts, per CPU. */
+#define KWATCH_ARM_IPI_MIN_INTERVAL_NS	1000000ULL
+
+static LIST_HEAD(kwatch_all_wp_list);
+static struct kwatch_watchpoint **kwatch_wp_slots;
+static u16 kwatch_wp_nr;
+static DEFINE_MUTEX(kwatch_all_wp_mutex);
+static unsigned long kwatch_dummy_holder __aligned(8);
+static int kwatch_hwbp_cpuhp_state = CPUHP_INVALID;
+static atomic_long_t kwatch_arm_ipi_suppressed;
+
+unsigned long kwatch_hwbp_arm_ipi_suppressed(void)
+{
+	return atomic_long_read(&kwatch_arm_ipi_suppressed);
+}
+
+#define CREATE_TRACE_POINTS
+#include <trace/events/kwatch.h>
+
+/*
+ * Render the saved stack like the ftrace built-in stacktrace / dump_stack()
+ * style. Symbol resolution runs at trace read time, not in the hit path.
+ */
+const char *kwatch_trace_print_stack(struct trace_seq *p,
+				     const unsigned long *stack,
+				     unsigned int nr)
+{
+	const char *ret = trace_seq_buffer_ptr(p);
+	unsigned int i;
+
+	for (i = 0; i < nr; i++)
+		trace_seq_printf(p, "\n => %pS", (void *)stack[i]);
+	trace_seq_putc(p, 0);
+	return ret;
+}
+
+static void kwatch_hwbp_handler(struct perf_event *bp,
+				struct perf_sample_data *data,
+				struct pt_regs *regs)
+{
+	struct kwatch_watchpoint *wp = bp->overflow_handler_context;
+	unsigned long stack_entries[KWATCH_STACK_DEPTH];
+	unsigned int stack_nr;
+
+	if (!kwatch_probe_validate_hit(regs, wp->arm_tsk))
+		return;
+
+	stack_nr = stack_trace_save_regs(regs, stack_entries, KWATCH_STACK_DEPTH, 2);
+	trace_kwatch_hit(instruction_pointer(regs), kernel_stack_pointer(regs),
+			 bp->attr.bp_addr, local_clock(),
+			 stack_entries, stack_nr);
+}
+
+static void kwatch_hwbp_arm_local(void *info)
+{
+	struct kwatch_watchpoint *wp = info;
+	struct perf_event *bp;
+	unsigned long flags;
+	int cpu, err;
+
+	local_irq_save(flags);
+
+	cpu = smp_processor_id();
+	bp = per_cpu(*wp->event, cpu);
+
+	if (unlikely(!bp))
+		goto out;
+
+	kwatch_probe_mute(true);
+	barrier();
+
+	/*
+	 * On success this also updates the per-CPU bp->attr, so the hit
+	 * handler reports what THIS CPU is watching instead of the shared
+	 * wp->attr, which another CPU may be re-pointing.
+	 */
+	err = modify_wide_hw_breakpoint_local(bp, &wp->attr);
+	if (unlikely(err))
+		WARN_ONCE(1,
+			  "KWatch: HWBP reinstall failed on CPU%d (err=%d, addr=0x%llx, len=%llu)\n",
+			  cpu, err, wp->attr.bp_addr, wp->attr.bp_len);
+
+	barrier();
+	kwatch_probe_mute(false);
+
+out:
+	local_irq_restore(flags);
+}
+
+static inline void kwatch_hwbp_try_recycle(struct kwatch_watchpoint *wp)
+{
+	if (atomic_dec_and_test(&wp->pending_ipis)) {
+		if (!READ_ONCE(wp->teardown))
+			atomic_set_release(&wp->in_use, 0);
+
+		atomic_dec(&wp->refcount);
+	}
+}
+
+static void kwatch_hwbp_disarm_local(void *info)
+{
+	struct kwatch_watchpoint *wp = info;
+
+	kwatch_hwbp_arm_local(info);
+	kwatch_hwbp_try_recycle(wp);
+}
+
+static int kwatch_hwbp_cpu_online(unsigned int cpu)
+{
+	struct perf_event_attr attr;
+	struct kwatch_watchpoint *wp;
+	struct perf_event *bp;
+
+	mutex_lock(&kwatch_all_wp_mutex);
+	list_for_each_entry(wp, &kwatch_all_wp_list, list) {
+		attr = wp->attr;
+		attr.bp_addr = (unsigned long)&kwatch_dummy_holder;
+		bp = perf_event_create_kernel_counter(&attr, cpu, NULL,
+						      kwatch_hwbp_handler, wp);
+		if (IS_ERR(bp)) {
+			pr_warn("%s failed to create watch on CPU %d: %ld\n",
+				__func__, cpu, PTR_ERR(bp));
+			continue;
+		}
+		per_cpu(*wp->event, cpu) = bp;
+	}
+	mutex_unlock(&kwatch_all_wp_mutex);
+	return 0;
+}
+
+static int kwatch_hwbp_cpu_offline(unsigned int cpu)
+{
+	struct kwatch_watchpoint *wp;
+	struct perf_event *bp;
+
+	mutex_lock(&kwatch_all_wp_mutex);
+	list_for_each_entry(wp, &kwatch_all_wp_list, list) {
+		bp = per_cpu(*wp->event, cpu);
+		if (bp) {
+			unregister_hw_breakpoint(bp);
+			per_cpu(*wp->event, cpu) = NULL;
+		}
+	}
+	mutex_unlock(&kwatch_all_wp_mutex);
+	return 0;
+}
+
+int kwatch_hwbp_get(struct kwatch_watchpoint **out_wp)
+{
+	struct kwatch_watchpoint *wp;
+	int i;
+
+	/*
+	 * Per-slot cmpxchg claim: safe for concurrent consumers on any CPU,
+	 * unlike llist_del_first() which requires a single consumer.
+	 */
+	for (i = 0; i < kwatch_wp_nr; i++) {
+		wp = kwatch_wp_slots[i];
+		if (atomic_read(&wp->in_use))
+			continue;
+		if (atomic_cmpxchg(&wp->in_use, 0, 1) == 0) {
+			atomic_inc(&wp->refcount);
+			*out_wp = wp;
+			return 0;
+		}
+	}
+	return -EBUSY;
+}
+
+void kwatch_hwbp_arm(struct kwatch_watchpoint *wp, unsigned long addr, u16 len)
+{
+	static DEFINE_PER_CPU(u64, last_ipi_time);
+	int cur_cpu;
+	call_single_data_t *csd;
+	int cpu;
+	bool is_disarm = (addr == (unsigned long)&kwatch_dummy_holder);
+	bool skip_remote = false;
+
+	wp->attr.bp_addr = addr;
+	wp->attr.bp_len = len;
+
+	if (!is_disarm)
+		wp->arm_tsk = current;
+
+	/* ensure attr update visible to other cpu before sending IPI */
+	smp_wmb();
+
+	atomic_set(&wp->pending_ipis, 1);
+	cur_cpu = get_cpu();
+
+	/*
+	 * Rate-limit only the cross-CPU broadcast, never the local re-point.
+	 * Arming the current CPU is free and must always reflect this window;
+	 * only the remote IPI fan-out is throttled to keep a hot function from
+	 * storming every CPU. A suppressed broadcast means remote CPUs keep
+	 * watching the previous address for that window (a missed remote-CPU
+	 * writer is possible) - hence the visible counter, and why kwatch
+	 * targets low-frequency functions. Disarm is never throttled: the
+	 * slot must always be released.
+	 */
+	if (!is_disarm) {
+		u64 now = local_clock();
+		u64 last = this_cpu_read(last_ipi_time);
+
+		if (now - last < KWATCH_ARM_IPI_MIN_INTERVAL_NS) {
+			atomic_long_inc(&kwatch_arm_ipi_suppressed);
+			skip_remote = true;
+		} else {
+			this_cpu_write(last_ipi_time, now);
+		}
+	}
+
+	if (!skip_remote) {
+		for_each_online_cpu(cpu) {
+			if (cpu == cur_cpu)
+				continue;
+
+			if (is_disarm)
+				atomic_inc(&wp->pending_ipis);
+
+			csd = per_cpu_ptr(is_disarm ? wp->csd_disarm : wp->csd_arm,
+					  cpu);
+			/*
+			 * The arm path ignores a -EBUSY return: a wp has a single
+			 * owner (claimed via kwatch_hwbp_get(), held until exit)
+			 * and is armed once per window, and the per-CPU csd queue
+			 * is FIFO, so this window's csd_arm cannot still be pending
+			 * from a prior window (its disarm, queued later, gates the
+			 * wp's reuse). Do not "fix" this into a retry.
+			 */
+			if (smp_call_function_single_async(cpu, csd) && is_disarm)
+				kwatch_hwbp_try_recycle(wp);
+		}
+	}
+	put_cpu();
+
+	if (is_disarm)
+		kwatch_hwbp_disarm_local(wp);
+	else
+		kwatch_hwbp_arm_local(wp);
+}
+
+int kwatch_hwbp_put(struct kwatch_watchpoint *wp)
+{
+	kwatch_hwbp_arm(wp, (unsigned long)&kwatch_dummy_holder,
+			sizeof(unsigned long));
+
+	return 0;
+}
+
+void kwatch_hwbp_free(void)
+{
+	struct kwatch_watchpoint *wp, *tmp;
+
+	kwatch_wp_nr = 0;
+	kfree(kwatch_wp_slots);
+	kwatch_wp_slots = NULL;
+
+	if (kwatch_hwbp_cpuhp_state != CPUHP_INVALID) {
+		cpuhp_remove_state_nocalls(kwatch_hwbp_cpuhp_state);
+		kwatch_hwbp_cpuhp_state = CPUHP_INVALID;
+	}
+
+	mutex_lock(&kwatch_all_wp_mutex);
+	list_for_each_entry_safe(wp, tmp, &kwatch_all_wp_list, list) {
+		list_del(&wp->list);
+
+		WRITE_ONCE(wp->teardown, true);
+		atomic_dec(&wp->refcount);
+
+		/* Wait for all async IPIs to finish */
+		while (atomic_read(&wp->refcount) > 0)
+			cpu_relax();
+
+		unregister_wide_hw_breakpoint(wp->event);
+		free_percpu(wp->csd_arm);
+		free_percpu(wp->csd_disarm);
+		kfree(wp);
+	}
+	mutex_unlock(&kwatch_all_wp_mutex);
+}
+
+int kwatch_hwbp_prealloc(u16 max_watch)
+{
+	struct kwatch_watchpoint *wp;
+	int success = 0, cpu;
+	int ret;
+
+	atomic_long_set(&kwatch_arm_ipi_suppressed, 0);
+
+	while (!max_watch || success < max_watch) {
+		wp = kzalloc_obj(*wp);
+		if (!wp)
+			break;
+
+		wp->csd_arm = alloc_percpu(call_single_data_t);
+		wp->csd_disarm = alloc_percpu(call_single_data_t);
+		if (!wp->csd_arm || !wp->csd_disarm) {
+			free_percpu(wp->csd_arm);
+			free_percpu(wp->csd_disarm);
+			kfree(wp);
+			break;
+		}
+
+		for_each_possible_cpu(cpu) {
+			INIT_CSD(per_cpu_ptr(wp->csd_arm, cpu),
+				 kwatch_hwbp_arm_local, wp);
+			INIT_CSD(per_cpu_ptr(wp->csd_disarm, cpu),
+				 kwatch_hwbp_disarm_local, wp);
+		}
+
+		wp->teardown = false;
+
+		hw_breakpoint_init(&wp->attr);
+		wp->attr.bp_addr = (unsigned long)&kwatch_dummy_holder;
+		wp->attr.bp_len = sizeof(unsigned long);
+		/* kwatch localizes corruption: it always watches for writes. */
+		wp->attr.bp_type = HW_BREAKPOINT_W;
+
+		wp->event = register_wide_hw_breakpoint(&wp->attr,
+							kwatch_hwbp_handler,
+							wp);
+		if (IS_ERR_PCPU(wp->event)) {
+			free_percpu(wp->csd_arm);
+			free_percpu(wp->csd_disarm);
+			kfree(wp);
+			break;
+		}
+
+		atomic_set(&wp->refcount, 1);
+
+		mutex_lock(&kwatch_all_wp_mutex);
+		list_add(&wp->list, &kwatch_all_wp_list);
+		mutex_unlock(&kwatch_all_wp_mutex);
+		success++;
+	}
+
+	if (!success)
+		return -EBUSY;
+
+	/*
+	 * A fresh prealloc must start from an empty slot array; warn if a
+	 * previous session was not torn down, since refilling without a reset
+	 * would index past the freshly sized array.
+	 */
+	WARN_ON_ONCE(kwatch_wp_slots || kwatch_wp_nr);
+	kwatch_wp_nr = 0;
+
+	kwatch_wp_slots = kcalloc(success, sizeof(*kwatch_wp_slots),
+				  GFP_KERNEL);
+	if (!kwatch_wp_slots) {
+		kwatch_hwbp_free();
+		return -ENOMEM;
+	}
+	mutex_lock(&kwatch_all_wp_mutex);
+	list_for_each_entry(wp, &kwatch_all_wp_list, list)
+		kwatch_wp_slots[kwatch_wp_nr++] = wp;
+	mutex_unlock(&kwatch_all_wp_mutex);
+
+	ret = cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN, "kwatch:online",
+					kwatch_hwbp_cpu_online,
+					kwatch_hwbp_cpu_offline);
+	if (ret < 0) {
+		kwatch_hwbp_free();
+		return ret;
+	}
+
+	kwatch_hwbp_cpuhp_state = ret;
+	return 0;
+}
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 09/13] mm/kwatch: add probe lifecycle runtime
From: Jinchao Wang @ 2026-07-17 13:05 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

Open and close the watch window with a kretprobe on the target
function: the entry handler tracks per-task nesting depth and, when
the configured depth is reached, resolves the watch expression and
arms a watchpoint; the exit handler disarms it. An optional kprobe
at func_offset arms mid-function instead of at entry.

Functions running in a real NMI(-like) context are rejected once, at
function entry, by comparing the NMI nesting count against the one
NMI-like layer that int3-based kprobe delivery itself adds; a
companion kprobe with a post_handler pins the probe point so jump
optimization cannot change the delivery mechanism after it is
sampled. Rejections are counted and exposed to the control plane.

A global epoch versioning scheme invalidates stale per-task state
across reconfigurations, and a per-CPU mute flag keeps window
management quiet while a CPU rewrites its own debug registers.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 mm/kwatch/Makefile |   2 +-
 mm/kwatch/probe.c  | 275 +++++++++++++++++++++++++++++++++++++++++++++
 2 files changed, 276 insertions(+), 1 deletion(-)
 create mode 100644 mm/kwatch/probe.c

diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index b2bc3003c89b..f04673cc5b1c 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
 obj-$(CONFIG_KWATCH) += kwatch.o
 
-kwatch-y := deref.o task_ctx.o hwbp.o
+kwatch-y := deref.o task_ctx.o hwbp.o probe.o
diff --git a/mm/kwatch/probe.c b/mm/kwatch/probe.c
new file mode 100644
index 000000000000..249aa50c9f78
--- /dev/null
+++ b/mm/kwatch/probe.c
@@ -0,0 +1,275 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/atomic.h>
+#include <linux/kprobes.h>
+#include <linux/kallsyms.h>
+#include <linux/percpu.h>
+#include <linux/preempt.h>
+#include <linux/sched.h>
+
+#include "kwatch.h"
+#define TRAMPOLINE_CHECK_DEPTH 16
+static DEFINE_PER_CPU(bool, kwatch_probe_cpu_muted);
+
+struct kwatch_probe_ctx {
+	struct kprobe kp;
+	struct kretprobe rp;
+	struct kprobe pin_kp;
+	const struct kwatch_config *cfg;
+	bool rp_via_int3;
+
+	u32 epoch;
+};
+
+static struct kwatch_probe_ctx kwatch_probe_ctx;
+static atomic_long_t kwatch_nmi_rejected;
+
+unsigned long kwatch_probe_nmi_rejected(void)
+{
+	return atomic_long_read(&kwatch_nmi_rejected);
+}
+
+/*
+ * True if the probed function itself runs in an NMI-like context.
+ * int3-based kprobe delivery adds one NMI-like layer of its own;
+ * delivery is pinned at registration so the subtraction stays exact.
+ */
+static bool kwatch_probed_ctx_in_nmi(bool via_int3)
+{
+	return (preempt_count() & NMI_MASK) > (via_int3 ? NMI_OFFSET : 0);
+}
+
+static void kwatch_pin_post_handler(struct kprobe *p, struct pt_regs *regs,
+				    unsigned long flags)
+{
+	/* a post_handler pins the probepoint: no jump optimization */
+}
+
+bool kwatch_probe_validate_hit(struct pt_regs *regs,
+			       struct task_struct *arm_tsk)
+{
+	struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+	const struct kwatch_config *cfg = kwatch_probe_ctx.cfg;
+
+	if (unlikely(!ctx || !cfg))
+		return true;
+
+	if (arm_tsk != current || ctx->depth != cfg->depth + 1)
+		return true;
+
+	return false;
+}
+
+void kwatch_probe_mute(bool mute)
+{
+	__this_cpu_write(kwatch_probe_cpu_muted, mute);
+}
+
+static inline bool kwatch_probe_is_muted(void)
+{
+	return __this_cpu_read(kwatch_probe_cpu_muted);
+}
+
+enum kwatch_probe_position {
+	KWATCH_PROBE_POSITION_ENTRY,
+	KWATCH_PROBE_POSITION_ACTIVE,
+	KWATCH_PROBE_POSITION_EXIT
+};
+
+static bool kwatch_tsk_ctx_check(enum kwatch_probe_position pos)
+{
+	struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(true);
+	u32 epoch;
+
+	if (unlikely(!ctx))
+		return false;
+
+	/* Pairs with smp_store_release() in kwatch_probe_start/stop() */
+	epoch = smp_load_acquire(&kwatch_probe_ctx.epoch);
+
+	if (unlikely(ctx->epoch != epoch))
+		kwatch_tsk_ctx_reset(ctx, epoch);
+
+	if (unlikely(!epoch)) {
+		/*
+		 * No active session (not yet published, or already stopped):
+		 * kwatch_tsk_ctx_get(true) above may have just claimed a slot
+		 * for current. Release it here, otherwise an entry that lands
+		 * in the register->epoch-publish window leaks the slot until
+		 * the pool is freed.
+		 */
+		kwatch_tsk_ctx_release(ctx);
+		return false;
+	}
+
+	switch (pos) {
+	case KWATCH_PROBE_POSITION_ENTRY:
+		ctx->depth++;
+		return true;
+	case KWATCH_PROBE_POSITION_ACTIVE:
+		return true;
+	case KWATCH_PROBE_POSITION_EXIT:
+		if (unlikely(ctx->depth == 0)) {
+			kwatch_tsk_ctx_put();
+			return false;
+		}
+
+		ctx->depth--;
+		if (ctx->depth == 0) {
+			kwatch_tsk_ctx_put();
+			return false;
+		}
+		return true;
+	}
+	return false;
+}
+
+static int kwatch_activate_handler(struct kprobe *p, struct pt_regs *regs)
+{
+	struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+	unsigned long watch_addr;
+	u16 watch_len;
+
+	if (unlikely(!ctx))
+		return 0;
+
+	if (unlikely(kwatch_probe_is_muted()))
+		return 0;
+
+	if (unlikely(!kwatch_tsk_ctx_check(KWATCH_PROBE_POSITION_ACTIVE)))
+		return 0;
+
+	if (ctx->depth != kwatch_probe_ctx.cfg->depth + 1 || ctx->wp)
+		return 0;
+
+	if (kwatch_deref_resolve(kwatch_probe_ctx.cfg, regs, &watch_addr,
+				 &watch_len))
+		return 0;
+
+	if (kwatch_hwbp_get(&ctx->wp))
+		return 0;
+
+	kwatch_hwbp_arm(ctx->wp, watch_addr, watch_len);
+	return 0;
+}
+
+static int kwatch_lifecycle_entry(struct kretprobe_instance *ri,
+				  struct pt_regs *regs)
+{
+	/*
+	 * Single policy point: the target function's context is judged once
+	 * here. A rejected invocation never increments depth, so the offset
+	 * kprobe path inherits the verdict through the depth check.
+	 */
+	if (unlikely(kwatch_probed_ctx_in_nmi(kwatch_probe_ctx.rp_via_int3))) {
+		atomic_long_inc(&kwatch_nmi_rejected);
+		return 1; /* NMI context is unsupported: no window, no return hook */
+	}
+
+	if (!kwatch_tsk_ctx_check(KWATCH_PROBE_POSITION_ENTRY))
+		return 0;
+
+	if (kwatch_probe_ctx.cfg->func_offset == 0)
+		kwatch_activate_handler(NULL, regs);
+
+	return 0;
+}
+
+static int kwatch_lifecycle_exit(struct kretprobe_instance *ri,
+				 struct pt_regs *regs)
+{
+	struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+
+	if (unlikely(!ctx))
+		return 0;
+
+	if (!kwatch_tsk_ctx_check(KWATCH_PROBE_POSITION_EXIT))
+		return 0;
+
+	if (ctx->depth == kwatch_probe_ctx.cfg->depth) {
+		struct kwatch_watchpoint *wp = xchg(&ctx->wp, NULL);
+
+		if (wp)
+			kwatch_hwbp_put(wp);
+	}
+
+	return 0;
+}
+
+int kwatch_probe_start(struct kwatch_config *cfg)
+{
+	static u32 next_epoch;
+	u32 current_epoch;
+	int ret;
+
+	/*
+	 * Lockless check to prevent concurrent starts. Strictly serialized
+	 * by the control plane mutex, but serves as a sanity check.
+	 */
+	if (smp_load_acquire(&kwatch_probe_ctx.epoch) != 0)
+		return -EBUSY;
+
+	memset(&kwatch_probe_ctx, 0, sizeof(kwatch_probe_ctx));
+	kwatch_probe_ctx.cfg = cfg;
+
+	/* Session-scoped, like arm_ipi_suppressed in kwatch_hwbp_prealloc() */
+	atomic_long_set(&kwatch_nmi_rejected, 0);
+
+	/*
+	 * Pin the entry probepoint before the kretprobe registers, so its
+	 * delivery (int3 vs ftrace) can never change under jump optimization.
+	 * register_kretprobe() clears kp.post_handler, hence the companion.
+	 */
+	kwatch_probe_ctx.pin_kp.symbol_name = cfg->func_name;
+	kwatch_probe_ctx.pin_kp.post_handler = kwatch_pin_post_handler;
+	ret = register_kprobe(&kwatch_probe_ctx.pin_kp);
+	if (ret < 0)
+		return ret;
+
+	kwatch_probe_ctx.rp.entry_handler = kwatch_lifecycle_entry;
+	kwatch_probe_ctx.rp.handler = kwatch_lifecycle_exit;
+	kwatch_probe_ctx.rp.kp.symbol_name = cfg->func_name;
+
+	ret = register_kretprobe(&kwatch_probe_ctx.rp);
+	if (ret < 0) {
+		unregister_kprobe(&kwatch_probe_ctx.pin_kp);
+		return ret;
+	}
+	kwatch_probe_ctx.rp_via_int3 = !kprobe_ftrace(&kwatch_probe_ctx.rp.kp);
+
+	if (cfg->func_offset) {
+		kwatch_probe_ctx.kp.symbol_name = cfg->func_name;
+		kwatch_probe_ctx.kp.offset = cfg->func_offset;
+		kwatch_probe_ctx.kp.pre_handler = kwatch_activate_handler;
+
+		ret = register_kprobe(&kwatch_probe_ctx.kp);
+		if (ret) {
+			unregister_kretprobe(&kwatch_probe_ctx.rp);
+			unregister_kprobe(&kwatch_probe_ctx.pin_kp);
+			return ret;
+		}
+	}
+
+	current_epoch = ++next_epoch;
+	if (unlikely(!current_epoch))
+		current_epoch = ++next_epoch;
+
+	/* Pairs with smp_load_acquire() in kwatch_tsk_ctx_check() */
+	smp_store_release(&kwatch_probe_ctx.epoch, current_epoch);
+
+	return 0;
+}
+
+void kwatch_probe_stop(void)
+{
+	if (!kwatch_probe_ctx.epoch)
+		return;
+
+	/* Pairs with smp_load_acquire() in kwatch_tsk_ctx_check() */
+	smp_store_release(&kwatch_probe_ctx.epoch, 0);
+
+	if (kwatch_probe_ctx.cfg->func_offset > 0)
+		unregister_kprobe(&kwatch_probe_ctx.kp);
+
+	unregister_kretprobe(&kwatch_probe_ctx.rp);
+	unregister_kprobe(&kwatch_probe_ctx.pin_kp);
+}
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 10/13] mm/kwatch: add anchor thread for global watchpoints
From: Jinchao Wang @ 2026-07-17 13:06 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

Global variables have no function whose execution can bound the
watch window. Provide one: a kernel thread sleeps for the configured
duration inside a dedicated noinline function,
kwatch_global_anchor(), and the probe runtime hooks that function
like any other target.

When the duration expires the thread schedules a work item that
tears the session down; the expired flag is cleared under the
control-plane mutex so a stale work item from a previous session
cannot stop a new one.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 mm/kwatch/Makefile |  2 +-
 mm/kwatch/anchor.c | 85 ++++++++++++++++++++++++++++++++++++++++++++++
 2 files changed, 86 insertions(+), 1 deletion(-)
 create mode 100644 mm/kwatch/anchor.c

diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index f04673cc5b1c..b196c794619a 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
 obj-$(CONFIG_KWATCH) += kwatch.o
 
-kwatch-y := deref.o task_ctx.o hwbp.o probe.o
+kwatch-y := deref.o task_ctx.o hwbp.o probe.o anchor.o
diff --git a/mm/kwatch/anchor.c b/mm/kwatch/anchor.c
new file mode 100644
index 000000000000..e87eb5e813ff
--- /dev/null
+++ b/mm/kwatch/anchor.c
@@ -0,0 +1,85 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/kthread.h>
+#include <linux/wait.h>
+#include <linux/jiffies.h>
+#include <linux/err.h>
+#include <linux/workqueue.h>
+
+#include "kwatch.h"
+
+static DECLARE_WAIT_QUEUE_HEAD(kwatch_anchor_wq);
+static struct task_struct *kwatch_anchor_tsk;
+static bool kwatch_anchor_expired;
+
+bool kwatch_anchor_has_expired(void)
+{
+	return READ_ONCE(kwatch_anchor_expired);
+}
+
+void kwatch_anchor_clear_expired(void)
+{
+	WRITE_ONCE(kwatch_anchor_expired, false);
+}
+
+static void kwatch_auto_stop_handler(struct work_struct *work)
+{
+	kwatch_auto_stop();
+}
+
+static DECLARE_WORK(kwatch_auto_stop_work, kwatch_auto_stop_handler);
+
+noinline void kwatch_global_anchor(unsigned long duration_sec)
+{
+	/* TASK_IDLE: a long timed sleep must not inflate loadavg or trip the
+	 * hung-task detector the way TASK_UNINTERRUPTIBLE would.
+	 */
+	wait_event_idle_timeout(kwatch_anchor_wq, kthread_should_stop(),
+				duration_sec * HZ);
+}
+
+static int kwatch_anchor_thread_fn(void *data)
+{
+	unsigned long duration = (unsigned long)data;
+
+	kwatch_global_anchor(duration);
+
+	if (!kthread_should_stop()) {
+		/* mark before scheduling; cleared under the control mutex */
+		WRITE_ONCE(kwatch_anchor_expired, true);
+		schedule_work(&kwatch_auto_stop_work);
+	}
+
+	while (!kthread_should_stop())
+		schedule_timeout_idle(HZ);
+
+	return 0;
+}
+
+int kwatch_anchor_start(u16 duration)
+{
+	kwatch_anchor_tsk = kthread_run(kwatch_anchor_thread_fn,
+					(void *)(unsigned long)duration,
+					"kwatch_anchor");
+	if (IS_ERR(kwatch_anchor_tsk)) {
+		int ret = PTR_ERR(kwatch_anchor_tsk);
+
+		kwatch_anchor_tsk = NULL;
+		return ret;
+	}
+	return 0;
+}
+
+void kwatch_anchor_stop(void)
+{
+	if (kwatch_anchor_tsk) {
+		kthread_stop(kwatch_anchor_tsk);
+		kwatch_anchor_tsk = NULL;
+	}
+}
+
+void kwatch_anchor_cancel_work(void)
+{
+	cancel_work_sync(&kwatch_auto_stop_work);
+}
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 11/13] mm/kwatch: add debugfs control plane
From: Jinchao Wang @ 2026-07-17 13:06 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

Wire the pieces together behind a single debugfs file,
/sys/kernel/debug/kwatch/config. Writing a key=value configuration
string stops any active session and starts a new one; reading shows
the active configuration and the nmi_rejected counter. An open-count
guard keeps the file single-open and a mutex serializes
start/stop/auto-stop against each other.

Add the Kconfig entry and hook mm/kwatch into the mm build. KWatch
can be built in or as a module; symbol-name watch expressions need
the built-in flavour (kallsyms_lookup_name is not exported).

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 MAINTAINERS        |   8 ++
 mm/Kconfig         |   1 +
 mm/Makefile        |   1 +
 mm/kwatch/Kconfig  |  16 +++
 mm/kwatch/Makefile |   2 +-
 mm/kwatch/core.c   | 324 +++++++++++++++++++++++++++++++++++++++++++++
 6 files changed, 351 insertions(+), 1 deletion(-)
 create mode 100644 mm/kwatch/Kconfig
 create mode 100644 mm/kwatch/core.c

diff --git a/MAINTAINERS b/MAINTAINERS
index 7cc4bca5a2c5..b6371f92fe5c 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -14578,6 +14578,14 @@ S:	Supported
 T:	git git://git.kernel.org/pub/scm/virt/kvm/kvm.git
 F:	arch/x86/kvm/xen.*
 
+KWATCH
+M:	Jinchao Wang <wangjinchao600@gmail.com>
+L:	linux-mm@kvack.org
+S:	Maintained
+F:	Documentation/dev-tools/kwatch.rst
+F:	include/trace/events/kwatch.h
+F:	mm/kwatch/
+
 L3MDEV
 M:	David Ahern <dsahern@kernel.org>
 L:	netdev@vger.kernel.org
diff --git a/mm/Kconfig b/mm/Kconfig
index 9e0ca4824905..cac75a46e21a 100644
--- a/mm/Kconfig
+++ b/mm/Kconfig
@@ -1510,5 +1510,6 @@ config LAZY_MMU_MODE_KUNIT_TEST
 	  If unsure, say N.
 
 source "mm/damon/Kconfig"
+source "mm/kwatch/Kconfig"
 
 endmenu
diff --git a/mm/Makefile b/mm/Makefile
index eff9f9e7e061..80c688330358 100644
--- a/mm/Makefile
+++ b/mm/Makefile
@@ -92,6 +92,7 @@ obj-$(CONFIG_PAGE_POISONING) += page_poison.o
 obj-$(CONFIG_KASAN)	+= kasan/
 obj-$(CONFIG_KFENCE) += kfence/
 obj-$(CONFIG_KMSAN)	+= kmsan/
+obj-$(CONFIG_KWATCH) += kwatch/
 obj-$(CONFIG_FAILSLAB) += failslab.o
 obj-$(CONFIG_FAIL_PAGE_ALLOC) += fail_page_alloc.o
 obj-$(CONFIG_MEMTEST)		+= memtest.o
diff --git a/mm/kwatch/Kconfig b/mm/kwatch/Kconfig
new file mode 100644
index 000000000000..9daf6d4463ef
--- /dev/null
+++ b/mm/kwatch/Kconfig
@@ -0,0 +1,16 @@
+config KWATCH
+	tristate "Kernel Watch Framework"
+	depends on PERF_EVENTS && HAVE_HW_BREAKPOINT && DEBUG_FS
+	depends on HAVE_REINSTALL_HW_BREAKPOINT
+	depends on KPROBES && KRETPROBES
+	depends on STACKTRACE
+	help
+	  A generalized hardware-assisted memory monitor utility.
+	  It provides a low-overhead, real-time trigger mechanism to monitor
+	  kernel memory safely in atomic contexts using hardware breakpoints.
+
+	  KWatch is designed to catch silent memory corruptions, stack
+	  overwrites, and complex Heisenbugs by synchronously trapping the
+	  exact instruction causing the illegal access.
+
+	  If unsure, say N.
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index b196c794619a..02d7917602f1 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
 obj-$(CONFIG_KWATCH) += kwatch.o
 
-kwatch-y := deref.o task_ctx.o hwbp.o probe.o anchor.o
+kwatch-y := core.o deref.o task_ctx.o hwbp.o probe.o anchor.o
diff --git a/mm/kwatch/core.c b/mm/kwatch/core.c
new file mode 100644
index 000000000000..d8526d5aae5c
--- /dev/null
+++ b/mm/kwatch/core.c
@@ -0,0 +1,324 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/kstrtox.h>
+#include <linux/module.h>
+#include <linux/slab.h>
+#include <linux/string.h>
+#include <linux/types.h>
+#include <linux/atomic.h>
+#include <linux/debugfs.h>
+#include <linux/mutex.h>
+#include "kwatch.h"
+
+static struct kwatch_config kwatch_config;
+static bool watching_active;
+
+static struct dentry *dbgfs_dir;
+static struct dentry *dbgfs_config;
+static DEFINE_MUTEX(kwatch_dbgfs_mutex);
+static atomic_t dbgfs_config_busy = ATOMIC_INIT(0);
+
+static int kwatch_start_watching(void)
+{
+	int ret;
+
+	if (!strlen(kwatch_config.func_name)) {
+		if (kwatch_config.duration > 0) {
+			strscpy(kwatch_config.func_name, "kwatch_global_anchor",
+				sizeof(kwatch_config.func_name));
+		} else {
+			pr_err("func_name or duration is required\n");
+			return -EINVAL;
+		}
+	} else if (kwatch_config.duration > 0 &&
+		   strcmp(kwatch_config.func_name, "kwatch_global_anchor")) {
+		pr_warn("duration is ignored when watching a specific function\n");
+	}
+
+	ret = kwatch_hwbp_prealloc(kwatch_config.max_watch);
+	if (ret) {
+		pr_err("kwatch_hwbp_prealloc ret: %d\n", ret);
+		return ret;
+	}
+
+	ret = kwatch_tsk_ctx_prealloc(kwatch_config.max_concurrency);
+	if (ret) {
+		kwatch_hwbp_free();
+		return ret;
+	}
+
+	ret = kwatch_probe_start(&kwatch_config);
+	if (ret) {
+		pr_err("kwatch_probe_start ret: %d\n", ret);
+		kwatch_tsk_ctx_free();
+		kwatch_hwbp_free();
+		return ret;
+	}
+
+	if (!strcmp(kwatch_config.func_name, "kwatch_global_anchor")) {
+		ret = kwatch_anchor_start(kwatch_config.duration);
+		if (ret) {
+			kwatch_probe_stop();
+			synchronize_rcu();
+			kwatch_tsk_ctx_release_wps();
+			kwatch_hwbp_free();
+			kwatch_tsk_ctx_free();
+			return ret;
+		}
+	}
+
+	watching_active = true;
+	return 0;
+}
+
+static void kwatch_stop_watching(void)
+{
+	watching_active = false;
+
+	kwatch_anchor_stop();
+	/* after kthread_stop: the dead thread cannot re-mark expiry */
+	kwatch_anchor_clear_expired();
+
+	kwatch_probe_stop();
+	synchronize_rcu();
+	kwatch_tsk_ctx_release_wps();
+	/*
+	 * Waits for disarm IPIs and unregisters breakpoints: no #DB can
+	 * reach the ctx pool once this returns.
+	 */
+	kwatch_hwbp_free();
+	kwatch_tsk_ctx_free();
+}
+
+void kwatch_auto_stop(void)
+{
+	mutex_lock(&kwatch_dbgfs_mutex);
+	/* the expired check neutralizes work items from torn-down sessions */
+	if (watching_active && kwatch_anchor_has_expired()) {
+		kwatch_stop_watching();
+		pr_info("watch duration expired, stopped watching\n");
+	}
+	mutex_unlock(&kwatch_dbgfs_mutex);
+}
+
+static int kwatch_config_parse(char *buf, struct kwatch_config *cfg)
+{
+	char *token, *key, *val;
+	int ret = 0;
+
+	memset(cfg, 0, sizeof(*cfg));
+	cfg->max_concurrency = 256;
+	cfg->max_watch = 4;
+	cfg->watch_len = 8;
+
+	while ((token = strsep(&buf, " \t\n")) != NULL) {
+		if (!*token)
+			continue;
+		key = strsep(&token, "=");
+		val = token;
+		if (!key || !val)
+			return -EINVAL;
+
+		if (!strcmp(key, "func_name")) {
+			strscpy(cfg->func_name, val, sizeof(cfg->func_name));
+		} else if (!strcmp(key, "func_offset")) {
+			ret = kstrtou16(val, 0, &cfg->func_offset);
+		} else if (!strcmp(key, "depth")) {
+			ret = kstrtou16(val, 0, &cfg->depth);
+		} else if (!strcmp(key, "max_concurrency")) {
+			ret = kstrtou16(val, 0, &cfg->max_concurrency);
+		} else if (!strcmp(key, "max_watch")) {
+			ret = kstrtou16(val, 0, &cfg->max_watch);
+		} else if (!strcmp(key, "watch_len")) {
+			ret = kstrtou16(val, 0, &cfg->watch_len);
+			if (!ret && cfg->watch_len != 1 &&
+			    cfg->watch_len != 2 && cfg->watch_len != 4 &&
+			    cfg->watch_len != 8)
+				ret = -EINVAL;
+		} else if (!strcmp(key, "duration")) {
+			ret = kstrtou16(val, 0, &cfg->duration);
+		} else if (!strcmp(key, "watch_expr")) {
+			strscpy(cfg->watch_expr, val, sizeof(cfg->watch_expr));
+			ret = kwatch_deref_parse(cfg, val);
+		}
+
+		if (ret)
+			return ret;
+	}
+	return 0;
+}
+
+static int kwatch_dbgfs_open(struct inode *inode, struct file *file)
+{
+	if (atomic_cmpxchg(&dbgfs_config_busy, 0, 1))
+		return -EBUSY;
+	return 0;
+}
+
+static int kwatch_dbgfs_release(struct inode *inode, struct file *file)
+{
+	atomic_set(&dbgfs_config_busy, 0);
+	return 0;
+}
+
+static ssize_t kwatch_dbgfs_read(struct file *file, char __user *user_buf,
+				 size_t count, loff_t *ppos)
+{
+	char *out_buf;
+	size_t len = 0;
+	ssize_t ret;
+
+	out_buf = kzalloc(MAX_CONFIG_STR_LEN, GFP_KERNEL);
+	if (!out_buf)
+		return -ENOMEM;
+
+	/*
+	 * Serialize against the write path and the auto-stop work item so the
+	 * config snapshot cannot tear or race a session teardown.
+	 */
+	mutex_lock(&kwatch_dbgfs_mutex);
+
+	if (watching_active) {
+		len += scnprintf(out_buf + len, MAX_CONFIG_STR_LEN - len,
+				 "func_name=%s\n"
+				 "func_offset=%u\n"
+				 "depth=%u\n"
+				 "duration=%u\n"
+				 "max_concurrency=%u\n"
+				 "max_watch=%u\n"
+				 "watch_len=%u\n",
+				 kwatch_config.func_name,
+				 kwatch_config.func_offset, kwatch_config.depth,
+				 kwatch_config.duration,
+				 kwatch_config.max_concurrency,
+				 kwatch_config.max_watch,
+				 kwatch_config.watch_len);
+
+		if (kwatch_config.base == KWATCH_BASE_GLOBAL_SYM) {
+			len += scnprintf(out_buf + len, MAX_CONFIG_STR_LEN - len,
+					 "sym_addr=0x%lx\n", kwatch_config.sym_addr);
+		}
+
+		len += scnprintf(out_buf + len, MAX_CONFIG_STR_LEN - len,
+				 "watch_expr=%s\n"
+				 "nmi_rejected=%lu\n"
+				 "arm_ipi_suppressed=%lu\n",
+				 kwatch_config.watch_expr,
+				 kwatch_probe_nmi_rejected(),
+				 kwatch_hwbp_arm_ipi_suppressed());
+	} else {
+		len = scnprintf(out_buf, MAX_CONFIG_STR_LEN, "not watching\n");
+	}
+
+	mutex_unlock(&kwatch_dbgfs_mutex);
+
+	ret = simple_read_from_buffer(user_buf, count, ppos, out_buf, len);
+	kfree(out_buf);
+	return ret;
+}
+
+static ssize_t kwatch_dbgfs_write(struct file *file, const char __user *buffer,
+				  size_t count, loff_t *ppos)
+{
+	char *input_alloc;
+	char *parse_str;
+	int ret;
+
+	if (count == 0 || count >= MAX_CONFIG_STR_LEN)
+		return -EINVAL;
+
+	input_alloc = memdup_user_nul(buffer, count);
+	if (IS_ERR(input_alloc))
+		return PTR_ERR(input_alloc);
+
+	mutex_lock(&kwatch_dbgfs_mutex);
+
+	if (watching_active)
+		kwatch_stop_watching();
+
+	parse_str = strim(input_alloc);
+
+	if (!strlen(parse_str)) {
+		ret = -EINVAL;
+		goto out;
+	}
+
+	ret = kwatch_config_parse(parse_str, &kwatch_config);
+	if (ret) {
+		pr_err("Failed to parse config %d\n", ret);
+		goto out;
+	}
+
+	ret = kwatch_start_watching();
+	if (ret) {
+		pr_err("Failed to start watching with %d\n", ret);
+		goto out;
+	}
+
+	ret = count;
+
+out:
+	mutex_unlock(&kwatch_dbgfs_mutex);
+	kfree(input_alloc);
+	return ret;
+}
+
+static const struct file_operations kwatch_fops = {
+	.owner = THIS_MODULE,
+	.open = kwatch_dbgfs_open,
+	.release = kwatch_dbgfs_release,
+	.read = kwatch_dbgfs_read,
+	.write = kwatch_dbgfs_write,
+};
+
+static int __init kwatch_init(void)
+{
+	int ret = 0;
+
+	memset(&kwatch_config, 0, sizeof(kwatch_config));
+
+	dbgfs_dir = debugfs_create_dir("kwatch", NULL);
+	if (IS_ERR(dbgfs_dir)) {
+		ret = PTR_ERR(dbgfs_dir);
+		goto err_dir;
+	}
+
+	dbgfs_config = debugfs_create_file("config", 0600, dbgfs_dir, NULL,
+					   &kwatch_fops);
+	if (IS_ERR(dbgfs_config)) {
+		ret = PTR_ERR(dbgfs_config);
+		goto err_file;
+	}
+
+	pr_info("module loaded\n");
+	return 0;
+
+err_file:
+	debugfs_remove_recursive(dbgfs_dir);
+	dbgfs_dir = NULL;
+err_dir:
+	return ret;
+}
+module_init(kwatch_init);
+
+static void __exit kwatch_exit(void)
+{
+	mutex_lock(&kwatch_dbgfs_mutex);
+	if (watching_active)
+		kwatch_stop_watching();
+	mutex_unlock(&kwatch_dbgfs_mutex);
+
+	/* the anchor thread is dead: nothing can schedule new work now */
+	kwatch_anchor_cancel_work();
+
+	debugfs_remove_recursive(dbgfs_dir);
+	dbgfs_dir = NULL;
+
+	pr_info("kwatch unloaded\n");
+}
+module_exit(kwatch_exit);
+
+MODULE_AUTHOR("Jinchao Wang <wangjinchao600@gmail.com>");
+MODULE_DESCRIPTION("Kernel watchpoint");
+MODULE_LICENSE("GPL");
-- 
2.53.0


^ permalink raw reply related

* [RFC PATCH v2 12/13] mm/kwatch: add KUnit tests for the watch expression parser
From: Jinchao Wang @ 2026-07-17 13:07 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

Cover base anchors (stack, argN, absolute address), positive and
negative offsets, dereference chains, and rejection of malformed
expressions (missing offsets, bad argument index, junk offsets).

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 mm/kwatch/.kunitconfig |   9 +++
 mm/kwatch/Kconfig      |  12 ++++
 mm/kwatch/Makefile     |   1 +
 mm/kwatch/deref_test.c | 146 +++++++++++++++++++++++++++++++++++++++++
 4 files changed, 168 insertions(+)
 create mode 100644 mm/kwatch/.kunitconfig
 create mode 100644 mm/kwatch/deref_test.c

diff --git a/mm/kwatch/.kunitconfig b/mm/kwatch/.kunitconfig
new file mode 100644
index 000000000000..7e977ddf0da1
--- /dev/null
+++ b/mm/kwatch/.kunitconfig
@@ -0,0 +1,9 @@
+CONFIG_KUNIT=y
+CONFIG_KWATCH=y
+CONFIG_KWATCH_KUNIT_TEST=y
+CONFIG_PERF_EVENTS=y
+CONFIG_HAVE_HW_BREAKPOINT=y
+CONFIG_HAVE_REINSTALL_HW_BREAKPOINT=y
+CONFIG_KPROBES=y
+CONFIG_KRETPROBES=y
+CONFIG_PRINTK=y
diff --git a/mm/kwatch/Kconfig b/mm/kwatch/Kconfig
index 9daf6d4463ef..6ec9aa448ece 100644
--- a/mm/kwatch/Kconfig
+++ b/mm/kwatch/Kconfig
@@ -14,3 +14,15 @@ config KWATCH
 	  exact instruction causing the illegal access.
 
 	  If unsure, say N.
+
+config KWATCH_KUNIT_TEST
+	bool "KUnit tests for KWatch" if !KUNIT_ALL_TESTS
+	# Built into the kwatch module, so it must be y; a bool cannot be
+	# enabled when KWATCH is a module (KWATCH=m would force it off).
+	depends on KWATCH=y && KUNIT
+	default KUNIT_ALL_TESTS
+	help
+	  Enable KUnit tests for the KWatch kernel module.
+	  This suite tests the core parsing logic, the pointer-chasing
+	  finite state machine, and edge cases involving complex watchpoint
+	  expressions. If unsure, say N.
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index 02d7917602f1..1d223d73b461 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,4 @@
 obj-$(CONFIG_KWATCH) += kwatch.o
 
 kwatch-y := core.o deref.o task_ctx.o hwbp.o probe.o anchor.o
+kwatch-$(CONFIG_KWATCH_KUNIT_TEST) += deref_test.o
diff --git a/mm/kwatch/deref_test.c b/mm/kwatch/deref_test.c
new file mode 100644
index 000000000000..35919dd24d92
--- /dev/null
+++ b/mm/kwatch/deref_test.c
@@ -0,0 +1,146 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <kunit/test.h>
+#include "kwatch.h"
+#include <linux/string.h>
+
+static void kwatch_test_parse_deref_chain(struct kunit *test)
+{
+	struct kwatch_config cfg;
+	int ret;
+
+	// Test 1: stack
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "stack");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_STACK);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+
+	// Test 2: arg1
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg1");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG1);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+
+	// Test 3: arg6+8
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg6+8");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG6);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 8);
+
+	// Test 4: arg2-16
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg2-16");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG2);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], -16);
+
+	// Test 5: arg3->8
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg3->8");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG3);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[1], 8);
+
+	// Test 6: arg4+8->16
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg4+8->16");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG4);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 8);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[1], 16);
+
+	// Test 7: arg5-8->-16
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg5-8->-16");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG5);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], -8);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[1], -16);
+
+	// Test 8: stack->0->8
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "stack->0->8");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_STACK);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 3);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[1], 0);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[2], 8);
+
+	// Test 9: arg1->+8
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg1->+8");
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG1);
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[1], 8);
+
+	// Test 9.1: arg1-> (implicit 0 should fail)
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg1->");
+	KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+	// Test 9.2: stack->->8 (implicit 0 should fail)
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "stack->->8");
+	KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+	// Test 10: Invalid base
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "invalid_base");
+	KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+	// Test 11: Invalid offset
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg1+abc");
+	KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+	// Test 12: Invalid arg
+	memset(&cfg, 0, sizeof(cfg));
+	ret = kwatch_deref_parse(&cfg, "arg7");
+	KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+	// Test 13: Absolute address. Use a width-appropriate literal: a 64-bit
+	// address would overflow unsigned long and fail kstrtoul() on 32-bit.
+	memset(&cfg, 0, sizeof(cfg));
+#if BITS_PER_LONG == 64
+	ret = kwatch_deref_parse(&cfg, "0xffffffff81000000+8");
+#else
+	ret = kwatch_deref_parse(&cfg, "0xc1000000+8");
+#endif
+	KUNIT_EXPECT_EQ(test, ret, 0);
+	KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ABS_ADDR);
+#if BITS_PER_LONG == 64
+	KUNIT_EXPECT_EQ(test, cfg.sym_addr, 0xffffffff81000000UL);
+#else
+	KUNIT_EXPECT_EQ(test, cfg.sym_addr, 0xc1000000UL);
+#endif
+	KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+	KUNIT_EXPECT_EQ(test, cfg.offsets[0], 8);
+}
+
+static struct kunit_case kwatch_deref_test_cases[] = {
+	KUNIT_CASE(kwatch_test_parse_deref_chain),
+	{}
+};
+
+static struct kunit_suite kwatch_deref_test_suite = {
+	.name = "kwatch_deref",
+	.test_cases = kwatch_deref_test_cases,
+};
+
+kunit_test_suite(kwatch_deref_test_suite);
+
+MODULE_DESCRIPTION("KUnit tests for the KWatch watch expression parser");
+MODULE_LICENSE("GPL");
-- 
2.53.0


^ permalink raw reply related

* Re: [PATCH] scripts/kernel-doc: Suggest possible names for excess descriptions
From: Knop, Ryszard @ 2026-07-17 13:07 UTC (permalink / raw)
  To: corbet@lwn.net, linux-doc@vger.kernel.org
  Cc: intel-xe@lists.freedesktop.org, Lin, Shuicheng,
	rdunlap@infradead.org, jani.nikula@linux.intel.com,
	linux-kernel@vger.kernel.org
In-Reply-To: <87tspy5zjy.fsf@trenco.lwn.net>

On Thu, 2026-07-16 at 16:40 -0600, Jonathan Corbet wrote:
> Ryszard Knop <ryszard.knop@intel.com> writes:
> 
> > Since check_sections() now warns if a documentation tag member name is
> > the same as defined in the struct, we can suggest names the checker
> > knows, so that it's more obvious how to deal with the warning.
> > 
> > Signed-off-by: Ryszard Knop <ryszard.knop@intel.com>
> > ---
> >  tools/lib/python/kdoc/kdoc_parser.py | 13 +++++++++++--
> >  1 file changed, 11 insertions(+), 2 deletions(-)
> 
> So I almost applied this version of this patch.  For future reference,
> please:
> 
> - Post new versions standalone, not as a reply
> 
> - Do not include version information in the changelog (put it below the
>   "---" line)
> 
> - Provide a more coherent changelog; the one above does not say what is
>   actually going on here.
> 
> - CC the maintainer (me).

Submitted a v3 with changes Mauro requested, fixed changelog and CC'd,
but accidentally sent it again with In-Reply-To which I had in the
command last time, sorry :/

> 
> Thanks,
> 
> jon

Thanks, Ryszard

^ permalink raw reply

* [RFC PATCH v2 13/13] Documentation/dev-tools: document KWatch
From: Jinchao Wang @ 2026-07-17 13:07 UTC (permalink / raw)
  To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
	Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

Describe what KWatch is for, how it compares with KASAN and KFENCE,
the debugfs configuration interface, the watch expression syntax,
how to read hits from the trace buffer (including after a crash),
and the current limitations.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
 Documentation/dev-tools/index.rst  |   1 +
 Documentation/dev-tools/kwatch.rst | 207 +++++++++++++++++++++++++++++
 2 files changed, 208 insertions(+)
 create mode 100644 Documentation/dev-tools/kwatch.rst

diff --git a/Documentation/dev-tools/index.rst b/Documentation/dev-tools/index.rst
index 59cbb77b33ff..f4c748da63db 100644
--- a/Documentation/dev-tools/index.rst
+++ b/Documentation/dev-tools/index.rst
@@ -30,6 +30,7 @@ Documentation/process/debugging/index.rst
    ubsan
    kmemleak
    kcsan
+   kwatch
    lkmm/index
    kfence
    kselftest
diff --git a/Documentation/dev-tools/kwatch.rst b/Documentation/dev-tools/kwatch.rst
new file mode 100644
index 000000000000..e58f3185ebbd
--- /dev/null
+++ b/Documentation/dev-tools/kwatch.rst
@@ -0,0 +1,207 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+======================================
+KWatch - Kernel Memory Watchpoint Tool
+======================================
+
+Overview
+========
+
+KWatch is a runtime-configurable debugging tool for locating kernel memory
+corruption. It arms hardware breakpoints (watchpoints) on a target address
+while a chosen function is executing, and reports the exact instruction that
+touches the watched memory, together with a stack trace, through a
+tracepoint.
+
+Unlike shadow-memory sanitizers, KWatch does not detect invalid accesses in
+general; it answers a narrower but common question during corruption hunts:
+"who writes to this address?". This includes in-bounds logical overwrites
+that KASAN cannot see, because the rogue writer modifies valid memory
+through a valid pointer, just at the wrong time or with the wrong data.
+
+Comparison with other tools:
+
+* KASAN detects out-of-bounds and use-after-free accesses, but reports the
+  symptom (the invalid access), not the writer that corrupted the data
+  earlier. It requires a rebuild and has significant CPU and memory
+  overhead, and its redzones perturb memory layout, which can hide
+  timing-sensitive bugs.
+* KFENCE is a low-overhead sampling detector for slab objects; it cannot be
+  pointed at one specific address.
+* Hardware breakpoints via kgdb or perf can watch an address, but only a
+  fixed one, system-wide, for the whole run. KWatch resolves the address
+  dynamically at function entry (for example "argument 2 of this function,
+  plus offset 8, dereferenced once") and disarms it again at function exit,
+  so short-lived and per-invocation objects can be watched too.
+
+KWatch has near-zero overhead while armed: the watched function pays for
+one kprobe/kretprobe pair plus programming of the debug registers; the rest
+of the system runs at full speed.
+
+Requirements
+============
+
+* ``CONFIG_KWATCH=y`` or ``m``. The Kconfig symbol depends on
+  ``CONFIG_PERF_EVENTS``, ``CONFIG_DEBUG_FS`` and an architecture that
+  provides ``HAVE_REINSTALL_HW_BREAKPOINT`` (currently x86 only).
+* Resolving symbol names in watch expressions requires ``CONFIG_KWATCH=y``
+  (built-in); a module can only watch absolute hexadecimal addresses.
+
+Usage
+=====
+
+KWatch is configured through a single debugfs file::
+
+    /sys/kernel/debug/kwatch/config
+
+Writing a configuration string starts a watch session (stopping any previous
+one); reading the file shows the active configuration and hit-rejection
+counters. The configuration is a whitespace-separated list of ``key=value``
+tokens:
+
+=================== ==========================================================
+Key                 Meaning
+=================== ==========================================================
+``func_name``       Function whose execution opens the watch window.
+``func_offset``     Instruction offset inside ``func_name`` at which the
+                    watchpoint is armed (default 0 = function entry).
+``watch_expr``      Expression describing the address to watch (see below).
+``watch_len``       Watched length in bytes: 1, 2, 4 or 8 (default 8).
+``depth``           Recursion depth at which the window opens (default 0).
+``max_watch``       Number of hardware watchpoints to preallocate
+                    (default 4).
+``max_concurrency`` Maximum number of tasks concurrently inside the watch
+                    window (default 256).
+``duration``        For global watches: seconds until automatic stop.
+=================== ==========================================================
+
+Watch expressions
+-----------------
+
+The address to watch is computed at function entry from::
+
+    watch_expr={base}[+-offset][->[+-]offset]...
+
+* ``base`` is one of:
+
+  - ``arg1`` ... ``arg6``: a function argument (register calling
+    convention). These are only meaningful at function entry, i.e. with
+    ``func_offset`` unset; combining ``argN`` with ``func_offset`` reads
+    the argument registers mid-function, where they no longer hold the
+    original arguments,
+  - ``stack``: the kernel stack pointer at the probe point,
+  - an absolute hexadecimal address, e.g. ``0xffffffff81234567``,
+  - a global symbol name (built-in KWatch only).
+
+* ``+offset`` / ``-offset`` adjusts the current address.
+* ``->offset`` loads the pointer stored at the current address (via
+  ``get_kernel_nofault()``) and then applies the offset. Up to four chain
+  elements are supported; offsets must be explicit (``->`` alone is
+  rejected).
+
+Given::
+
+    struct some_struct {
+        struct some_struct *ptr;    /* offset 0 */
+        int num;                    /* offset 8 */
+    };
+
+    void target_function(struct some_struct *arg1);
+
+typical expressions are:
+
+=========================== ==============================================
+Expression                  Watches
+=========================== ==============================================
+``watch_expr=arg1``         ``&arg1->ptr`` (the pointer field itself)
+``watch_expr=arg1+8``       ``&arg1->num``
+``watch_expr=arg1->0``      ``&arg1->ptr->ptr`` (one dereference)
+``watch_expr=arg1->8``      ``&arg1->ptr->num``
+``watch_expr=0xffff...+8``  absolute address plus 8
+=========================== ==============================================
+
+Example: catch whoever overwrites ``arg1->num`` of a function while that
+function runs::
+
+    echo "func_name=target_function watch_expr=arg1+8 watch_len=4" \
+        > /sys/kernel/debug/kwatch/config
+
+Watching global variables
+-------------------------
+
+A global variable has no natural function window. When ``duration`` is
+given without ``func_name``, KWatch starts an internal anchor kernel thread
+that sleeps inside a dummy function, and uses that function as the window::
+
+    echo "watch_expr=jiffies_wobble duration=60 watch_len=8" \
+        > /sys/kernel/debug/kwatch/config
+
+The session tears itself down when the duration expires.
+
+Reading hits
+------------
+
+Hits are emitted as the ``kwatch:kwatch_hit`` tracepoint, which is safe in
+NMI-like contexts where printk is not. Each event carries the timestamp,
+the instruction pointer, the watched address and a short stack trace::
+
+    echo 1 > /sys/kernel/debug/tracing/events/kwatch/kwatch_hit/enable
+    cat /sys/kernel/debug/tracing/trace_pipe
+
+If the corruption crashes the machine, the ring buffer can still be
+recovered:
+
+* ``echo 1 > /proc/sys/kernel/ftrace_dump_on_oops`` (or the
+  ``ftrace_dump_on_oops`` boot parameter) dumps the buffer to the console
+  on an oops.
+* With kdump, the buffer is present in the vmcore and can be read with
+  ``crash> trace``.
+* ``CONFIG_PSTORE_FTRACE`` persists it across reboots on supported
+  platforms.
+
+Limitations
+===========
+
+* Functions that run in a genuine NMI(-like) context are rejected at
+  function entry; rejected invocations never open a watch window and are
+  counted in the ``nmi_rejected`` field of the config file. Watching
+  functions reachable from NMI handlers is out of scope.
+* The number of concurrent watchpoints is bounded by the CPU's debug
+  registers (typically 4).
+* Cross-CPU re-arming of a watchpoint is rate-limited per CPU: the local
+  CPU is always re-pointed, but the broadcast to other CPUs is throttled
+  so a very hot watched function cannot storm the system with IPIs. While
+  a broadcast is suppressed the other CPUs keep watching the previous
+  address, so a writer that runs on another CPU during that window can be
+  missed; the number of suppressed broadcasts is reported in the
+  ``arm_ipi_suppressed`` field of the config file. KWatch therefore
+  targets functions that are entered at a moderate rate.
+* If the target address cannot be resolved at arming time (for example a
+  ``get_kernel_nofault()`` failure on a swapped or unmapped page), the
+  watchpoint is not armed for that invocation.
+* Offsets in watch expressions are static; dynamic indexing such as
+  ``arg1->ptr[arg2]`` is not supported.
+* If a task is torn down while still inside the watched function without
+  the function returning (an oops or BUG in the window, which abandons the
+  stack), its watch window is not closed until the session stops. This is
+  the same best-effort cleanup that applies to every resource a task holds
+  when it dies abnormally. Do not target the task-exit path itself.
+* arm64 is not yet supported: stepping over a hit that has a custom
+  overflow handler needs a generic mechanism in the arch code, which is
+  planned as a follow-up series.
+
+Implementation notes
+====================
+
+The implementation lives in ``mm/kwatch/`` and is split into a control
+plane (``core.c``, the debugfs interface), an execution plane (``probe.c``
+and ``deref.c``: kprobe/kretprobe window management and address
+resolution), and a resource plane (``hwbp.c`` and ``task_ctx.c``).
+
+Hardware watchpoints are preallocated as perf events on every CPU and
+re-pointed at hit time with ``modify_wide_hw_breakpoint_local()``, a new
+hw_breakpoint API that updates the breakpoint on the local CPU without
+releasing its slot; other CPUs are updated by asynchronous IPIs. Per-task
+window state is kept in a fixed-size, lockless open-addressing array
+claimed with ``cmpxchg()``, so the hit path performs no allocation and
+takes no locks, which keeps it safe in atomic and NMI-like contexts.
-- 
2.53.0


^ permalink raw reply related

* Re:
From: Jonathan Corbet @ 2026-07-17 13:08 UTC (permalink / raw)
  To: Daniel Pereira; +Cc: linux-doc
In-Reply-To: <CAMAsx6cO588DcqAT9c=-7aXWLwzAbjrAp=jJ9Ya9dC7draaPZA@mail.gmail.com>

Daniel Pereira <danielmaraboo@gmail.com> writes:

> Regarding the conflict issue, I would be happy to take on the
> responsibility of organizing these patches to help save you time.
> Would you prefer if I acted as a point of contact for these
> translations? I could collect submissions from contributors, merge
> them into the structure correctly, and then send them to you as a
> single, cleaned-up set of patches.
>
> Please let me know if this workflow would be helpful or if you have a
> different preference for how I should handle these conflicts.

I think that if you organize the index.rst files as I suggested
(i.e. roughly along the lines of the English versions) most of the
conflicts will go away, and the result will better generally.  That's
the best place to start, I think.

Thanks,

jon

^ permalink raw reply

* Re: [PATCH v2] scripts/kernel-doc: Suggest possible names for excess descriptions
From: Knop, Ryszard @ 2026-07-17 13:17 UTC (permalink / raw)
  To: mchehab+huawei@kernel.org
  Cc: intel-xe@lists.freedesktop.org, Lin, Shuicheng,
	linux-doc@vger.kernel.org, rdunlap@infradead.org,
	jani.nikula@linux.intel.com, linux-kernel@vger.kernel.org
In-Reply-To: <20260715155247.3b9fb363@localhost>

On Wed, 2026-07-15 at 15:52 +0200, Mauro Carvalho Chehab wrote:
> On Wed, 15 Jul 2026 13:21:27 +0000
> "Knop, Ryszard" <ryszard.knop@intel.com> wrote:
> 
> > On Wed, 2026-07-15 at 14:42 +0200, Mauro Carvalho Chehab wrote:
> > > On Wed, 15 Jul 2026 13:17:26 +0200
> > > Ryszard Knop <ryszard.knop@intel.com> wrote:
> > >   
> > > > Since check_sections() now warns if a documentation tag member name is
> > > > the same as defined in the struct, we can suggest names the checker
> > > > knows, so that it's more obvious how to deal with the warning.
> > > > 
> > > > v2 (rdunlap):
> > > > - Strip whitespace from warnings, nicer when the hint is empty
> > > > 
> > > > Signed-off-by: Ryszard Knop <ryszard.knop@intel.com>
> > > > ---
> > > >  tools/lib/python/kdoc/kdoc_parser.py | 13 +++++++++++--
> > > >  1 file changed, 11 insertions(+), 2 deletions(-)
> > > > 
> > > > diff --git a/tools/lib/python/kdoc/kdoc_parser.py b/tools/lib/python/kdoc/kdoc_parser.py
> > > > index 2dedda215c22..a22c3e3182f0 100644
> > > > --- a/tools/lib/python/kdoc/kdoc_parser.py
> > > > +++ b/tools/lib/python/kdoc/kdoc_parser.py
> > > > @@ -558,6 +558,13 @@ class KernelDoc:
> > > >                          self.push_parameter(ln, decl_type, param, dtype,
> > > >                                              arg, declaration_name)
> > > >  
> > > > +    def get_suggestions_hint(self, decl_name, possible_names):
> > > > +        suggestions = set(name for name in possible_names if decl_name in name)
> > > > +        if not suggestions:
> > > > +            return ""
> > > > +
> > > > +        return f"(did you mean one of: '{"', '".join(suggestions)}')"
> > > > +  
> > > 
> > > There is a better way to propose suggestions. See:
> > > 	Documentation/sphinx/kernel_include.py
> > > 
> > > E.g. use something like:
> > > 
> > > 	from difflib import get_close_matches
> > > 
> > > 	matches = get_close_matches(decl_name, possible_names)
> > > 
> > > See: https://docs.python.org/3/library/difflib.html#difflib.get_close_matches
> > > 
> > > If the problem is due to a typo, this will likely return the
> > > right name.  
> > 
> > The checks here specifically were added to deal with situations like
> > [1] which boils down to:
> > 
> > struct {
> >     /** @flags: good description */
> >     int flags;
> > 
> >     /** @substruct: also good */
> >     struct {
> >         /** @mode: bad, wrong, no good */
> >         int mode;
> >     } substruct;
> > } big_block_o_data;
> > 
> > The docs should say "@substruct.mode" instead of just "@mode", so this
> > is distant enough from the actual input that difflib would not suggest
> > it. 
> 
> Ok, but there should be cases like, instead of "mode", someone writes
> for instance "modes".
> 
> > I could merge suggestions from both difflib and the plain substring
> > comparison if you'd like me to?
> 
> Makes sense to me. Just ensure that they aren't duplicated.

Submitted v3 with slightly more complex suggestions set up like this:

- First, we suggest nested struct names. For "substruct.member", we
compare the kdoc declaration name with "member" after the last dot.
Exact matches go first, then substrings, then the difflib suggestion
(so that 'flgas' still matches 'substruct.flags').
- Then we compare decl name on the full known possible member name,
first with substrings and then with difflib again.
- All that gets deduplicated and merged in the order listed above.

Link to v3:
https://lore.kernel.org/linux-doc/20260717125753.634550-1-ryszard.knop@intel.com/

> 
> > 
> > [1] https://patchwork.freedesktop.org/patch/734307/?series=168905&rev=1
> > 
> > > 
> > > Regards,
> > > Mauro  
> > 
> > Thanks, Ryszard
> 

Thanks, Ryszard

^ permalink raw reply

* Re: [RFC PATCH v1 0/8] Arm Core Local Accelerator Driver
From: Jason Gunthorpe @ 2026-07-17 13:32 UTC (permalink / raw)
  To: Will Deacon
  Cc: Ryan Roberts, Greg Kroah-Hartman, Arnd Bergmann, Catalin Marinas,
	Mark Rutland, Jean-Philippe Brucker, Oded Gabbay, Jonathan Corbet,
	linux-kernel, linux-arm-kernel, dri-devel, linux-doc, maz, oupton,
	hch
In-Reply-To: <aloS_T3LQvpHeZUV@willie-the-truck>

On Fri, Jul 17, 2026 at 12:33:17PM +0100, Will Deacon wrote:

> first. So, at the moment, this just looks like a burden to me, especially
> as it appears to create a brand new, device-specific UAPI for what is
> ostensibly a form of SVA - something which the community is actively
> working on already.

Yeah, I think it was a mistake to hardwire the invalidation to the
CPU. It would fit much better into Linux if the invalidation was
separate and independently controllable with an option to follow the
CPU.

Then you could use a normal S2 page table through iommufd and not mess
with kvm. (though the VMID sharing with KVM for a BTM-like scheme was
never solved upstream)

I also wouldn't expect you to use vfio-mdev, if the devices are
discovered and known at boot then it should be a normal vfio driver.

Jason

^ permalink raw reply

* Re: [RFC PATCH v2 00/13] mm/kwatch: dynamic hardware watchpoints for hunting memory corruption
From: Dave Hansen @ 2026-07-17 13:41 UTC (permalink / raw)
  To: Jinchao Wang, Andrew Morton, Peter Zijlstra, Thomas Gleixner,
	Steven Rostedt, Masami Hiramatsu
  Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
	Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
	Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
	Matthew Wilcox, Alan Stern, Randy Dunlap, Alexander Potapenko,
	Marco Elver, Mike Rapoport, linux-kernel, linux-mm,
	linux-trace-kernel, linux-perf-users, linux-doc
In-Reply-To: <20260717125023.1895892-1-wangjinchao600@gmail.com>

On 7/17/26 05:50, Jinchao Wang wrote:
>  24 files changed, 2115 insertions(+), 63 deletions(-)
Reading this, I wonder how many kernel debugging features we need. I
don't even think we have a centralized list of them. They all just live
in their own silos.

This one really seems like a super specialized tool. It has to be
enabled at compile time and specifically aimed at a specific function.

Maybe this should live off on the side for a while. If folks end up
actually needing it, they can point their friendly LLM over to its tree.

^ permalink raw reply

* Re: [RFC PATCH v1 0/8] Arm Core Local Accelerator Driver
From: Ryan Roberts @ 2026-07-17 13:42 UTC (permalink / raw)
  To: Jason Gunthorpe, Will Deacon
  Cc: Greg Kroah-Hartman, Arnd Bergmann, Catalin Marinas, Mark Rutland,
	Jean-Philippe Brucker, Oded Gabbay, Jonathan Corbet, linux-kernel,
	linux-arm-kernel, dri-devel, linux-doc, maz, oupton, hch
In-Reply-To: <20260717133200.GB699082@nvidia.com>

On 17/07/2026 14:32, Jason Gunthorpe wrote:
> On Fri, Jul 17, 2026 at 12:33:17PM +0100, Will Deacon wrote:
> 
>> first. So, at the moment, this just looks like a burden to me, especially
>> as it appears to create a brand new, device-specific UAPI for what is
>> ostensibly a form of SVA - something which the community is actively
>> working on already.
> 
> Yeah, I think it was a mistake to hardwire the invalidation to the
> CPU. It would fit much better into Linux if the invalidation was
> separate and independently controllable with an option to follow the
> CPU.

Well I'm certainly not going to disagree, but the HW is somewhat fixed at this
point :-|

> 
> Then you could use a normal S2 page table through iommufd and not mess
> with kvm. (though the VMID sharing with KVM for a BTM-like scheme was
> never solved upstream)
> 
> I also wouldn't expect you to use vfio-mdev, if the devices are
> discovered and known at boot then it should be a normal vfio driver.

OK good feedback - I'm not going to claim to be a VFIO expert (especially not in
front of an actual VFIO expert) - We'll look closer at this and come back if we
have questions (although unlikely over the summer holiday).

Thanks,
Ryan


> 
> Jason


^ permalink raw reply

* Re: [RFC PATCH v1 1/8] misc/arm-cla: Add driver skeleton and documentation
From: Arnd Bergmann @ 2026-07-17 13:49 UTC (permalink / raw)
  To: Ryan Roberts, Greg Kroah-Hartman, Catalin Marinas, Will Deacon,
	Mark Rutland, Jean-Philippe Brucker, Oded Gabbay, Jonathan Corbet
  Cc: linux-kernel, linux-arm-kernel, dri-devel, linux-doc
In-Reply-To: <20260717104759.123203-2-ryan.roberts@arm.com>

On Fri, Jul 17, 2026, at 12:47, Ryan Roberts wrote:
> From: Jean-Philippe Brucker <jpb@kernel.org>
>
> Add the initial Kconfig and build-system plumbing for the Arm Core Local
> Accelerator driver.
>
> Introduce the common driver header and register definitions used by
> later CLA support. The definitions cover the CLA MMIO frame, launch
> response and status fields, standard accelerator registers, launch
> opcodes, error codes and memory translation context state.
>
> Add documentation describing the CLA programming model, its CPU-local
> MMIO access rules, userspace assignment model, domain grouping and
> expected boot state.

I have a few more questions here. Most of the description and
the design decisions make perfect sense to me, but there are
a few things I don't understand from your current document.

> +The CLA supports up to 8 attached accelerators, which are accessed by
> +programming the CLA's MMIO registers. Operations are launched to an 
> accelerator
> +and are polled for completion. CLA does not raise interrupts.
> +
> +            CPU                     CLA              Accel
> +             |--- write DATA[7:0] -->|                 |
> +             |--- write LAUNCH ----->|---- launch ---->|
> +             |<--- poll LRESP -------|                 |
> +             |                       |                 |
> +             |<--- poll STATUS ------|<--- complete ---|
> +
> +Each operation can take a 512-bit payload in the DATA registers. After handling
> +a LAUNCH write, CLA indicates the launch status in the LRESP register. A further
> +operation can only be launched after LRESP indicates completion of the previous
> +launch.

This sounds a lot like st64bv or st64bv0, passing an 8-word payload and returning
a single word per accelerator operation with shared addressing.

Why are there now two interfaces to do the same thing?

Can a user process use st64bv to do the four steps in a
single instruction?

> +Some operations continue to run asynchronously on the accelerator after launch
> +completion. In this case progress is tracked by polling the STATUS register.
> +When the CLA updates the STATUS register, it also raises an event which will
> +wake an in-progress WFE (wait for event) instruction on the local CPU.

The asynchronous interface seems very confusing.

How are page faults from the SVA master resolved during an
asynchronous operation?

Can a CPU start multiple asynchronous operations concurrently?

Do these continue to run if the starting process is scheduled out
and another process also tries to use CLA?

> +Faults during address translation are reported by the accelerator in its
> +registers and in STATUS. While polling for work completion, software fixes up
> +the faults and notifies the accelerator with RESOLVE operations.

I would like to understand the faulting part better. Which instruction
specifically causes the fault, is that the poll STATUS read?

> +Inter-Accelerator Communication
> +-------------------------------
> +
> +On some platforms, multiple accelerators, each attached to a separate CLA within
> +a cluster, are also directly connected to each other via a shared bus to
> +accelerate cooperation between accelerators. The accelerators sharing a bus
> +cannot be isolated from each other. When collaborative operations are launched
> +on each of the participating accelerators, they synchronize over the bus,
> +stalling until all are ready.

Could you give an example what this model might be good for?

Does this mean a user may have to start one operation on each CPU
from a thread of the same process in order to get a result efficiently
across a shared accelerator?

I assume this will become clearer once you can show an example userspace
application that uses this type of accelerator.

> +Intended SW Usage Model
> +=======================
> +
> +CLA is designed for its PL0 MMIO frame to be mapped into user space  and for user
> +space to directly launch accelerator operations and poll for completion. It has
> +been observed that for some use cases, the operation execution time is small and
> +a trip through the kernel would consume a significant amount of the CPU budget
> +for preparing the next operation leading to a significant reduction in bandwidth
> +through the accelerator.

Do you have any plans for in-kernel usage of the accelerators?
I would assume that for things like cryptographic features, these
make sense to be exposed to the kernel itself.

> +User space software is expected to create a thread to drive each CLA it is
> +using, and for each thread to be pinned to the CLA's local CPU.

What happens if multiple processes have the same chardev open and
each mmap() that, e.g. after a fork()? Does each process see its
own virtual instance of the accelerator and interact with it through
the same physical MMIO register range but its own process address space,
or do you have to rely on the registers being mapped only into a
single mm_struct to prevent a process from messing with another process
data?

> +Saving and restoring the internal state of the accelerator is an optional
> +feature. Current platforms only support it when the accelerator is idle, so
> +preempting an accelerator causes work cancellation. Software must carefully
> +consider how to balance forward-progress guarantees with preemption 
> latency.

I'm not sure I understand this point. Do you mean any async operation
that was started on an accelerator may fail due to preemption, so user
space must be able to restart it?

     Arnd

^ permalink raw reply

* Re: [PATCH v6 1/5] spi: dt-bindings: Add spi-device-addr peripheral property
From: Nuno Sá @ 2026-07-17 13:52 UTC (permalink / raw)
  To: Mark Brown
  Cc: Conor Dooley, Janani Sunil, Lars-Peter Clausen, Michael Hennerich,
	Jonathan Cameron, David Lechner, Nuno Sá, Andy Shevchenko,
	Rob Herring, Krzysztof Kozlowski, Conor Dooley, Philipp Zabel,
	Jonathan Corbet, Shuah Khan, Marius Cristea, Marcus Folkesson,
	Kent Gustavsson, linux-iio, devicetree, linux-kernel, linux-doc,
	Janani Sunil, linux-spi, Kent Gustavsson
In-Reply-To: <27746e91-ab2b-40a0-822b-6d3a8997cefc@sirena.org.uk>

On Fri, Jul 17, 2026 at 12:51:53PM +0100, Mark Brown wrote:
> On Thu, Jul 16, 2026 at 06:06:13PM +0100, Conor Dooley wrote:
> > On Thu, Jul 16, 2026 at 01:28:15PM +0100, Mark Brown wrote:
> 
> > > Oh, isn't that just multi-pin chip selects then:
> 
> > >    https://patch.msgid.link/cover.1783729282.git.Jonathan.Santos@analog.com
> 
> > No, I think that's something different.
> 
> > In this case, there is one chip select that all instances of the device
> > share. Both the ADI and Microchip devices then use the upper bits of the
> > first/address byte during reads and write to access individual devices.
> > In the microchip case that I'm familiar with, what upper bits the device
> > responds to are set by fuses in the factory. They also share the same MOSI
> > and MISO.
> 
> I'm struggling to identify a way in which this is functionally different
> to a multi-pin chip select.
> 
> > The SPI core already supports two of these devices from Microchip, since
> > it just modifies the contents of tx_buf in a spi_transfer.
> 
> When I said we'd have to modify the bitstream I was told that wasn't the
> case and this was just asserting more pins along with the "chip select".

Oh sorry,

Likely I was not clear. The part as some ID pins which are just
configuration! So they are two pins and depending on how they are set
(low/high) each chip can have four possible addresses! Now all of the chips
(to a max of 4) are connected to same CS of the controller so just one CS get's
asserted.

Of course the above will assert on all devices on that same CS but only
the one where the ID pins combo match the spi xfer high bits (part of
the address) will reply.

Hopefully the above makes it clear.

- Nuno Sá

^ permalink raw reply

* [PATCH v9 00/10] tracing: wprobe: x86: Add wprobe for watchpoint
From: Masami Hiramatsu (Google) @ 2026-07-17 14:19 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users

Hi,

Here is the 9th version of the series for adding new wprobe (watch probe)
which provides memory access tracing event. Moreover, this can be used
via event trigger. Thus it can trace memory access on a dynamically
allocated objects too.
The previous version is here:

  https://lore.kernel.org/all/178417033089.209165.16717079876036408877.stgit@devnote2/

This version fixes issues commented by Sashiko[1] and add BTF typecast
support [10/10]. Let me list it up.

[1] https://sashiko.dev/#/patchset/178417033089.209165.16717079876036408877.stgit%40devnote2

[PATCH 1/10] 
      - Rebased on probes/for-next branch.
      - Add WPROBE_NO_SIBLING error log to explicitly reject sibling probes
        since event triggers identify the target wprobe by event name.
      - Use traceprobe_parse_event_name() to properly validate group/event
        names instead of using the raw command string directly.
      - Generate unique event name (w_0x<addr>) for anonymous address-based
        watchpoints to avoid naming collisions.
      - Call traceprobe_update_arg() in __register_trace_wprobe() to resolve
        @symbol fetch arguments, consistent with kprobe and fprobe.
[PATCH 3/10]
      - Fix command check logic to prevent early exit under 'set -e'
        (errexit) when grep or test fails.
      - Simplify enable/disable status checks by removing cat pipes.
[PATCH 5/10]
      - Define DR_LEN_MASK and DR_TYPE_MASK locally to eliminate magic
        numbers in decode_dr7() and setup_hwbp().
      - Temporarily disable the active slot in setup_hwbp() before updating
        the address register to avoid spurious debug exceptions.
[PATCH 6/10]
      - Update commit message.
[PATCH 7/10]
      - Add lockdep_assert_irqs_disabled() to enforce and document the
        interrupt-disabled calling context requirement.
      - Implement state rollback logic in modify_wide_hw_breakpoint_local()
        to prevent inconsistent state on architecture update failure.
[PATCH 8/10]
      - Make event_trigger_free() non-static to solve build dependency.
      - Sync irq_work and work inside __unregister_trace_wprobe() before
        clearing tw->bp_event to prevent concurrent NULL pointer dereference.
      - Introduce private_free destructor in struct event_trigger_data to
        safely release wprobe_trigger_data after tracepoint readers exit.
      - Fix filter memory leak in wprobe_trigger_cmd_parse() on the error
        handling path of trace_event_try_get_ref() failure.
      - Avoid overwriting tw->addr on clear_wprobe trigger registration to
        prevent active watchpoint corruption and hardware breakpoint leak.
[PATCH 9/10]
      - Drop dentry cache after removing tempfile for on-disk
        filesystem.

To support kprobe events, we need to check the previous NMI context
or need to use an atomic refcount etc. But that should be another
patch for review.

To support arm64, we need to avoid major pagefault on single-stepping
issue. I think we can solve it with checking whether the address is the
kernel (which must not cause a major page fault), or not.

Usage
-----

The basic usage of this wprobe is similar to other probes;

  w:[GRP/][EVENT] [r|w|rw]@<ADDRESS|SYMBOL[+OFFS]> [FETCHARGS]

This defines a new wprobe event. For example, to trace jiffies update,
you can do;

 echo 'w:my_jiffies w@jiffies:8 value=+0($addr)' >> dynamic_events
 echo 1 > events/wprobes/my_jiffies/enable

Moreover, this can be combined with event trigger to trace the memory
accecss on slab objects. The trigger syntax is;

  set_wprobe:WPROBE_EVENT:FIELD[+OFFSET] [if FILTER]
  clear_wprobe:WPROBE_EVENT[:FIELD[+OFFSET]] [if FILTER]

set_wprobe sets WPROBE_EVENT's watch address on FIELD[+OFFSET].
clear_wprobe clears WPROBE_EVENT's watch address if it is set to
FIELD[+OFFSET]. If FIELD is omitted, forcibly clear the watch address
when trigger event is hit.

For example, trace the first 8 byte of the dentry data structure passed
to do_truncate() until it is deleted by dentry_kill().
(Note: all tracefs setup uses '>>' so that it does not kick do_truncate())

  # echo 'w:watch rw@0:8 address=$addr value=+0($addr)' > dynamic_events

  # echo 'f:truncate do_truncate dentry=$arg2' >> dynamic_events
  # echo 'set_wprobe:watch:dentry' >> events/fprobes/truncate/trigger

  # echo 'f:dentry_kill dentry_kill dentry=$arg1' >> dynamic_events
  # echo 'clear_wprobe:watch:dentry' >> events/fprobes/dentry_kill/trigger

  # echo 1 >> events/fprobes/truncate/enable
  # echo 1 >> events/fprobes/dentry_kill/enable

  # echo aaa > /tmp/hoge
  # echo bbb > /tmp/hoge
  # echo ccc > /tmp/hoge
  # rm /tmp/hoge

Then, the trace data will show;

 # tracer: nop
 #
 # entries-in-buffer/entries-written: 32/32   #P:8
 #
 #                                _-----=> irqs-off/BH-disabled
 #                               / _----=> need-resched
 #                              | / _---=> hardirq/softirq
 #                              || / _--=> preempt-depth
 #                              ||| / _-=> migrate-disable
 #                              |||| /     delay
 #           TASK-PID     CPU#  |||||  TIMESTAMP  FUNCTION
 #              | |         |   |||||     |         |
               sh-107     [004] ...1.     9.990418: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004ad6618
               sh-107     [004] ...1.     9.990914: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b3de78
               sh-107     [004] ...1.     9.993175: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049ddd40
               sh-107     [004] .....     9.995198: truncate: (do_truncate+0x4/0x120) dentry=0xffff8880048083a8
               sh-107     [004] ...1.     9.995389: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db998
               sh-107     [004] ..Zff     9.997503: watch: (lookup_fast+0xaa/0x150) address=0xffff8880048083a8 value=0x8200080
               sh-107     [004] ..Zff     9.997509: watch: (path_openat+0x211/0xda0) address=0xffff8880048083a8 value=0x8200080
               sh-107     [004] ..Zff     9.997514: watch: (path_openat+0xa56/0xda0) address=0xffff8880048083a8 value=0x8200080
               sh-107     [004] ..Zff     9.997518: watch: (path_openat+0xae2/0xda0) address=0xffff8880048083a8 value=0x8200080
               sh-107     [004] .....     9.997521: truncate: (do_truncate+0x4/0x120) dentry=0xffff8880048083a8
               sh-107     [004] ...1.     9.997582: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004808270
               sh-107     [004] ...1.     9.999365: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db728
               sh-107     [004] ...1.     9.999388: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b1c000
               rm-113     [005] ..Zff    10.000965: watch: (lookup_fast+0xaa/0x150) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.000971: watch: (path_lookupat+0x97/0x1e0) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.000984: watch: (lookup_fast+0xaa/0x150) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.000988: watch: (path_lookupat+0x97/0x1e0) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.001010: watch: (lookup_one_qstr_excl+0x28/0x140) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.001014: watch: (lookup_one_qstr_excl+0xd1/0x140) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.001018: watch: (may_delete_dentry+0x1c/0x200) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.001021: watch: (may_delete_dentry+0x195/0x200) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] ..Zff    10.001031: watch: (vfs_unlink+0x5e/0x260) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] d.Z..    10.001067: watch: (d_make_discardable+0x1b/0x40) address=0xffff8880048083a8 value=0x8200080
               rm-113     [005] d.Z..    10.001071: watch: (d_make_discardable+0x29/0x40) address=0xffff8880048083a8 value=0x200080
               rm-113     [005] ...1.    10.001072: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880048083a8
               rm-113     [005] ...1.    10.001218: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880048083a8
               sh-107     [004] ...1.    10.001416: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db110
               sh-107     [004] ...1.    10.001444: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db248
               sh-107     [004] ...1.    10.001500: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004ad6618
               sh-107     [004] ...1.    10.002067: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b41e78
               sh-107     [004] ...1.    10.904920: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b41e78
               sh-107     [004] ...1.    10.905129: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004ad6618

Thank you,

---

Jinchao Wang (2):
      x86/hw_breakpoint: Unify breakpoint install/uninstall
      x86/hw_breakpoint: Add arch_reinstall_hw_breakpoint

Masami Hiramatsu (Google) (8):
      tracing: wprobe: Add watchpoint probe event based on hardware breakpoint
      x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires
      selftests: tracing: Add a basic testcase for wprobe
      selftests: tracing: Add syntax testcase for wprobe
      HWBP: Add modify_wide_hw_breakpoint_local() API
      tracing: wprobe: Add wprobe event trigger
      selftests: ftrace: Add wprobe trigger testcase
      tracing/wprobe: Support BTF typecast in fetchargs


 Documentation/trace/index.rst                      |    1 
 Documentation/trace/wprobetrace.rst                |  163 +++
 arch/Kconfig                                       |   20 
 arch/x86/Kconfig                                   |    2 
 arch/x86/include/asm/hw_breakpoint.h               |    8 
 arch/x86/kernel/hw_breakpoint.c                    |  164 ++-
 include/linux/hw_breakpoint.h                      |    6 
 include/linux/trace_events.h                       |    3 
 kernel/events/hw_breakpoint.c                      |   61 +
 kernel/trace/Kconfig                               |   24 
 kernel/trace/Makefile                              |    1 
 kernel/trace/trace.c                               |    9 
 kernel/trace/trace.h                               |    7 
 kernel/trace/trace_events_trigger.c                |   12 
 kernel/trace/trace_probe.c                         |   24 
 kernel/trace/trace_probe.h                         |   14 
 kernel/trace/trace_wprobe.c                        | 1185 ++++++++++++++++++++
 tools/testing/selftests/ftrace/config              |    2 
 .../ftrace/test.d/dynevent/add_remove_wprobe.tc    |   63 +
 .../test.d/dynevent/wprobes_syntax_errors.tc       |   20 
 .../test.d/trigger/trigger-wprobe-btf-typecast.tc  |   80 +
 .../ftrace/test.d/trigger/trigger-wprobe.tc        |   70 +
 22 files changed, 1868 insertions(+), 71 deletions(-)
 create mode 100644 Documentation/trace/wprobetrace.rst
 create mode 100644 kernel/trace/trace_wprobe.c
 create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
 create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
 create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-typecast.tc
 create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc

--
Masami Hiramatsu (Google) <mhiramat@kernel.org>

^ permalink raw reply

* [PATCH v9 01/10] tracing: wprobe: Add watchpoint probe event based on hardware breakpoint
From: Masami Hiramatsu (Google) @ 2026-07-17 14:19 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users
In-Reply-To: <178429796992.157981.3393977217853767915.stgit@devnote2>

From: Masami Hiramatsu (Google) <mhiramat@kernel.org>

Add a new probe event for the hardware breakpoint called wprobe-event.
This wprobe allows user to trace (watch) the memory access at the
specified memory address.
The new syntax is;

 w[:[GROUP/]EVENT] [r|w|rw]@[ADDR|SYM][:SIZE] [FETCH_ARGs]

User also can use $addr to fetch the accessed address and $value to fetch
the accessed memory value (shorthand for '+0($addr)'). No other variables
are supported.

For example, tracing updates of the jiffies;

 /sys/kernel/tracing # echo 'w:my_jiffies w@jiffies' >> dynamic_events
 /sys/kernel/tracing # cat dynamic_events
 w:wprobes/my_jiffies w@jiffies:4
 /sys/kernel/tracing # echo 1 > events/wprobes/my_jiffies/enable
 /sys/kernel/tracing # head -n 20 trace | tail -n 5
 #           TASK-PID     CPU#  |||||  TIMESTAMP  FUNCTION
 #              | |         |   |||||     |         |
          <idle>-0       [000] d.Z1.   206.547317: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
          <idle>-0       [000] d.Z1.   206.548341: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
          <idle>-0       [000] d.Z1.   206.549346: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)

Assisted-by: Antigravity:gemini-3.5-flash
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
 Changes in v9:
  - Rebased on probes/for-next branch.
  - Add WPROBE_NO_SIBLING error log to explicitly reject sibling probes
    since event triggers identify the target wprobe by event name.
  - Use traceprobe_parse_event_name() to properly validate group/event
    names instead of using the raw command string directly.
  - Generate unique event name (w_0x<addr>) for anonymous address-based
    watchpoints to avoid naming collisions.
  - Call traceprobe_update_arg() in __register_trace_wprobe() to resolve
    @symbol fetch arguments, consistent with kprobe and fprobe.
 Changes in v8:
  - Include required header files.
  - Use READ_ONCE(tw->addr) in trace handler to safely check dynamically
    updated addresses.
  - Prohibit unsafe perf support by returning -EOPNOTSUPP in
    wprobe_register().
  - Add rollback logic to unregister already-enabled sibling probes if
    registration fails mid-loop.
  - Resolve symbol offsets dynamically in trace_wprobe_show() using
    kallsyms_lookup_name().
  - Fix memory leak of parse_address_spec()'s symbol output in
    __trace_wprobe_create().
  - Print "rw" instead of "x" for read-write type breakpoints in
    trace_wprobe_show().
  - Document the $value fetcharg in wprobetrace.rst.
 Changes in v7:
  - Include IS_ERR_PCPU fix.
  - use seq_print_ip_sym_offset().
  - fix checkpatch warning on DEFINE_FREE()
  - Use bp->attr.bp_addr instead of tw->addr because it can be updated from another CPU.
---
 Documentation/trace/index.rst       |    1 
 Documentation/trace/wprobetrace.rst |   70 +++
 include/linux/trace_events.h        |    2 
 kernel/trace/Kconfig                |   13 +
 kernel/trace/Makefile               |    1 
 kernel/trace/trace.c                |    9 
 kernel/trace/trace.h                |    5 
 kernel/trace/trace_probe.c          |   21 +
 kernel/trace/trace_probe.h          |    9 
 kernel/trace/trace_wprobe.c         |  747 +++++++++++++++++++++++++++++++++++
 10 files changed, 874 insertions(+), 4 deletions(-)
 create mode 100644 Documentation/trace/wprobetrace.rst
 create mode 100644 kernel/trace/trace_wprobe.c

diff --git a/Documentation/trace/index.rst b/Documentation/trace/index.rst
index 5d9bf4694d5d..2f04f32001ed 100644
--- a/Documentation/trace/index.rst
+++ b/Documentation/trace/index.rst
@@ -36,6 +36,7 @@ the Linux kernel.
    kprobes
    kprobetrace
    fprobetrace
+   wprobetrace
    eprobetrace
    fprobe
    ring-buffer-design
diff --git a/Documentation/trace/wprobetrace.rst b/Documentation/trace/wprobetrace.rst
new file mode 100644
index 000000000000..eb4f10607530
--- /dev/null
+++ b/Documentation/trace/wprobetrace.rst
@@ -0,0 +1,70 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+=======================================
+Watchpoint probe (wprobe) Event Tracing
+=======================================
+
+.. Author: Masami Hiramatsu <mhiramat@kernel.org>
+
+Overview
+--------
+
+Wprobe event is a dynamic event based on the hardware breakpoint, which is
+similar to other probe events, but it is for watching data access. It allows
+you to trace which code accesses a specified data.
+
+As same as other dynamic events, wprobe events are defined via
+`dynamic_events` interface file on tracefs.
+
+Synopsis of wprobe-events
+-------------------------
+::
+
+  w:[GRP/][EVENT] SPEC [FETCHARGS]                       : Probe on data access
+
+ GRP            : Group name for wprobe. If omitted, use "wprobes" for it.
+ EVENT          : Event name for wprobe. If omitted, an event name is
+                  generated based on the address or symbol.
+ SPEC           : Breakpoint specification.
+                  [r|w|rw]@<ADDRESS|SYMBOL[+|-OFFS]>[:LENGTH]
+
+   r|w|rw       : Access type, r for read, w for write, and rw for both.
+                  Default is rw if omitted.
+   ADDRESS      : Address to trace (hexadecimal).
+   SYMBOL       : Symbol name to trace.
+   LENGTH       : Length of the data to trace in bytes. (1, 2, 4, or 8)
+
+  FETCHARGS      : Arguments. Each probe can have up to 128 args.
+   $addr         : Fetch the accessing address.
+   $value        : Fetch the memory value at the accessing address (same as +0($addr)).
+   @ADDR         : Fetch memory at ADDR (ADDR should be in kernel)
+  @SYM[+|-offs] : Fetch memory at SYM +|- offs (SYM should be a data symbol)
+  +|-[u]OFFS(FETCHARG) : Fetch memory at FETCHARG +|- OFFS address.(\*1)(\*2)
+  \IMM          : Store an immediate value to the argument.
+  NAME=FETCHARG : Set NAME as the argument name of FETCHARG.
+  FETCHARG:TYPE : Set TYPE as the type of FETCHARG. Currently, basic types
+                  (u8/u16/u32/u64/s8/s16/s32/s64), hexadecimal types
+                  (x8/x16/x32/x64), "char", "string", "ustring", "symbol", "symstr"
+                  and bitfield are supported.
+
+  (\*1) this is useful for fetching a field of data structures.
+  (\*2) "u" means user-space dereference.
+
+For the details of TYPE, see :ref:`kprobetrace documentation <kprobetrace_types>`.
+
+Usage examples
+--------------
+Here is an example to add a wprobe event on a variable `jiffies`.
+::
+
+  # echo 'w:my_jiffies w@jiffies' >> dynamic_events
+  # cat dynamic_events
+  w:wprobes/my_jiffies w@jiffies
+  # echo 1 > events/wprobes/enable
+  # cat trace | head
+  #           TASK-PID     CPU#  |||||  TIMESTAMP  FUNCTION
+  #              | |         |   |||||     |         |
+           <idle>-0       [000] d.Z1.  717.026259: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
+           <idle>-0       [000] d.Z1.  717.026373: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
+
+You can see the code which writes to `jiffies` is `tick_do_update_jiffies64()`.
diff --git a/include/linux/trace_events.h b/include/linux/trace_events.h
index 308c76b57d13..d1e5ab71d928 100644
--- a/include/linux/trace_events.h
+++ b/include/linux/trace_events.h
@@ -328,6 +328,7 @@ enum {
 	TRACE_EVENT_FL_UPROBE_BIT,
 	TRACE_EVENT_FL_EPROBE_BIT,
 	TRACE_EVENT_FL_FPROBE_BIT,
+	TRACE_EVENT_FL_WPROBE_BIT,
 	TRACE_EVENT_FL_CUSTOM_BIT,
 	TRACE_EVENT_FL_TEST_STR_BIT,
 };
@@ -358,6 +359,7 @@ enum {
 	TRACE_EVENT_FL_UPROBE		= (1 << TRACE_EVENT_FL_UPROBE_BIT),
 	TRACE_EVENT_FL_EPROBE		= (1 << TRACE_EVENT_FL_EPROBE_BIT),
 	TRACE_EVENT_FL_FPROBE		= (1 << TRACE_EVENT_FL_FPROBE_BIT),
+	TRACE_EVENT_FL_WPROBE		= (1 << TRACE_EVENT_FL_WPROBE_BIT),
 	TRACE_EVENT_FL_CUSTOM		= (1 << TRACE_EVENT_FL_CUSTOM_BIT),
 	TRACE_EVENT_FL_TEST_STR		= (1 << TRACE_EVENT_FL_TEST_STR_BIT),
 };
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index 0ab5916575a9..b58c2565024f 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -862,6 +862,19 @@ config EPROBE_EVENTS
 	  convert the type of an event field. For example, turn an
 	  address into a string.
 
+config WPROBE_EVENTS
+	bool "Enable wprobe-based dynamic events"
+	depends on TRACING
+	depends on HAVE_HW_BREAKPOINT
+	select PROBE_EVENTS
+	select DYNAMIC_EVENTS
+	help
+	  This allows the user to add watchpoint tracing events based on
+	  hardware breakpoints on the fly via the ftrace interface.
+
+	  Those events can be inserted wherever hardware breakpoints can be
+	  set, and record accessed memory address and values.
+
 config BPF_EVENTS
 	depends on BPF_SYSCALL
 	depends on (KPROBE_EVENTS || UPROBE_EVENTS) && PERF_EVENTS
diff --git a/kernel/trace/Makefile b/kernel/trace/Makefile
index f934ff586bd4..141c8323de20 100644
--- a/kernel/trace/Makefile
+++ b/kernel/trace/Makefile
@@ -126,6 +126,7 @@ obj-$(CONFIG_FTRACE_RECORD_RECURSION) += trace_recursion_record.o
 obj-$(CONFIG_FPROBE) += fprobe.o
 obj-$(CONFIG_RETHOOK) += rethook.o
 obj-$(CONFIG_FPROBE_EVENTS) += trace_fprobe.o
+obj-$(CONFIG_WPROBE_EVENTS) += trace_wprobe.o
 
 obj-$(CONFIG_TRACEPOINT_BENCHMARK) += trace_benchmark.o
 obj-$(CONFIG_RV) += rv/
diff --git a/kernel/trace/trace.c b/kernel/trace/trace.c
index c9e182d40059..1bc27c0ad029 100644
--- a/kernel/trace/trace.c
+++ b/kernel/trace/trace.c
@@ -4294,8 +4294,12 @@ static const char readme_msg[] =
 	"  uprobe_events\t\t- Create/append/remove/show the userspace dynamic events\n"
 	"\t\t\t  Write into this file to define/undefine new trace events.\n"
 #endif
+#ifdef CONFIG_WPROBE_EVENTS
+	"  wprobe_events\t\t- Create/append/remove/show the hardware breakpoint dynamic events\n"
+	"\t\t\t  Write into this file to define/undefine new trace events.\n"
+#endif
 #if defined(CONFIG_KPROBE_EVENTS) || defined(CONFIG_UPROBE_EVENTS) || \
-    defined(CONFIG_FPROBE_EVENTS)
+    defined(CONFIG_FPROBE_EVENTS) || defined(CONFIG_WPROBE_EVENTS)
 	"\t  accepts: event-definitions (one definition per line)\n"
 #if defined(CONFIG_KPROBE_EVENTS) || defined(CONFIG_UPROBE_EVENTS)
 	"\t   Format: p[:[<group>/][<event>]] <place> [<args>]\n"
@@ -4305,6 +4309,9 @@ static const char readme_msg[] =
 	"\t           f[:[<group>/][<event>]] <func-name>[%return] [<args>]\n"
 	"\t           t[:[<group>/][<event>]] <tracepoint> [<args>]\n"
 #endif
+#ifdef CONFIG_WPROBE_EVENTS
+	"\t           w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]\n"
+#endif
 #ifdef CONFIG_HIST_TRIGGERS
 	"\t           s:[synthetic/]<event> <field> [<field>]\n"
 #endif
diff --git a/kernel/trace/trace.h b/kernel/trace/trace.h
index 80fe152af1dd..2f07c5c4ffc8 100644
--- a/kernel/trace/trace.h
+++ b/kernel/trace/trace.h
@@ -179,6 +179,11 @@ struct fexit_trace_entry_head {
 	unsigned long		ret_ip;
 };
 
+struct wprobe_trace_entry_head {
+	struct trace_entry	ent;
+	unsigned long		ip;
+};
+
 #define TRACE_BUF_SIZE		1024
 
 struct trace_array;
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 7568f5e68de7..5d5e9b477b86 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -1405,6 +1405,23 @@ static int parse_probe_vars(char *orig_arg, const struct fetch_type *t,
 			return 0;
 	}
 
+	/* wprobe only support "$addr" and "$value" variable */
+	if (ctx->flags & TPARG_FL_WPROBE) {
+		if (!strcmp(arg, "addr")) {
+			code->op = FETCH_OP_BADDR;
+			return 0;
+		}
+		if (!strcmp(arg, "value")) {
+			code->op = FETCH_OP_BADDR;
+			code++;
+			code->op = FETCH_OP_DEREF;
+			code->offset = 0;
+			*pcode = code;
+			return 0;
+		}
+		goto inval;
+	}
+
 	if (strcmp(arg, "comm") == 0 || strcmp(arg, "COMM") == 0) {
 		code->op = FETCH_OP_COMM;
 		return 0;
@@ -1464,8 +1481,8 @@ static int parse_probe_arg_register(char *arg, struct fetch_insn *code,
 {
 	int ret;
 
-	if (ctx->flags & (TPARG_FL_TEVENT | TPARG_FL_FPROBE)) {
-		/* eprobe and fprobe do not handle registers */
+	if (ctx->flags & (TPARG_FL_TEVENT | TPARG_FL_FPROBE | TPARG_FL_WPROBE)) {
+		/* eprobe, fprobe and wprobe do not handle registers */
 		trace_probe_log_err(ctx->offset, BAD_VAR);
 		return -EINVAL;
 	}
diff --git a/kernel/trace/trace_probe.h b/kernel/trace/trace_probe.h
index ebdc706e7cb6..7380502a85af 100644
--- a/kernel/trace/trace_probe.h
+++ b/kernel/trace/trace_probe.h
@@ -91,6 +91,7 @@ typedef int (*print_type_func_t)(struct trace_seq *, void *, void *);
 	FETCH_OP(STACK, param),		/* Stack: .param = index */	\
 	FETCH_OP(STACKP, none),		/* Stack pointer */		\
 	FETCH_OP(RETVAL, none),		/* Return value */		\
+	FETCH_OP(BADDR, none),		/* Break address */		\
 	FETCH_OP(IMM, imm),		/* Immediate: .immediate */	\
 	FETCH_OP(COMM, none),		/* Current comm */		\
 	FETCH_OP(CURRENT, none),	/* Current task_struct address */\
@@ -420,6 +421,7 @@ static inline int traceprobe_get_entry_data_size(struct trace_probe *tp)
 #define TPARG_FL_USER   BIT(4)
 #define TPARG_FL_FPROBE BIT(5)
 #define TPARG_FL_TPOINT BIT(6)
+#define TPARG_FL_WPROBE BIT(7)
 #define TPARG_FL_LOC_MASK	GENMASK(4, 0)
 
 static inline bool tparg_is_function_entry(unsigned int flags)
@@ -546,6 +548,10 @@ extern int traceprobe_define_arg_fields(struct trace_event_call *event_call,
 	C(ARG_TOO_LONG,		"Argument expression is too long"),		\
 	C(ARRAY_NO_CLOSE,	"Array is not closed"),		\
 	C(ARRAY_TOO_BIG,	"Array number is too big"),		\
+	C(BAD_ACCESS_ADDR,	"Invalid access memory address"),		\
+	C(BAD_ACCESS_FMT,	"Access memory address requires @"),		\
+	C(BAD_ACCESS_LEN,	"This memory access length is not supported"),	\
+	C(BAD_ACCESS_TYPE,	"Bad memory access type"),			\
 	C(BAD_ADDR_SUFFIX,	"Invalid probed address suffix"),		\
 	C(BAD_ARG_NAME,		"Argument name must follow the same rules as C identifiers"),	\
 	C(BAD_ARG_NUM,		"Invalid argument number"),		\
@@ -628,7 +634,8 @@ extern int traceprobe_define_arg_fields(struct trace_event_call *event_call,
 	C(TYPECAST_NOT_EVENT,	"Typecasts are only for eprobe fields"),	\
 	C(TYPECAST_REQ_FIELD,	"Typecast requires a field access"),	\
 	C(TYPECAST_SYM_OFFSET,	"@SYM+/-OFFSET with typecast needs parentheses"),	\
-	C(USED_ARG_NAME,	"This argument name is already used"),
+	C(USED_ARG_NAME,	"This argument name is already used"),		\
+	C(WPROBE_NO_SIBLING,	"Watchpoint probe does not support sibling probes"),
 
 #undef C
 #define C(a, b)		TP_ERR_##a
diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
new file mode 100644
index 000000000000..08a2829b9eaa
--- /dev/null
+++ b/kernel/trace/trace_wprobe.c
@@ -0,0 +1,747 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Hardware-breakpoint-based tracing events
+ *
+ * Copyright (C) 2023, Masami Hiramatsu <mhiramat@kernel.org>
+ */
+#define pr_fmt(fmt)	"trace_wprobe: " fmt
+
+#include <linux/compiler.h>
+#include <linux/hw_breakpoint.h>
+#include <linux/kallsyms.h>
+#include <linux/list.h>
+#include <linux/module.h>
+#include <linux/mutex.h>
+#include <linux/perf_event.h>
+#include <linux/rculist.h>
+#include <linux/security.h>
+#include <linux/tracepoint.h>
+#include <linux/uaccess.h>
+
+#include <asm/ptrace.h>
+
+#include "trace_dynevent.h"
+#include "trace_probe.h"
+#include "trace_probe_kernel.h"
+#include "trace_probe_tmpl.h"
+#include "trace_output.h"
+
+#define WPROBE_EVENT_SYSTEM "wprobes"
+
+static int trace_wprobe_create(const char *raw_command);
+static int trace_wprobe_show(struct seq_file *m, struct dyn_event *ev);
+static int trace_wprobe_release(struct dyn_event *ev);
+static bool trace_wprobe_is_busy(struct dyn_event *ev);
+static bool trace_wprobe_match(const char *system, const char *event,
+			       int argc, const char **argv, struct dyn_event *ev);
+
+static struct dyn_event_operations trace_wprobe_ops = {
+	.create = trace_wprobe_create,
+	.show = trace_wprobe_show,
+	.is_busy = trace_wprobe_is_busy,
+	.free = trace_wprobe_release,
+	.match = trace_wprobe_match,
+};
+
+struct trace_wprobe {
+	struct dyn_event	devent;
+	struct perf_event * __percpu *bp_event;
+	unsigned long		addr;
+	int			len;
+	int			type;
+	const char		*symbol;
+	struct trace_probe	tp;
+};
+
+static bool is_trace_wprobe(struct dyn_event *ev)
+{
+	return ev->ops == &trace_wprobe_ops;
+}
+
+static struct trace_wprobe *to_trace_wprobe(struct dyn_event *ev)
+{
+	return container_of(ev, struct trace_wprobe, devent);
+}
+
+#define for_each_trace_wprobe(pos, dpos)			\
+	for_each_dyn_event(dpos)				\
+		if (is_trace_wprobe(dpos) && (pos = to_trace_wprobe(dpos)))
+
+static bool trace_wprobe_is_busy(struct dyn_event *ev)
+{
+	struct trace_wprobe *tw = to_trace_wprobe(ev);
+
+	return trace_probe_is_enabled(&tw->tp);
+}
+
+static bool trace_wprobe_match(const char *system, const char *event,
+			       int argc, const char **argv, struct dyn_event *ev)
+{
+	struct trace_wprobe *tw = to_trace_wprobe(ev);
+
+	if (event[0] != '\0' && strcmp(trace_probe_name(&tw->tp), event))
+		return false;
+
+	if (system && strcmp(trace_probe_group_name(&tw->tp), system))
+		return false;
+
+	/* TODO: match arguments */
+	return true;
+}
+
+/*
+ * Note that we don't verify the fetch_insn code, since it does not come
+ * from user space.
+ */
+static int
+process_fetch_insn(struct fetch_insn *code, void *rec, void *edata,
+		   void *dest, void *base)
+{
+	void *baddr = rec;
+	unsigned long val;
+	int ret;
+
+retry:
+	/* 1st stage: get value from context */
+	switch (code->op) {
+	case FETCH_OP_BADDR:
+		val = (unsigned long)baddr;
+		break;
+	case FETCH_NOP_SYMBOL:	/* Ignore a place holder */
+		code++;
+		goto retry;
+	default:
+		ret = process_common_fetch_insn(code, &val);
+		if (ret < 0)
+			return ret;
+	}
+	code++;
+
+	return process_fetch_insn_bottom(code, val, dest, base);
+}
+NOKPROBE_SYMBOL(process_fetch_insn)
+
+static void wprobe_trace_handler(struct trace_wprobe *tw,
+				 unsigned long addr,
+				 struct pt_regs *regs,
+				 struct trace_event_file *trace_file)
+{
+	struct wprobe_trace_entry_head *entry;
+	struct trace_event_call *call = trace_probe_event_call(&tw->tp);
+	struct trace_event_buffer fbuffer;
+	int dsize;
+
+	if (WARN_ON_ONCE(call != trace_file->event_call))
+		return;
+
+	if (trace_trigger_soft_disabled(trace_file))
+		return;
+
+	if (READ_ONCE(tw->addr) != addr)
+		return;
+
+	dsize = __get_data_size(&tw->tp, (void *)addr, NULL);
+
+	entry = trace_event_buffer_reserve(&fbuffer, trace_file,
+					   sizeof(*entry) + tw->tp.size + dsize);
+	if (!entry)
+		return;
+
+	entry->ip = instruction_pointer(regs);
+	store_trace_args(&entry[1], &tw->tp, (void *)addr, NULL, sizeof(*entry), dsize);
+
+	fbuffer.regs = regs;
+	trace_event_buffer_commit(&fbuffer);
+}
+
+static void wprobe_perf_handler(struct perf_event *bp,
+			      struct perf_sample_data *data,
+			      struct pt_regs *regs)
+{
+	struct trace_wprobe *tw = bp->overflow_handler_context;
+	struct event_file_link *link;
+	unsigned long addr = bp->attr.bp_addr;
+
+	trace_probe_for_each_link_rcu(link, &tw->tp)
+		wprobe_trace_handler(tw, addr, regs, link->file);
+}
+
+static int __register_trace_wprobe(struct trace_wprobe *tw)
+{
+	struct perf_event_attr attr;
+	int i, ret;
+
+	if (tw->bp_event)
+		return -EINVAL;
+
+	for (i = 0; i < tw->tp.nr_args; i++) {
+		ret = traceprobe_update_arg(&tw->tp.args[i]);
+		if (ret)
+			return ret;
+	}
+
+	hw_breakpoint_init(&attr);
+	attr.bp_addr = tw->addr;
+	attr.bp_len = tw->len;
+	attr.bp_type = tw->type;
+
+	tw->bp_event = register_wide_hw_breakpoint(&attr, wprobe_perf_handler, tw);
+	if (IS_ERR_PCPU(tw->bp_event)) {
+		int ret = PTR_ERR_PCPU(tw->bp_event);
+
+		tw->bp_event = NULL;
+		return ret;
+	}
+
+	return 0;
+}
+
+static void __unregister_trace_wprobe(struct trace_wprobe *tw)
+{
+	if (tw->bp_event) {
+		unregister_wide_hw_breakpoint(tw->bp_event);
+		tw->bp_event = NULL;
+	}
+}
+
+static void free_trace_wprobe(struct trace_wprobe *tw)
+{
+	if (tw) {
+		trace_probe_cleanup(&tw->tp);
+		kfree(tw->symbol);
+		kfree(tw);
+	}
+}
+DEFINE_FREE(free_trace_wprobe, struct trace_wprobe *,
+	if (!IS_ERR_OR_NULL(_T))
+		free_trace_wprobe(_T))
+
+
+static struct trace_wprobe *alloc_trace_wprobe(const char *group,
+					       const char *event,
+					       const char *symbol,
+					       unsigned long addr,
+					       int len, int type, int nargs)
+{
+	struct trace_wprobe *tw __free(free_trace_wprobe) = NULL;
+	int ret;
+
+	tw = kzalloc(struct_size(tw, tp.args, nargs), GFP_KERNEL);
+	if (!tw)
+		return ERR_PTR(-ENOMEM);
+
+	if (symbol) {
+		tw->symbol = kstrdup(symbol, GFP_KERNEL);
+		if (!tw->symbol)
+			return ERR_PTR(-ENOMEM);
+	}
+	tw->addr = addr;
+	tw->len = len;
+	tw->type = type;
+
+	ret = trace_probe_init(&tw->tp, event, group, false, nargs);
+	if (ret < 0)
+		return ERR_PTR(ret);
+
+	dyn_event_init(&tw->devent, &trace_wprobe_ops);
+	return_ptr(tw);
+}
+
+static struct trace_wprobe *find_trace_wprobe(const char *event,
+					      const char *group)
+{
+	struct dyn_event *pos;
+	struct trace_wprobe *tw;
+
+	for_each_trace_wprobe(tw, pos)
+		if (strcmp(trace_probe_name(&tw->tp), event) == 0 &&
+		    strcmp(trace_probe_group_name(&tw->tp), group) == 0)
+			return tw;
+	return NULL;
+}
+
+static enum print_line_t
+print_wprobe_event(struct trace_iterator *iter, int flags,
+		   struct trace_event *event)
+{
+	struct wprobe_trace_entry_head *field;
+	struct trace_seq *s = &iter->seq;
+	struct trace_probe *tp;
+
+	field = (struct wprobe_trace_entry_head *)iter->ent;
+	tp = trace_probe_primary_from_call(
+		container_of(event, struct trace_event_call, event));
+	if (WARN_ON_ONCE(!tp))
+		goto out;
+
+	trace_seq_printf(s, "%s: (", trace_probe_name(tp));
+
+	if (!seq_print_ip_sym_offset(s, field->ip, flags))
+		goto out;
+
+	trace_seq_putc(s, ')');
+
+	if (trace_probe_print_args(s, tp->args, tp->nr_args,
+			     (u8 *)&field[1], field) < 0)
+		goto out;
+
+	trace_seq_putc(s, '\n');
+out:
+	return trace_handle_return(s);
+}
+
+static int wprobe_event_define_fields(struct trace_event_call *event_call)
+{
+	int ret;
+	struct wprobe_trace_entry_head field;
+	struct trace_probe *tp;
+
+	tp = trace_probe_primary_from_call(event_call);
+	if (WARN_ON_ONCE(!tp))
+		return -ENOENT;
+
+	DEFINE_FIELD(unsigned long, ip, FIELD_STRING_IP, 0);
+
+	return traceprobe_define_arg_fields(event_call, sizeof(field), tp);
+}
+
+static struct trace_event_functions wprobe_funcs = {
+	.trace	= print_wprobe_event
+};
+
+static struct trace_event_fields wprobe_fields_array[] = {
+	{ .type = TRACE_FUNCTION_TYPE,
+	  .define_fields = wprobe_event_define_fields },
+	{}
+};
+
+static int wprobe_register(struct trace_event_call *event,
+			   enum trace_reg type, void *data);
+
+static inline void init_trace_event_call(struct trace_wprobe *tw)
+{
+	struct trace_event_call *call = trace_probe_event_call(&tw->tp);
+
+	call->event.funcs = &wprobe_funcs;
+	call->class->fields_array = wprobe_fields_array;
+	call->flags = TRACE_EVENT_FL_WPROBE;
+	call->class->reg = wprobe_register;
+}
+
+static int register_wprobe_event(struct trace_wprobe *tw)
+{
+	init_trace_event_call(tw);
+	return trace_probe_register_event_call(&tw->tp);
+}
+
+static int register_trace_wprobe_event(struct trace_wprobe *tw)
+{
+	struct trace_wprobe *old_tw;
+	int ret;
+
+	guard(mutex)(&event_mutex);
+
+	old_tw = find_trace_wprobe(trace_probe_name(&tw->tp),
+				   trace_probe_group_name(&tw->tp));
+	if (old_tw) {
+		/*
+		 * Wprobe does not support sibling probes because the event
+		 * trigger (set_wprobe/clear_wprobe) identifies the target
+		 * wprobe by its event name. Having multiple wprobes sharing
+		 * the same event name would make the target ambiguous.
+		 */
+		trace_probe_log_set_index(0);
+		trace_probe_log_err(0, WPROBE_NO_SIBLING);
+		return -EBUSY;
+	}
+
+	ret = register_wprobe_event(tw);
+	if (ret)
+		return ret;
+
+	dyn_event_add(&tw->devent, trace_probe_event_call(&tw->tp));
+	return 0;
+}
+static int unregister_wprobe_event(struct trace_wprobe *tw)
+{
+	return trace_probe_unregister_event_call(&tw->tp);
+}
+
+static int unregister_trace_wprobe(struct trace_wprobe *tw)
+{
+	if (trace_probe_has_sibling(&tw->tp))
+		goto unreg;
+
+	if (trace_probe_is_enabled(&tw->tp))
+		return -EBUSY;
+
+	if (trace_event_dyn_busy(trace_probe_event_call(&tw->tp)))
+		return -EBUSY;
+
+	if (unregister_wprobe_event(tw))
+		return -EBUSY;
+
+unreg:
+	__unregister_trace_wprobe(tw);
+	dyn_event_remove(&tw->devent);
+	trace_probe_unlink(&tw->tp);
+
+	return 0;
+}
+
+static int enable_trace_wprobe(struct trace_event_call *call,
+			       struct trace_event_file *file)
+{
+	struct trace_probe *tp;
+	struct trace_wprobe *tw;
+	bool enabled;
+	int ret = 0;
+
+	tp = trace_probe_primary_from_call(call);
+	if (WARN_ON_ONCE(!tp))
+		return -ENODEV;
+	enabled = trace_probe_is_enabled(tp);
+
+	if (file) {
+		ret = trace_probe_add_file(tp, file);
+		if (ret)
+			return ret;
+	} else {
+		trace_probe_set_flag(tp, TP_FLAG_PROFILE);
+	}
+
+	if (!enabled) {
+		list_for_each_entry(tw, trace_probe_probe_list(tp), tp.list) {
+			ret = __register_trace_wprobe(tw);
+			if (ret < 0) {
+				struct trace_wprobe *tmp;
+
+				list_for_each_entry(tmp, trace_probe_probe_list(tp), tp.list) {
+					if (tmp == tw)
+						break;
+					__unregister_trace_wprobe(tmp);
+				}
+				if (file)
+					trace_probe_remove_file(tp, file);
+				else
+					trace_probe_clear_flag(tp, TP_FLAG_PROFILE);
+				return ret;
+			}
+		}
+	}
+
+	return 0;
+}
+
+static int disable_trace_wprobe(struct trace_event_call *call,
+				struct trace_event_file *file)
+{
+	struct trace_wprobe *tw;
+	struct trace_probe *tp;
+
+	tp = trace_probe_primary_from_call(call);
+	if (WARN_ON_ONCE(!tp))
+		return -ENODEV;
+
+	if (file) {
+		if (!trace_probe_get_file_link(tp, file))
+			return -ENOENT;
+		if (!trace_probe_has_single_file(tp))
+			goto out;
+		trace_probe_clear_flag(tp, TP_FLAG_TRACE);
+	} else {
+		trace_probe_clear_flag(tp, TP_FLAG_PROFILE);
+	}
+
+	if (!trace_probe_is_enabled(tp)) {
+		list_for_each_entry(tw, trace_probe_probe_list(tp), tp.list) {
+			__unregister_trace_wprobe(tw);
+		}
+	}
+
+out:
+	if (file)
+		trace_probe_remove_file(tp, file);
+
+	return 0;
+}
+
+static int wprobe_register(struct trace_event_call *event,
+			   enum trace_reg type, void *data)
+{
+	struct trace_event_file *file = data;
+
+	switch (type) {
+	case TRACE_REG_REGISTER:
+		return enable_trace_wprobe(event, file);
+	case TRACE_REG_UNREGISTER:
+		return disable_trace_wprobe(event, file);
+
+#ifdef CONFIG_PERF_EVENTS
+	case TRACE_REG_PERF_REGISTER:
+	case TRACE_REG_PERF_UNREGISTER:
+	case TRACE_REG_PERF_OPEN:
+	case TRACE_REG_PERF_CLOSE:
+	case TRACE_REG_PERF_ADD:
+	case TRACE_REG_PERF_DEL:
+		return -EOPNOTSUPP;
+#endif
+	}
+	return 0;
+}
+
+static int parse_address_spec(const char *spec, unsigned long *addr, int *type,
+			      int *len, char **symbol)
+{
+	char *_spec __free(kfree) = NULL;
+	int _len = HW_BREAKPOINT_LEN_4;
+	int _type = HW_BREAKPOINT_RW;
+	unsigned long _addr = 0;
+	char *at, *col;
+
+	_spec = kstrdup(spec, GFP_KERNEL);
+	if (!_spec)
+		return -ENOMEM;
+
+	at = strchr(_spec, '@');
+	col = strchr(_spec, ':');
+
+	if (!at) {
+		trace_probe_log_err(0, BAD_ACCESS_FMT);
+		return -EINVAL;
+	}
+
+	if (at != _spec) {
+		*at = '\0';
+
+		if (strcmp(_spec, "r") == 0)
+			_type = HW_BREAKPOINT_R;
+		else if (strcmp(_spec, "w") == 0)
+			_type = HW_BREAKPOINT_W;
+		else if (strcmp(_spec, "rw") == 0)
+			_type = HW_BREAKPOINT_RW;
+		else {
+			trace_probe_log_err(0, BAD_ACCESS_TYPE);
+			return -EINVAL;
+		}
+	}
+
+	if (col) {
+		*col = '\0';
+		if (kstrtoint(col + 1, 0, &_len)) {
+			trace_probe_log_err(col + 1 - _spec, BAD_ACCESS_LEN);
+			return -EINVAL;
+		}
+
+		switch (_len) {
+		case 1:
+			_len = HW_BREAKPOINT_LEN_1;
+			break;
+		case 2:
+			_len = HW_BREAKPOINT_LEN_2;
+			break;
+		case 4:
+			_len = HW_BREAKPOINT_LEN_4;
+			break;
+		case 8:
+			_len = HW_BREAKPOINT_LEN_8;
+			break;
+		default:
+			trace_probe_log_err(col + 1 - _spec, BAD_ACCESS_LEN);
+			return -EINVAL;
+		}
+	}
+
+	if (kstrtoul(at + 1, 0, &_addr) != 0) {
+		char *off_str = strpbrk(at + 1, "+-");
+		int offset = 0;
+
+		if (off_str) {
+			if (kstrtoint(off_str, 0, &offset) != 0) {
+				trace_probe_log_err(off_str - _spec, BAD_PROBE_ADDR);
+				return -EINVAL;
+			}
+			*off_str = '\0';
+		}
+		_addr = kallsyms_lookup_name(at + 1);
+		if (!_addr) {
+			trace_probe_log_err(at + 1 - _spec, BAD_ACCESS_ADDR);
+			return -ENOENT;
+		}
+		_addr += offset;
+		*symbol = kstrdup(at + 1, GFP_KERNEL);
+		if (!*symbol)
+			return -ENOMEM;
+	}
+
+	*addr = _addr;
+	*type = _type;
+	*len = _len;
+	return 0;
+}
+
+static int __trace_wprobe_create(int argc, const char *argv[])
+{
+	/*
+	 * Argument syntax:
+	 *  b[:[GRP/][EVENT]] SPEC
+	 *
+	 * SPEC:
+	 *  [r|w|rw]@[ADDR|SYMBOL[+OFFS]][:LEN]
+	 */
+	struct traceprobe_parse_context *ctx __free(traceprobe_parse_context) = NULL;
+	struct trace_wprobe *tw __free(free_trace_wprobe) = NULL;
+	const char *event = NULL, *group = WPROBE_EVENT_SYSTEM;
+	const char *tplog __free(trace_probe_log_clear) = NULL;
+	char *symbol __free(kfree) = NULL;
+	char *gbuf __free(kfree) = NULL;
+	char *ebuf __free(kfree) = NULL;
+	unsigned long addr;
+	int len, type, i;
+	int ret = 0;
+
+	if (argv[0][0] != 'w')
+		return -ECANCELED;
+
+	if (argc < 2)
+		return -EINVAL;
+
+	tplog = trace_probe_log_init("wprobe", argc, argv);
+
+	if (argv[0][1] != '\0') {
+		if (argv[0][1] != ':') {
+			trace_probe_log_set_index(0);
+			trace_probe_log_err(1, BAD_MAXACT_TYPE);
+			return -EINVAL;
+		}
+		event = &argv[0][2];
+	}
+
+	trace_probe_log_set_index(1);
+	ret = parse_address_spec(argv[1], &addr, &type, &len, &symbol);
+	if (ret < 0)
+		return ret;
+
+	trace_probe_log_set_index(0);
+	if (event) {
+		gbuf = kmalloc(MAX_EVENT_NAME_LEN, GFP_KERNEL);
+		if (!gbuf)
+			return -ENOMEM;
+		ret = traceprobe_parse_event_name(&event, &group, gbuf,
+						  event - argv[0]);
+		if (ret)
+			return ret;
+	}
+
+	if (!event) {
+		/* Make a new event name */
+		ebuf = kmalloc(MAX_EVENT_NAME_LEN, GFP_KERNEL);
+		if (!ebuf)
+			return -ENOMEM;
+		if (symbol)
+			snprintf(ebuf, MAX_EVENT_NAME_LEN, "%s", symbol);
+		else
+			snprintf(ebuf, MAX_EVENT_NAME_LEN, "w_0x%lx", addr);
+		sanitize_event_name(ebuf);
+		event = ebuf;
+	}
+
+	argc -= 2; argv += 2;
+	tw = alloc_trace_wprobe(group, event, symbol, addr, len, type, argc);
+	if (IS_ERR(tw))
+		return PTR_ERR(tw);
+
+	ctx = kzalloc_obj(*ctx);
+	if (!ctx)
+		return -ENOMEM;
+
+	ctx->flags = TPARG_FL_KERNEL | TPARG_FL_WPROBE;
+
+	/* parse arguments */
+	for (i = 0; i < argc; i++) {
+		trace_probe_log_set_index(i + 2);
+		ctx->offset = 0;
+		ret = traceprobe_parse_probe_arg(&tw->tp, i, argv[i], ctx);
+		if (ret)
+			return ret;	/* This can be -ENOMEM */
+	}
+
+	ret = traceprobe_set_print_fmt(&tw->tp, PROBE_PRINT_NORMAL);
+	if (ret < 0)
+		return ret;
+
+	ret = register_trace_wprobe_event(tw);
+	if (!ret)
+		tw = NULL; /* To avoid free */
+
+	return ret;
+}
+
+static int trace_wprobe_create(const char *raw_command)
+{
+	return trace_probe_create(raw_command, __trace_wprobe_create);
+}
+
+static int trace_wprobe_release(struct dyn_event *ev)
+{
+	struct trace_wprobe *tw = to_trace_wprobe(ev);
+	int ret = unregister_trace_wprobe(tw);
+
+	if (!ret)
+		free_trace_wprobe(tw);
+	return ret;
+}
+
+static int trace_wprobe_show(struct seq_file *m, struct dyn_event *ev)
+{
+	struct trace_wprobe *tw = to_trace_wprobe(ev);
+	int i;
+
+	seq_printf(m, "w:%s/%s", trace_probe_group_name(&tw->tp),
+		   trace_probe_name(&tw->tp));
+
+	const char *type_str;
+
+	if (tw->type == HW_BREAKPOINT_R)
+		type_str = "r";
+	else if (tw->type == HW_BREAKPOINT_W)
+		type_str = "w";
+	else
+		type_str = "rw";
+
+	int len;
+
+	if (tw->len == HW_BREAKPOINT_LEN_1)
+		len = 1;
+	else if (tw->len == HW_BREAKPOINT_LEN_2)
+		len = 2;
+	else if (tw->len == HW_BREAKPOINT_LEN_4)
+		len = 4;
+	else
+		len = 8;
+
+	if (tw->symbol) {
+		unsigned long sym_addr = kallsyms_lookup_name(tw->symbol);
+		long offset = sym_addr ? (long)(tw->addr - sym_addr) : 0;
+
+		if (offset)
+			seq_printf(m, " %s@%s%+ld:%d", type_str, tw->symbol, offset, len);
+		else
+			seq_printf(m, " %s@%s:%d", type_str, tw->symbol, len);
+	} else {
+		seq_printf(m, " %s@0x%lx:%d", type_str, tw->addr, len);
+	}
+
+	for (i = 0; i < tw->tp.nr_args; i++)
+		seq_printf(m, " %s=%s", tw->tp.args[i].name, tw->tp.args[i].comm);
+	seq_putc(m, '\n');
+
+	return 0;
+}
+
+static __init int init_wprobe_trace(void)
+{
+	return dyn_event_register(&trace_wprobe_ops);
+}
+fs_initcall(init_wprobe_trace);
+


^ permalink raw reply related

* [PATCH v9 02/10] x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires
From: Masami Hiramatsu (Google) @ 2026-07-17 14:19 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users
In-Reply-To: <178429796992.157981.3393977217853767915.stgit@devnote2>

From: Masami Hiramatsu (Google) <mhiramat@kernel.org>

Add CONFIG_HAVE_POST_BREAKPOINT_HOOK which indicates the hw_breakpoint
on that architecture fires after the target memory has been modified.
This is currently x86 only behavior.

Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
 arch/Kconfig         |   10 ++++++++++
 arch/x86/Kconfig     |    1 +
 kernel/trace/Kconfig |    1 +
 3 files changed, 12 insertions(+)

diff --git a/arch/Kconfig b/arch/Kconfig
index fa7507ac8e13..959aee9568ff 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -457,6 +457,16 @@ config HAVE_MIXED_BREAKPOINTS_REGS
 	  Select this option if your arch implements breakpoints under the
 	  latter fashion.
 
+config HAVE_POST_BREAKPOINT_HOOK
+	bool
+	depends on HAVE_HW_BREAKPOINT
+	help
+	  Depending on the arch implementation of hardware breakpoints,
+	  some of them provide breakpoint hook after the target memory
+	  is modified.
+	  Select this option if your arch implements breakpoints overflow
+	  handler hooks after the target memory is modified.
+
 config HAVE_USER_RETURN_NOTIFIER
 	bool
 
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index bdad90f210e4..6b7e14ef8cfb 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -246,6 +246,7 @@ config X86
 	select HAVE_FUNCTION_TRACER
 	select HAVE_GCC_PLUGINS
 	select HAVE_HW_BREAKPOINT
+	select HAVE_POST_BREAKPOINT_HOOK
 	select HAVE_IOREMAP_PROT
 	select HAVE_IRQ_EXIT_ON_IRQ_STACK	if X86_64
 	select HAVE_IRQ_TIME_ACCOUNTING
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index b58c2565024f..d9b6fa5c35d9 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -866,6 +866,7 @@ config WPROBE_EVENTS
 	bool "Enable wprobe-based dynamic events"
 	depends on TRACING
 	depends on HAVE_HW_BREAKPOINT
+	depends on HAVE_POST_BREAKPOINT_HOOK
 	select PROBE_EVENTS
 	select DYNAMIC_EVENTS
 	help


^ permalink raw reply related

* [PATCH v9 03/10] selftests: tracing: Add a basic testcase for wprobe
From: Masami Hiramatsu (Google) @ 2026-07-17 14:20 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users
In-Reply-To: <178429796992.157981.3393977217853767915.stgit@devnote2>

From: Masami Hiramatsu (Google) <mhiramat@kernel.org>

Add 'add_remove_wprobe.tc' testcase for testing wprobe event that
tests adding and removing operations of the wprobe event.

Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
 Changes in v9:
  - Fix command check logic to prevent early exit under 'set -e'
    (errexit) when grep or test fails.
  - Simplify enable/disable status checks by removing cat pipes.
 Changes in v8:
  - Fixed silently test failure path.
---
 tools/testing/selftests/ftrace/config              |    1 
 .../ftrace/test.d/dynevent/add_remove_wprobe.tc    |   63 ++++++++++++++++++++
 2 files changed, 64 insertions(+)
 create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc

diff --git a/tools/testing/selftests/ftrace/config b/tools/testing/selftests/ftrace/config
index 544de0db5f58..d2f503722020 100644
--- a/tools/testing/selftests/ftrace/config
+++ b/tools/testing/selftests/ftrace/config
@@ -27,3 +27,4 @@ CONFIG_STACK_TRACER=y
 CONFIG_TRACER_SNAPSHOT=y
 CONFIG_UPROBES=y
 CONFIG_UPROBE_EVENTS=y
+CONFIG_WPROBE_EVENTS=y
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
new file mode 100644
index 000000000000..647c37d5e4c8
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
@@ -0,0 +1,63 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: Generic dynamic event - add/remove wprobe events
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README
+
+echo 0 > events/enable
+echo > dynamic_events
+
+# Use jiffies as a variable that is frequently written to.
+TARGET=jiffies
+
+echo "w:my_wprobe w@$TARGET" >> dynamic_events
+
+if ! grep -q my_wprobe dynamic_events; then
+    echo "Failed to create wprobe event"
+    exit_fail
+fi
+
+if [ ! -d events/wprobes/my_wprobe ]; then
+    echo "Failed to create wprobe event directory"
+    exit_fail
+fi
+
+echo 1 > events/wprobes/my_wprobe/enable
+
+# Check if the event is enabled
+if ! grep -q 1 events/wprobes/my_wprobe/enable; then
+    echo "Failed to enable wprobe event"
+    exit_fail
+fi
+
+# Let some time pass to trigger the breakpoint
+sleep 1
+
+# Check if we got any trace output
+if ! grep -q my_wprobe trace; then
+    echo "wprobe event was not triggered"
+    exit_fail
+fi
+
+echo 0 > events/wprobes/my_wprobe/enable
+
+# Check if the event is disabled
+if ! grep -q 0 events/wprobes/my_wprobe/enable; then
+    echo "Failed to disable wprobe event"
+    exit_fail
+fi
+
+echo "-:my_wprobe" >> dynamic_events
+
+if grep -q my_wprobe dynamic_events; then
+    echo "Failed to remove wprobe event"
+    exit_fail
+fi
+
+if [ -d events/wprobes/my_wprobe ]; then
+    echo "Failed to remove wprobe event directory"
+    exit_fail
+fi
+
+clear_trace
+
+exit 0


^ permalink raw reply related

* [PATCH v9 04/10] selftests: tracing: Add syntax testcase for wprobe
From: Masami Hiramatsu (Google) @ 2026-07-17 14:20 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users
In-Reply-To: <178429796992.157981.3393977217853767915.stgit@devnote2>

From: Masami Hiramatsu (Google) <mhiramat@kernel.org>

Add "wprobe_syntax_errors.tc" testcase for testing syntax errors
of the watch probe events.

Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
 .../test.d/dynevent/wprobes_syntax_errors.tc       |   20 ++++++++++++++++++++
 1 file changed, 20 insertions(+)
 create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc

diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
new file mode 100644
index 000000000000..56ac579d60ae
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
@@ -0,0 +1,20 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: Watch probe event parser error log check
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README
+
+check_error() { # command-with-error-pos-by-^
+    ftrace_errlog_check 'wprobe' "$1" 'dynamic_events'
+}
+
+check_error 'w ^symbol'			# BAD_ACCESS_FMT
+check_error 'w ^a@symbol'		# BAD_ACCESS_TYPE
+check_error 'w w@^symbol'		# BAD_ACCESS_ADDR
+check_error 'w w@jiffies^+offset'	# BAD_ACCESS_ADDR
+check_error 'w w@jiffies:^100'		# BAD_ACCESS_LEN
+check_error 'w w@jiffies ^$arg1'	# BAD_VAR
+check_error 'w w@jiffies ^$retval'	# BAD_VAR
+check_error 'w w@jiffies ^$stack'	# BAD_VAR
+check_error 'w w@jiffies ^%ax'		# BAD_VAR
+
+exit 0


^ permalink raw reply related

* [PATCH v9 05/10] x86/hw_breakpoint: Unify breakpoint install/uninstall
From: Masami Hiramatsu (Google) @ 2026-07-17 14:20 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users
In-Reply-To: <178429796992.157981.3393977217853767915.stgit@devnote2>

From: Jinchao Wang <wangjinchao600@gmail.com>

Consolidate breakpoint management to reduce code duplication.
The diffstat was misleading, so the stripped code size is compared instead.
After refactoring, it is reduced from 11976 bytes to 11448 bytes on my
x86_64 system built with clang.

This also makes it easier to introduce arch_reinstall_hw_breakpoint().

In addition, including linux/types.h to fix a missing build dependency.

Assisted-by: Antigravity:gemini-3.5-flash
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v8:
  - Add missing barrier() on the disable path of setup_hwbp() to prevent
    the compiler from reordering this_cpu_write(cpu_dr7, ...) before
    set_debugreg(dr7, 7).
  - Clear existing slot control and enable bits in cpu_dr7 inside
    setup_hwbp() using __encode_dr7() before setting new ones to prevent
    register state corruption on update/reinstall.
---
 arch/x86/include/asm/hw_breakpoint.h |    6 +
 arch/x86/kernel/hw_breakpoint.c      |  144 +++++++++++++++++++---------------
 2 files changed, 86 insertions(+), 64 deletions(-)

diff --git a/arch/x86/include/asm/hw_breakpoint.h b/arch/x86/include/asm/hw_breakpoint.h
index 0bc931cd0698..aa6adac6c3a2 100644
--- a/arch/x86/include/asm/hw_breakpoint.h
+++ b/arch/x86/include/asm/hw_breakpoint.h
@@ -5,6 +5,7 @@
 #include <uapi/asm/hw_breakpoint.h>
 
 #define	__ARCH_HW_BREAKPOINT_H
+#include <linux/types.h>
 
 /*
  * The name should probably be something dealt in
@@ -18,6 +19,11 @@ struct arch_hw_breakpoint {
 	u8		type;
 };
 
+enum bp_slot_action {
+	BP_SLOT_ACTION_INSTALL,
+	BP_SLOT_ACTION_UNINSTALL,
+};
+
 #include <linux/kdebug.h>
 #include <linux/percpu.h>
 #include <linux/list.h>
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index f846c15f21ca..c323c2aab2af 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -49,7 +49,6 @@ static DEFINE_PER_CPU(unsigned long, cpu_debugreg[HBP_NUM]);
  */
 static DEFINE_PER_CPU(struct perf_event *, bp_per_reg[HBP_NUM]);
 
-
 static inline unsigned long
 __encode_dr7(int drnum, unsigned int len, unsigned int type)
 {
@@ -86,96 +85,113 @@ int decode_dr7(unsigned long dr7, int bpnum, unsigned *len, unsigned *type)
 }
 
 /*
- * Install a perf counter breakpoint.
- *
- * We seek a free debug address register and use it for this
- * breakpoint. Eventually we enable it in the debug control register.
- *
- * Atomic: we hold the counter->ctx->lock and we only handle variables
- * and registers local to this cpu.
+ * We seek a slot and change it or keep it based on the action.
+ * Returns slot number on success, negative error on failure.
+ * Must be called with IRQs disabled.
  */
-int arch_install_hw_breakpoint(struct perf_event *bp)
+static int manage_bp_slot(struct perf_event *bp, enum bp_slot_action action)
 {
-	struct arch_hw_breakpoint *info = counter_arch_bp(bp);
-	unsigned long *dr7;
-	int i;
-
-	lockdep_assert_irqs_disabled();
+	struct perf_event *old_bp;
+	struct perf_event *new_bp;
+	int slot;
+
+	switch (action) {
+	case BP_SLOT_ACTION_INSTALL:
+		old_bp = NULL;
+		new_bp = bp;
+		break;
+	case BP_SLOT_ACTION_UNINSTALL:
+		old_bp = bp;
+		new_bp = NULL;
+		break;
+	default:
+		return -EINVAL;
+	}
 
-	for (i = 0; i < HBP_NUM; i++) {
-		struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
+	for (slot = 0; slot < HBP_NUM; slot++) {
+		struct perf_event **curr = this_cpu_ptr(&bp_per_reg[slot]);
 
-		if (!*slot) {
-			*slot = bp;
-			break;
+		if (*curr == old_bp) {
+			*curr = new_bp;
+			return slot;
 		}
 	}
 
-	if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
-		return -EBUSY;
+	if (old_bp) {
+		WARN_ONCE(1, "Can't find matching breakpoint slot");
+		return -EINVAL;
+	}
 
-	set_debugreg(info->address, i);
-	__this_cpu_write(cpu_debugreg[i], info->address);
+	WARN_ONCE(1, "No free breakpoint slots");
+	return -EBUSY;
+}
 
-	dr7 = this_cpu_ptr(&cpu_dr7);
-	*dr7 |= encode_dr7(i, info->len, info->type);
+static void setup_hwbp(struct arch_hw_breakpoint *info, int slot, bool enable)
+{
+	unsigned long dr7;
+
+	set_debugreg(info->address, slot);
+	__this_cpu_write(cpu_debugreg[slot], info->address);
+
+	dr7 = this_cpu_read(cpu_dr7);
+	dr7 &= ~(__encode_dr7(slot, 0xc, 0x3) |
+		 (DR_LOCAL_ENABLE << (slot * DR_ENABLE_SIZE)));
+	if (enable)
+		dr7 |= encode_dr7(slot, info->len, info->type);
 
 	/*
-	 * Ensure we first write cpu_dr7 before we set the DR7 register.
-	 * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
+	 * Enabling:
+	 *   Ensure we first write cpu_dr7 before we set the DR7 register.
+	 *   This ensures an NMI never see cpu_dr7 0 when DR7 is not.
 	 */
+	if (enable)
+		this_cpu_write(cpu_dr7, dr7);
+
 	barrier();
 
-	set_debugreg(*dr7, 7);
-	if (info->mask)
-		amd_set_dr_addr_mask(info->mask, i);
+	set_debugreg(dr7, 7);
+
+	amd_set_dr_addr_mask(enable ? info->mask : 0, slot);
 
-	return 0;
+	barrier();
+
+	/*
+	 * Disabling:
+	 *   Ensure the write to cpu_dr7 is after we've set the DR7 register.
+	 *   This ensures an NMI never see cpu_dr7 0 when DR7 is not.
+	 */
+	if (!enable)
+		this_cpu_write(cpu_dr7, dr7);
 }
 
 /*
- * Uninstall the breakpoint contained in the given counter.
- *
- * First we search the debug address register it uses and then we disable
- * it.
- *
- * Atomic: we hold the counter->ctx->lock and we only handle variables
- * and registers local to this cpu.
+ * find suitable breakpoint slot and set it up based on the action
  */
-void arch_uninstall_hw_breakpoint(struct perf_event *bp)
+static int arch_manage_bp(struct perf_event *bp, enum bp_slot_action action)
 {
-	struct arch_hw_breakpoint *info = counter_arch_bp(bp);
-	unsigned long dr7;
-	int i;
+	struct arch_hw_breakpoint *info;
+	int slot;
 
 	lockdep_assert_irqs_disabled();
 
-	for (i = 0; i < HBP_NUM; i++) {
-		struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
-
-		if (*slot == bp) {
-			*slot = NULL;
-			break;
-		}
-	}
+	slot = manage_bp_slot(bp, action);
+	if (slot < 0)
+		return slot;
 
-	if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
-		return;
+	info = counter_arch_bp(bp);
+	setup_hwbp(info, slot, action != BP_SLOT_ACTION_UNINSTALL);
 
-	dr7 = this_cpu_read(cpu_dr7);
-	dr7 &= ~__encode_dr7(i, info->len, info->type);
-
-	set_debugreg(dr7, 7);
-	if (info->mask)
-		amd_set_dr_addr_mask(0, i);
+	return 0;
+}
 
-	/*
-	 * Ensure the write to cpu_dr7 is after we've set the DR7 register.
-	 * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
-	 */
-	barrier();
+int arch_install_hw_breakpoint(struct perf_event *bp)
+{
+	return arch_manage_bp(bp, BP_SLOT_ACTION_INSTALL);
+}
 
-	this_cpu_write(cpu_dr7, dr7);
+void arch_uninstall_hw_breakpoint(struct perf_event *bp)
+{
+	arch_manage_bp(bp, BP_SLOT_ACTION_UNINSTALL);
 }
 
 static int arch_bp_generic_len(int x86_len)


^ permalink raw reply related

* [PATCH v9 06/10] x86/hw_breakpoint: Add arch_reinstall_hw_breakpoint
From: Masami Hiramatsu (Google) @ 2026-07-17 14:20 UTC (permalink / raw)
  To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
  Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
	Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
	Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
	linux-doc, linux-perf-users
In-Reply-To: <178429796992.157981.3393977217853767915.stgit@devnote2>

From: Jinchao Wang <wangjinchao600@gmail.com>

The new arch_reinstall_hw_breakpoint() function can be used in an
atomic context, unlike the more expensive free and re-allocation path.
This allows callers to efficiently re-establish an existing breakpoint.

Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Reviewed-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
 Changes in v9:
  - Update commit message.
  - Temporarily disable the active slot in setup_hwbp() before updating
    the address register to avoid spurious debug exceptions.
---
 arch/x86/include/asm/hw_breakpoint.h |    2 ++
 arch/x86/kernel/hw_breakpoint.c      |   34 ++++++++++++++++++++++++++++------
 2 files changed, 30 insertions(+), 6 deletions(-)

diff --git a/arch/x86/include/asm/hw_breakpoint.h b/arch/x86/include/asm/hw_breakpoint.h
index aa6adac6c3a2..c22cc4e87fc5 100644
--- a/arch/x86/include/asm/hw_breakpoint.h
+++ b/arch/x86/include/asm/hw_breakpoint.h
@@ -21,6 +21,7 @@ struct arch_hw_breakpoint {
 
 enum bp_slot_action {
 	BP_SLOT_ACTION_INSTALL,
+	BP_SLOT_ACTION_REINSTALL,
 	BP_SLOT_ACTION_UNINSTALL,
 };
 
@@ -65,6 +66,7 @@ extern int hw_breakpoint_exceptions_notify(struct notifier_block *unused,
 
 
 int arch_install_hw_breakpoint(struct perf_event *bp);
+int arch_reinstall_hw_breakpoint(struct perf_event *bp);
 void arch_uninstall_hw_breakpoint(struct perf_event *bp);
 void hw_breakpoint_pmu_read(struct perf_event *bp);
 void hw_breakpoint_pmu_unthrottle(struct perf_event *bp);
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index c323c2aab2af..0df3ff556f47 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -100,6 +100,10 @@ static int manage_bp_slot(struct perf_event *bp, enum bp_slot_action action)
 		old_bp = NULL;
 		new_bp = bp;
 		break;
+	case BP_SLOT_ACTION_REINSTALL:
+		old_bp = bp;
+		new_bp = bp;
+		break;
 	case BP_SLOT_ACTION_UNINSTALL:
 		old_bp = bp;
 		new_bp = NULL;
@@ -129,23 +133,36 @@ static int manage_bp_slot(struct perf_event *bp, enum bp_slot_action action)
 static void setup_hwbp(struct arch_hw_breakpoint *info, int slot, bool enable)
 {
 	unsigned long dr7;
-
-	set_debugreg(info->address, slot);
-	__this_cpu_write(cpu_debugreg[slot], info->address);
+	bool enabled;
 
 	dr7 = this_cpu_read(cpu_dr7);
+	enabled = dr7 & ((DR_LOCAL_ENABLE | DR_GLOBAL_ENABLE) << (slot * DR_ENABLE_SIZE));
 	dr7 &= ~(__encode_dr7(slot, 0xc, 0x3) |
 		 (DR_LOCAL_ENABLE << (slot * DR_ENABLE_SIZE)));
-	if (enable)
-		dr7 |= encode_dr7(slot, info->len, info->type);
+
+	/*
+	 * If the slot is currently enabled, disable it first before updating
+	 * the address register to prevent spurious debug exceptions.
+	 */
+	if (enable && enabled) {
+		barrier();
+		set_debugreg(dr7, 7);
+		barrier();
+		this_cpu_write(cpu_dr7, dr7);
+	}
+
+	set_debugreg(info->address, slot);
+	__this_cpu_write(cpu_debugreg[slot], info->address);
 
 	/*
 	 * Enabling:
 	 *   Ensure we first write cpu_dr7 before we set the DR7 register.
 	 *   This ensures an NMI never see cpu_dr7 0 when DR7 is not.
 	 */
-	if (enable)
+	if (enable) {
+		dr7 |= encode_dr7(slot, info->len, info->type);
 		this_cpu_write(cpu_dr7, dr7);
+	}
 
 	barrier();
 
@@ -189,6 +206,11 @@ int arch_install_hw_breakpoint(struct perf_event *bp)
 	return arch_manage_bp(bp, BP_SLOT_ACTION_INSTALL);
 }
 
+int arch_reinstall_hw_breakpoint(struct perf_event *bp)
+{
+	return arch_manage_bp(bp, BP_SLOT_ACTION_REINSTALL);
+}
+
 void arch_uninstall_hw_breakpoint(struct perf_event *bp)
 {
 	arch_manage_bp(bp, BP_SLOT_ACTION_UNINSTALL);


^ permalink raw reply related


This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox