* [PATCH v7 08/10] HWBP: Add modify_wide_hw_breakpoint_local() API
From: Masami Hiramatsu (Google) @ 2026-07-15 1:45 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add modify_wide_hw_breakpoint_local() arch-wide interface which allows
hwbp users to update watch address on-line. This is available if the
arch supports CONFIG_HAVE_REINSTALL_HW_BREAKPOINT.
Note that this allows to change the type only for compatible types,
because it does not release and reserve the hwbp slot based on type.
For instance, you can not change HW_BREAKPOINT_W to HW_BREAKPOINT_X.
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v7:
- Update bp->attr.bp_attr so that we can correctly check the
address on it.
- Use -EOPNOTSUPP instead of -ENOSYS.
Changes in v4:
- Update kerneldoc comment about modify_wide_hw_breakpoint_local
according to Randy's comment.
Changes in v2:
- Check type compatibility by checking slot. (Thanks Jinchao!)
---
arch/Kconfig | 10 ++++++++++
arch/x86/Kconfig | 1 +
include/linux/hw_breakpoint.h | 6 ++++++
kernel/events/hw_breakpoint.c | 39 +++++++++++++++++++++++++++++++++++++++
4 files changed, 56 insertions(+)
diff --git a/arch/Kconfig b/arch/Kconfig
index 959aee9568ff..4a87e843bb4c 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -467,6 +467,16 @@ config HAVE_POST_BREAKPOINT_HOOK
Select this option if your arch implements breakpoints overflow
handler hooks after the target memory is modified.
+config HAVE_REINSTALL_HW_BREAKPOINT
+ bool
+ depends on HAVE_HW_BREAKPOINT
+ help
+ Depending on the arch implementation of hardware breakpoints,
+ some of them are able to update the breakpoint configuration
+ without release and reserve the hardware breakpoint register.
+ What configuration is able to update depends on hardware and
+ software implementation.
+
config HAVE_USER_RETURN_NOTIFIER
bool
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index 6b7e14ef8cfb..588218da8f41 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -247,6 +247,7 @@ config X86
select HAVE_GCC_PLUGINS
select HAVE_HW_BREAKPOINT
select HAVE_POST_BREAKPOINT_HOOK
+ select HAVE_REINSTALL_HW_BREAKPOINT
select HAVE_IOREMAP_PROT
select HAVE_IRQ_EXIT_ON_IRQ_STACK if X86_64
select HAVE_IRQ_TIME_ACCOUNTING
diff --git a/include/linux/hw_breakpoint.h b/include/linux/hw_breakpoint.h
index db199d653dd1..6754ffbee9ed 100644
--- a/include/linux/hw_breakpoint.h
+++ b/include/linux/hw_breakpoint.h
@@ -81,6 +81,9 @@ register_wide_hw_breakpoint(struct perf_event_attr *attr,
perf_overflow_handler_t triggered,
void *context);
+extern int modify_wide_hw_breakpoint_local(struct perf_event *bp,
+ struct perf_event_attr *attr);
+
extern int register_perf_hw_breakpoint(struct perf_event *bp);
extern void unregister_hw_breakpoint(struct perf_event *bp);
extern void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events);
@@ -124,6 +127,9 @@ register_wide_hw_breakpoint(struct perf_event_attr *attr,
perf_overflow_handler_t triggered,
void *context) { return NULL; }
static inline int
+modify_wide_hw_breakpoint_local(struct perf_event *bp,
+ struct perf_event_attr *attr) { return -EOPNOTSUPP; }
+static inline int
register_perf_hw_breakpoint(struct perf_event *bp) { return -ENOSYS; }
static inline void unregister_hw_breakpoint(struct perf_event *bp) { }
static inline void
diff --git a/kernel/events/hw_breakpoint.c b/kernel/events/hw_breakpoint.c
index 789add0c185a..4337688da397 100644
--- a/kernel/events/hw_breakpoint.c
+++ b/kernel/events/hw_breakpoint.c
@@ -888,6 +888,45 @@ void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events)
}
EXPORT_SYMBOL_GPL(unregister_wide_hw_breakpoint);
+/**
+ * modify_wide_hw_breakpoint_local - update breakpoint config for local CPU
+ * @bp: the hwbp perf event for this CPU
+ * @attr: the new attribute for @bp
+ *
+ * This does not release and reserve the slot of a HWBP; it just reuses the
+ * current slot on local CPU. So the users must update the other CPUs by
+ * themselves.
+ * Also, since this does not release/reserve the slot, this can not change the
+ * type to incompatible type of the HWBP.
+ * Return err if attr is invalid or the CPU fails to update debug register
+ * for new @attr.
+ */
+#ifdef CONFIG_HAVE_REINSTALL_HW_BREAKPOINT
+int modify_wide_hw_breakpoint_local(struct perf_event *bp,
+ struct perf_event_attr *attr)
+{
+ int ret;
+
+ if (find_slot_idx(bp->attr.bp_type) != find_slot_idx(attr->bp_type))
+ return -EINVAL;
+
+ ret = hw_breakpoint_arch_parse(bp, attr, counter_arch_bp(bp));
+ if (ret)
+ return ret;
+
+ bp->attr.bp_addr = attr->bp_addr;
+
+ return arch_reinstall_hw_breakpoint(bp);
+}
+#else
+int modify_wide_hw_breakpoint_local(struct perf_event *bp,
+ struct perf_event_attr *attr)
+{
+ return -EOPNOTSUPP;
+}
+#endif
+EXPORT_SYMBOL_GPL(modify_wide_hw_breakpoint_local);
+
/**
* hw_breakpoint_is_used - check if breakpoints are currently used
*
^ permalink raw reply related
* [PATCH v7 07/10] x86/hw_breakpoint: Add arch_reinstall_hw_breakpoint
From: Masami Hiramatsu (Google) @ 2026-07-15 1:45 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Jinchao Wang <wangjinchao600@gmail.com>
The new arch_reinstall_hw_breakpoint() function can be used in an
atomic context, unlike the more expensive free and re-allocation path.
This allows callers to efficiently re-establish an existing breakpoint.
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Reviewed-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
arch/x86/include/asm/hw_breakpoint.h | 2 ++
arch/x86/kernel/hw_breakpoint.c | 9 +++++++++
2 files changed, 11 insertions(+)
diff --git a/arch/x86/include/asm/hw_breakpoint.h b/arch/x86/include/asm/hw_breakpoint.h
index aa6adac6c3a2..c22cc4e87fc5 100644
--- a/arch/x86/include/asm/hw_breakpoint.h
+++ b/arch/x86/include/asm/hw_breakpoint.h
@@ -21,6 +21,7 @@ struct arch_hw_breakpoint {
enum bp_slot_action {
BP_SLOT_ACTION_INSTALL,
+ BP_SLOT_ACTION_REINSTALL,
BP_SLOT_ACTION_UNINSTALL,
};
@@ -65,6 +66,7 @@ extern int hw_breakpoint_exceptions_notify(struct notifier_block *unused,
int arch_install_hw_breakpoint(struct perf_event *bp);
+int arch_reinstall_hw_breakpoint(struct perf_event *bp);
void arch_uninstall_hw_breakpoint(struct perf_event *bp);
void hw_breakpoint_pmu_read(struct perf_event *bp);
void hw_breakpoint_pmu_unthrottle(struct perf_event *bp);
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index 877509539300..9af8d81075db 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -100,6 +100,10 @@ static int manage_bp_slot(struct perf_event *bp, enum bp_slot_action action)
old_bp = NULL;
new_bp = bp;
break;
+ case BP_SLOT_ACTION_REINSTALL:
+ old_bp = bp;
+ new_bp = bp;
+ break;
case BP_SLOT_ACTION_UNINSTALL:
old_bp = bp;
new_bp = NULL;
@@ -188,6 +192,11 @@ int arch_install_hw_breakpoint(struct perf_event *bp)
return arch_manage_bp(bp, BP_SLOT_ACTION_INSTALL);
}
+int arch_reinstall_hw_breakpoint(struct perf_event *bp)
+{
+ return arch_manage_bp(bp, BP_SLOT_ACTION_REINSTALL);
+}
+
void arch_uninstall_hw_breakpoint(struct perf_event *bp)
{
arch_manage_bp(bp, BP_SLOT_ACTION_UNINSTALL);
^ permalink raw reply related
* [PATCH v7 06/10] x86/hw_breakpoint: Unify breakpoint install/uninstall
From: Masami Hiramatsu (Google) @ 2026-07-15 1:45 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Jinchao Wang <wangjinchao600@gmail.com>
Consolidate breakpoint management to reduce code duplication.
The diffstat was misleading, so the stripped code size is compared instead.
After refactoring, it is reduced from 11976 bytes to 11448 bytes on my
x86_64 system built with clang.
This also makes it easier to introduce arch_reinstall_hw_breakpoint().
In addition, including linux/types.h to fix a missing build dependency.
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Reviewed-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
arch/x86/include/asm/hw_breakpoint.h | 6 +
arch/x86/kernel/hw_breakpoint.c | 141 +++++++++++++++++++---------------
2 files changed, 84 insertions(+), 63 deletions(-)
diff --git a/arch/x86/include/asm/hw_breakpoint.h b/arch/x86/include/asm/hw_breakpoint.h
index 0bc931cd0698..aa6adac6c3a2 100644
--- a/arch/x86/include/asm/hw_breakpoint.h
+++ b/arch/x86/include/asm/hw_breakpoint.h
@@ -5,6 +5,7 @@
#include <uapi/asm/hw_breakpoint.h>
#define __ARCH_HW_BREAKPOINT_H
+#include <linux/types.h>
/*
* The name should probably be something dealt in
@@ -18,6 +19,11 @@ struct arch_hw_breakpoint {
u8 type;
};
+enum bp_slot_action {
+ BP_SLOT_ACTION_INSTALL,
+ BP_SLOT_ACTION_UNINSTALL,
+};
+
#include <linux/kdebug.h>
#include <linux/percpu.h>
#include <linux/list.h>
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index f846c15f21ca..877509539300 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -49,7 +49,6 @@ static DEFINE_PER_CPU(unsigned long, cpu_debugreg[HBP_NUM]);
*/
static DEFINE_PER_CPU(struct perf_event *, bp_per_reg[HBP_NUM]);
-
static inline unsigned long
__encode_dr7(int drnum, unsigned int len, unsigned int type)
{
@@ -86,96 +85,112 @@ int decode_dr7(unsigned long dr7, int bpnum, unsigned *len, unsigned *type)
}
/*
- * Install a perf counter breakpoint.
- *
- * We seek a free debug address register and use it for this
- * breakpoint. Eventually we enable it in the debug control register.
- *
- * Atomic: we hold the counter->ctx->lock and we only handle variables
- * and registers local to this cpu.
+ * We seek a slot and change it or keep it based on the action.
+ * Returns slot number on success, negative error on failure.
+ * Must be called with IRQs disabled.
*/
-int arch_install_hw_breakpoint(struct perf_event *bp)
+static int manage_bp_slot(struct perf_event *bp, enum bp_slot_action action)
{
- struct arch_hw_breakpoint *info = counter_arch_bp(bp);
- unsigned long *dr7;
- int i;
-
- lockdep_assert_irqs_disabled();
+ struct perf_event *old_bp;
+ struct perf_event *new_bp;
+ int slot;
+
+ switch (action) {
+ case BP_SLOT_ACTION_INSTALL:
+ old_bp = NULL;
+ new_bp = bp;
+ break;
+ case BP_SLOT_ACTION_UNINSTALL:
+ old_bp = bp;
+ new_bp = NULL;
+ break;
+ default:
+ return -EINVAL;
+ }
- for (i = 0; i < HBP_NUM; i++) {
- struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
+ for (slot = 0; slot < HBP_NUM; slot++) {
+ struct perf_event **curr = this_cpu_ptr(&bp_per_reg[slot]);
- if (!*slot) {
- *slot = bp;
- break;
+ if (*curr == old_bp) {
+ *curr = new_bp;
+ return slot;
}
}
- if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
- return -EBUSY;
+ if (old_bp) {
+ WARN_ONCE(1, "Can't find matching breakpoint slot");
+ return -EINVAL;
+ }
+
+ WARN_ONCE(1, "No free breakpoint slots");
+ return -EBUSY;
+}
+
+static void setup_hwbp(struct arch_hw_breakpoint *info, int slot, bool enable)
+{
+ unsigned long dr7;
- set_debugreg(info->address, i);
- __this_cpu_write(cpu_debugreg[i], info->address);
+ set_debugreg(info->address, slot);
+ __this_cpu_write(cpu_debugreg[slot], info->address);
- dr7 = this_cpu_ptr(&cpu_dr7);
- *dr7 |= encode_dr7(i, info->len, info->type);
+ dr7 = this_cpu_read(cpu_dr7);
+ if (enable)
+ dr7 |= encode_dr7(slot, info->len, info->type);
+ else
+ dr7 &= ~__encode_dr7(slot, info->len, info->type);
/*
- * Ensure we first write cpu_dr7 before we set the DR7 register.
- * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
+ * Enabling:
+ * Ensure we first write cpu_dr7 before we set the DR7 register.
+ * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
*/
+ if (enable)
+ this_cpu_write(cpu_dr7, dr7);
+
barrier();
- set_debugreg(*dr7, 7);
+ set_debugreg(dr7, 7);
+
if (info->mask)
- amd_set_dr_addr_mask(info->mask, i);
+ amd_set_dr_addr_mask(enable ? info->mask : 0, slot);
- return 0;
+ /*
+ * Disabling:
+ * Ensure the write to cpu_dr7 is after we've set the DR7 register.
+ * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
+ */
+ if (!enable)
+ this_cpu_write(cpu_dr7, dr7);
}
/*
- * Uninstall the breakpoint contained in the given counter.
- *
- * First we search the debug address register it uses and then we disable
- * it.
- *
- * Atomic: we hold the counter->ctx->lock and we only handle variables
- * and registers local to this cpu.
+ * find suitable breakpoint slot and set it up based on the action
*/
-void arch_uninstall_hw_breakpoint(struct perf_event *bp)
+static int arch_manage_bp(struct perf_event *bp, enum bp_slot_action action)
{
- struct arch_hw_breakpoint *info = counter_arch_bp(bp);
- unsigned long dr7;
- int i;
+ struct arch_hw_breakpoint *info;
+ int slot;
lockdep_assert_irqs_disabled();
- for (i = 0; i < HBP_NUM; i++) {
- struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
-
- if (*slot == bp) {
- *slot = NULL;
- break;
- }
- }
-
- if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
- return;
+ slot = manage_bp_slot(bp, action);
+ if (slot < 0)
+ return slot;
- dr7 = this_cpu_read(cpu_dr7);
- dr7 &= ~__encode_dr7(i, info->len, info->type);
+ info = counter_arch_bp(bp);
+ setup_hwbp(info, slot, action != BP_SLOT_ACTION_UNINSTALL);
- set_debugreg(dr7, 7);
- if (info->mask)
- amd_set_dr_addr_mask(0, i);
+ return 0;
+}
- /*
- * Ensure the write to cpu_dr7 is after we've set the DR7 register.
- * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
- */
- barrier();
+int arch_install_hw_breakpoint(struct perf_event *bp)
+{
+ return arch_manage_bp(bp, BP_SLOT_ACTION_INSTALL);
+}
- this_cpu_write(cpu_dr7, dr7);
+void arch_uninstall_hw_breakpoint(struct perf_event *bp)
+{
+ arch_manage_bp(bp, BP_SLOT_ACTION_UNINSTALL);
}
static int arch_bp_generic_len(int x86_len)
^ permalink raw reply related
* [PATCH v7 05/10] tracing: wprobe: Use a new seq_print_ip_sym_offset() wrapper
From: Masami Hiramatsu (Google) @ 2026-07-15 1:44 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Use a new seq_print_ip_sym_offset() wrapper function instead of
using TRACE_ITER(SYM_OFFSET) mask directly.
Link: https://lore.kernel.org/all/176226550596.59499.18020648957674458755.stgit@devnote2/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
kernel/trace/trace_wprobe.c | 1 +
1 file changed, 1 insertion(+)
diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
index b52f3eac719f..dd310a87b333 100644
--- a/kernel/trace/trace_wprobe.c
+++ b/kernel/trace/trace_wprobe.c
@@ -20,6 +20,7 @@
#include <asm/ptrace.h>
#include "trace_dynevent.h"
+#include "trace_output.h"
#include "trace_probe.h"
#include "trace_probe_kernel.h"
#include "trace_probe_tmpl.h"
^ permalink raw reply related
* [PATCH v7 04/10] selftests: tracing: Add syntax testcase for wprobe
From: Masami Hiramatsu (Google) @ 2026-07-15 1:44 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add "wprobe_syntax_errors.tc" testcase for testing syntax errors
of the watch probe events.
Link: https://lore.kernel.org/all/175859027842.374439.6402700780945714048.stgit@devnote2/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
.../test.d/dynevent/wprobes_syntax_errors.tc | 20 ++++++++++++++++++++
1 file changed, 20 insertions(+)
create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
new file mode 100644
index 000000000000..56ac579d60ae
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
@@ -0,0 +1,20 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: Watch probe event parser error log check
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README
+
+check_error() { # command-with-error-pos-by-^
+ ftrace_errlog_check 'wprobe' "$1" 'dynamic_events'
+}
+
+check_error 'w ^symbol' # BAD_ACCESS_FMT
+check_error 'w ^a@symbol' # BAD_ACCESS_TYPE
+check_error 'w w@^symbol' # BAD_ACCESS_ADDR
+check_error 'w w@jiffies^+offset' # BAD_ACCESS_ADDR
+check_error 'w w@jiffies:^100' # BAD_ACCESS_LEN
+check_error 'w w@jiffies ^$arg1' # BAD_VAR
+check_error 'w w@jiffies ^$retval' # BAD_VAR
+check_error 'w w@jiffies ^$stack' # BAD_VAR
+check_error 'w w@jiffies ^%ax' # BAD_VAR
+
+exit 0
^ permalink raw reply related
* [PATCH v7 03/10] selftests: tracing: Add a basic testcase for wprobe
From: Masami Hiramatsu (Google) @ 2026-07-15 1:44 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add 'add_remove_wprobe.tc' testcase for testing wprobe event that
tests adding and removing operations of the wprobe event.
Link: https://lore.kernel.org/all/175859026716.374439.14852239332989324292.stgit@devnote2/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
tools/testing/selftests/ftrace/config | 1
.../ftrace/test.d/dynevent/add_remove_wprobe.tc | 68 ++++++++++++++++++++
2 files changed, 69 insertions(+)
create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
diff --git a/tools/testing/selftests/ftrace/config b/tools/testing/selftests/ftrace/config
index 544de0db5f58..d2f503722020 100644
--- a/tools/testing/selftests/ftrace/config
+++ b/tools/testing/selftests/ftrace/config
@@ -27,3 +27,4 @@ CONFIG_STACK_TRACER=y
CONFIG_TRACER_SNAPSHOT=y
CONFIG_UPROBES=y
CONFIG_UPROBE_EVENTS=y
+CONFIG_WPROBE_EVENTS=y
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
new file mode 100644
index 000000000000..20774c7f69f8
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
@@ -0,0 +1,68 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: Generic dynamic event - add/remove wprobe events
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README
+
+echo 0 > events/enable
+echo > dynamic_events
+
+# Use jiffies as a variable that is frequently written to.
+TARGET=jiffies
+
+echo "w:my_wprobe w@$TARGET" >> dynamic_events
+
+grep -q my_wprobe dynamic_events
+if [ $? -ne 0 ]; then
+ echo "Failed to create wprobe event"
+ exit_fail
+fi
+
+test -d events/wprobes/my_wprobe
+if [ $? -ne 0 ]; then
+ echo "Failed to create wprobe event directory"
+ exit_fail
+fi
+
+echo 1 > events/wprobes/my_wprobe/enable
+
+# Check if the event is enabled
+cat events/wprobes/my_wprobe/enable | grep -q 1
+if [ $? -ne 0 ]; then
+ echo "Failed to enable wprobe event"
+ exit_fail
+fi
+
+# Let some time pass to trigger the breakpoint
+sleep 1
+
+# Check if we got any trace output
+if !grep -q my_wprobe trace; then
+ echo "wprobe event was not triggered"
+fi
+
+echo 0 > events/wprobes/my_wprobe/enable
+
+# Check if the event is disabled
+cat events/wprobes/my_wprobe/enable | grep -q 0
+if [ $? -ne 0 ]; then
+ echo "Failed to disable wprobe event"
+ exit_fail
+fi
+
+echo "-:my_wprobe" >> dynamic_events
+
+! grep -q my_wprobe dynamic_events
+if [ $? -ne 0 ]; then
+ echo "Failed to remove wprobe event"
+ exit_fail
+fi
+
+! test -d events/wprobes/my_wprobe
+if [ $? -ne 0 ]; then
+ echo "Failed to remove wprobe event directory"
+ exit_fail
+fi
+
+clear_trace
+
+exit 0
^ permalink raw reply related
* [PATCH v7 02/10] x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires
From: Masami Hiramatsu (Google) @ 2026-07-15 1:44 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add CONFIG_HAVE_POST_BREAKPOINT_HOOK which indicates the hw_breakpoint
on that architecture fires after the target memory has been modified.
This is currently x86 only behavior.
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
arch/Kconfig | 10 ++++++++++
arch/x86/Kconfig | 1 +
kernel/trace/Kconfig | 1 +
3 files changed, 12 insertions(+)
diff --git a/arch/Kconfig b/arch/Kconfig
index fa7507ac8e13..959aee9568ff 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -457,6 +457,16 @@ config HAVE_MIXED_BREAKPOINTS_REGS
Select this option if your arch implements breakpoints under the
latter fashion.
+config HAVE_POST_BREAKPOINT_HOOK
+ bool
+ depends on HAVE_HW_BREAKPOINT
+ help
+ Depending on the arch implementation of hardware breakpoints,
+ some of them provide breakpoint hook after the target memory
+ is modified.
+ Select this option if your arch implements breakpoints overflow
+ handler hooks after the target memory is modified.
+
config HAVE_USER_RETURN_NOTIFIER
bool
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index bdad90f210e4..6b7e14ef8cfb 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -246,6 +246,7 @@ config X86
select HAVE_FUNCTION_TRACER
select HAVE_GCC_PLUGINS
select HAVE_HW_BREAKPOINT
+ select HAVE_POST_BREAKPOINT_HOOK
select HAVE_IOREMAP_PROT
select HAVE_IRQ_EXIT_ON_IRQ_STACK if X86_64
select HAVE_IRQ_TIME_ACCOUNTING
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index b58c2565024f..d9b6fa5c35d9 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -866,6 +866,7 @@ config WPROBE_EVENTS
bool "Enable wprobe-based dynamic events"
depends on TRACING
depends on HAVE_HW_BREAKPOINT
+ depends on HAVE_POST_BREAKPOINT_HOOK
select PROBE_EVENTS
select DYNAMIC_EVENTS
help
^ permalink raw reply related
* [PATCH v7 01/10] tracing: wprobe: Add watchpoint probe event based on hardware breakpoint
From: Masami Hiramatsu (Google) @ 2026-07-15 1:44 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
In-Reply-To: <178407983818.95826.12714571928538799781.stgit@devnote2>
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add a new probe event for the hardware breakpoint called wprobe-event.
This wprobe allows user to trace (watch) the memory access at the
specified memory address.
The new syntax is;
w[:[GROUP/]EVENT] [r|w|rw]@[ADDR|SYM][:SIZE] [FETCH_ARGs]
User also can use $addr to fetch the accessed address. But no other
variables are supported. To record updated value, use '+0($addr)'.
For example, tracing updates of the jiffies;
/sys/kernel/tracing # echo 'w:my_jiffies w@jiffies' >> dynamic_events
/sys/kernel/tracing # cat dynamic_events
w:wprobes/my_jiffies w@jiffies:4
/sys/kernel/tracing # echo 1 > events/wprobes/my_jiffies/enable
/sys/kernel/tracing # head -n 20 trace | tail -n 5
# TASK-PID CPU# ||||| TIMESTAMP FUNCTION
# | | | ||||| | |
<idle>-0 [000] d.Z1. 206.547317: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
<idle>-0 [000] d.Z1. 206.548341: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
<idle>-0 [000] d.Z1. 206.549346: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
Link: https://lore.kernel.org/all/175859021100.374439.8723137923620348816.stgit@devnote2/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v7:
- Include IS_ERR_PCPU fix.
- use seq_print_ip_sym_offset().
- fix checkpatch warning on DEFINE_FREE()
- Use bp->attr.bp_addr instead of tw->addr because it can be updated from another CPU.
---
Documentation/trace/index.rst | 1
Documentation/trace/wprobetrace.rst | 69 +++
include/linux/trace_events.h | 2
kernel/trace/Kconfig | 13 +
kernel/trace/Makefile | 1
kernel/trace/trace.c | 9
kernel/trace/trace.h | 5
kernel/trace/trace_probe.c | 22 +
kernel/trace/trace_probe.h | 8
kernel/trace/trace_wprobe.c | 692 +++++++++++++++++++++++++++++++++++
10 files changed, 819 insertions(+), 3 deletions(-)
create mode 100644 Documentation/trace/wprobetrace.rst
create mode 100644 kernel/trace/trace_wprobe.c
diff --git a/Documentation/trace/index.rst b/Documentation/trace/index.rst
index 5d9bf4694d5d..2f04f32001ed 100644
--- a/Documentation/trace/index.rst
+++ b/Documentation/trace/index.rst
@@ -36,6 +36,7 @@ the Linux kernel.
kprobes
kprobetrace
fprobetrace
+ wprobetrace
eprobetrace
fprobe
ring-buffer-design
diff --git a/Documentation/trace/wprobetrace.rst b/Documentation/trace/wprobetrace.rst
new file mode 100644
index 000000000000..025b4c39b809
--- /dev/null
+++ b/Documentation/trace/wprobetrace.rst
@@ -0,0 +1,69 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+=======================================
+Watchpoint probe (wprobe) Event Tracing
+=======================================
+
+.. Author: Masami Hiramatsu <mhiramat@kernel.org>
+
+Overview
+--------
+
+Wprobe event is a dynamic event based on the hardware breakpoint, which is
+similar to other probe events, but it is for watching data access. It allows
+you to trace which code accesses a specified data.
+
+As same as other dynamic events, wprobe events are defined via
+`dynamic_events` interface file on tracefs.
+
+Synopsis of wprobe-events
+-------------------------
+::
+
+ w:[GRP/][EVENT] SPEC [FETCHARGS] : Probe on data access
+
+ GRP : Group name for wprobe. If omitted, use "wprobes" for it.
+ EVENT : Event name for wprobe. If omitted, an event name is
+ generated based on the address or symbol.
+ SPEC : Breakpoint specification.
+ [r|w|rw]@<ADDRESS|SYMBOL[+|-OFFS]>[:LENGTH]
+
+ r|w|rw : Access type, r for read, w for write, and rw for both.
+ Default is rw if omitted.
+ ADDRESS : Address to trace (hexadecimal).
+ SYMBOL : Symbol name to trace.
+ LENGTH : Length of the data to trace in bytes. (1, 2, 4, or 8)
+
+ FETCHARGS : Arguments. Each probe can have up to 128 args.
+ $addr : Fetch the accessing address.
+ @ADDR : Fetch memory at ADDR (ADDR should be in kernel)
+ @SYM[+|-offs] : Fetch memory at SYM +|- offs (SYM should be a data symbol)
+ +|-[u]OFFS(FETCHARG) : Fetch memory at FETCHARG +|- OFFS address.(\*1)(\*2)
+ \IMM : Store an immediate value to the argument.
+ NAME=FETCHARG : Set NAME as the argument name of FETCHARG.
+ FETCHARG:TYPE : Set TYPE as the type of FETCHARG. Currently, basic types
+ (u8/u16/u32/u64/s8/s16/s32/s64), hexadecimal types
+ (x8/x16/x32/x64), "char", "string", "ustring", "symbol", "symstr"
+ and bitfield are supported.
+
+ (\*1) this is useful for fetching a field of data structures.
+ (\*2) "u" means user-space dereference.
+
+For the details of TYPE, see :ref:`kprobetrace documentation <kprobetrace_types>`.
+
+Usage examples
+--------------
+Here is an example to add a wprobe event on a variable `jiffies`.
+::
+
+ # echo 'w:my_jiffies w@jiffies' >> dynamic_events
+ # cat dynamic_events
+ w:wprobes/my_jiffies w@jiffies
+ # echo 1 > events/wprobes/enable
+ # cat trace | head
+ # TASK-PID CPU# ||||| TIMESTAMP FUNCTION
+ # | | | ||||| | |
+ <idle>-0 [000] d.Z1. 717.026259: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
+ <idle>-0 [000] d.Z1. 717.026373: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
+
+You can see the code which writes to `jiffies` is `tick_do_update_jiffies64()`.
diff --git a/include/linux/trace_events.h b/include/linux/trace_events.h
index 308c76b57d13..d1e5ab71d928 100644
--- a/include/linux/trace_events.h
+++ b/include/linux/trace_events.h
@@ -328,6 +328,7 @@ enum {
TRACE_EVENT_FL_UPROBE_BIT,
TRACE_EVENT_FL_EPROBE_BIT,
TRACE_EVENT_FL_FPROBE_BIT,
+ TRACE_EVENT_FL_WPROBE_BIT,
TRACE_EVENT_FL_CUSTOM_BIT,
TRACE_EVENT_FL_TEST_STR_BIT,
};
@@ -358,6 +359,7 @@ enum {
TRACE_EVENT_FL_UPROBE = (1 << TRACE_EVENT_FL_UPROBE_BIT),
TRACE_EVENT_FL_EPROBE = (1 << TRACE_EVENT_FL_EPROBE_BIT),
TRACE_EVENT_FL_FPROBE = (1 << TRACE_EVENT_FL_FPROBE_BIT),
+ TRACE_EVENT_FL_WPROBE = (1 << TRACE_EVENT_FL_WPROBE_BIT),
TRACE_EVENT_FL_CUSTOM = (1 << TRACE_EVENT_FL_CUSTOM_BIT),
TRACE_EVENT_FL_TEST_STR = (1 << TRACE_EVENT_FL_TEST_STR_BIT),
};
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index 0ab5916575a9..b58c2565024f 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -862,6 +862,19 @@ config EPROBE_EVENTS
convert the type of an event field. For example, turn an
address into a string.
+config WPROBE_EVENTS
+ bool "Enable wprobe-based dynamic events"
+ depends on TRACING
+ depends on HAVE_HW_BREAKPOINT
+ select PROBE_EVENTS
+ select DYNAMIC_EVENTS
+ help
+ This allows the user to add watchpoint tracing events based on
+ hardware breakpoints on the fly via the ftrace interface.
+
+ Those events can be inserted wherever hardware breakpoints can be
+ set, and record accessed memory address and values.
+
config BPF_EVENTS
depends on BPF_SYSCALL
depends on (KPROBE_EVENTS || UPROBE_EVENTS) && PERF_EVENTS
diff --git a/kernel/trace/Makefile b/kernel/trace/Makefile
index f934ff586bd4..141c8323de20 100644
--- a/kernel/trace/Makefile
+++ b/kernel/trace/Makefile
@@ -126,6 +126,7 @@ obj-$(CONFIG_FTRACE_RECORD_RECURSION) += trace_recursion_record.o
obj-$(CONFIG_FPROBE) += fprobe.o
obj-$(CONFIG_RETHOOK) += rethook.o
obj-$(CONFIG_FPROBE_EVENTS) += trace_fprobe.o
+obj-$(CONFIG_WPROBE_EVENTS) += trace_wprobe.o
obj-$(CONFIG_TRACEPOINT_BENCHMARK) += trace_benchmark.o
obj-$(CONFIG_RV) += rv/
diff --git a/kernel/trace/trace.c b/kernel/trace/trace.c
index c9e182d40059..1bc27c0ad029 100644
--- a/kernel/trace/trace.c
+++ b/kernel/trace/trace.c
@@ -4294,8 +4294,12 @@ static const char readme_msg[] =
" uprobe_events\t\t- Create/append/remove/show the userspace dynamic events\n"
"\t\t\t Write into this file to define/undefine new trace events.\n"
#endif
+#ifdef CONFIG_WPROBE_EVENTS
+ " wprobe_events\t\t- Create/append/remove/show the hardware breakpoint dynamic events\n"
+ "\t\t\t Write into this file to define/undefine new trace events.\n"
+#endif
#if defined(CONFIG_KPROBE_EVENTS) || defined(CONFIG_UPROBE_EVENTS) || \
- defined(CONFIG_FPROBE_EVENTS)
+ defined(CONFIG_FPROBE_EVENTS) || defined(CONFIG_WPROBE_EVENTS)
"\t accepts: event-definitions (one definition per line)\n"
#if defined(CONFIG_KPROBE_EVENTS) || defined(CONFIG_UPROBE_EVENTS)
"\t Format: p[:[<group>/][<event>]] <place> [<args>]\n"
@@ -4305,6 +4309,9 @@ static const char readme_msg[] =
"\t f[:[<group>/][<event>]] <func-name>[%return] [<args>]\n"
"\t t[:[<group>/][<event>]] <tracepoint> [<args>]\n"
#endif
+#ifdef CONFIG_WPROBE_EVENTS
+ "\t w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]\n"
+#endif
#ifdef CONFIG_HIST_TRIGGERS
"\t s:[synthetic/]<event> <field> [<field>]\n"
#endif
diff --git a/kernel/trace/trace.h b/kernel/trace/trace.h
index 80fe152af1dd..2f07c5c4ffc8 100644
--- a/kernel/trace/trace.h
+++ b/kernel/trace/trace.h
@@ -179,6 +179,11 @@ struct fexit_trace_entry_head {
unsigned long ret_ip;
};
+struct wprobe_trace_entry_head {
+ struct trace_entry ent;
+ unsigned long ip;
+};
+
#define TRACE_BUF_SIZE 1024
struct trace_array;
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 18c212122344..2600d9699bd8 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -1344,6 +1344,24 @@ static int parse_probe_vars(char *orig_arg, const struct fetch_type *t,
return 0;
}
+ /* wprobe only support "$addr" and "$value" variable */
+ if (ctx->flags & TPARG_FL_WPROBE) {
+ if (!strcmp(arg, "addr")) {
+ code->op = FETCH_OP_BADDR;
+ return 0;
+ }
+ if (!strcmp(arg, "value")) {
+ code->op = FETCH_OP_BADDR;
+ code++;
+ code->op = FETCH_OP_DEREF;
+ code->offset = 0;
+ *pcode = code;
+ return 0;
+ }
+ err = TP_ERR_BAD_VAR;
+ goto inval;
+ }
+
if (str_has_prefix(arg, "retval")) {
if (!(ctx->flags & TPARG_FL_RETURN)) {
err = TP_ERR_RETVAL_ON_PROBE;
@@ -1491,8 +1509,9 @@ parse_probe_arg(char *arg, const struct fetch_type *type,
ret = parse_probe_vars(arg, type, pcode, end, ctx);
break;
+#ifdef CONFIG_HAVE_FUNCTION_ARG_ACCESS_API
case '%': /* named register */
- if (ctx->flags & (TPARG_FL_TEVENT | TPARG_FL_FPROBE)) {
+ if (ctx->flags & (TPARG_FL_TEVENT | TPARG_FL_FPROBE | TPARG_FL_WPROBE)) {
/* eprobe and fprobe do not handle registers */
trace_probe_log_err(ctx->offset, BAD_VAR);
break;
@@ -1505,6 +1524,7 @@ parse_probe_arg(char *arg, const struct fetch_type *type,
} else
trace_probe_log_err(ctx->offset, BAD_REG_NAME);
break;
+#endif
case '@': /* memory, file-offset or symbol */
if (isdigit(arg[1])) {
diff --git a/kernel/trace/trace_probe.h b/kernel/trace/trace_probe.h
index e6268a8dc378..64c5fe9bdfc9 100644
--- a/kernel/trace/trace_probe.h
+++ b/kernel/trace/trace_probe.h
@@ -90,6 +90,7 @@ typedef int (*print_type_func_t)(struct trace_seq *, void *, void *);
FETCH_OP(STACK, param), /* Stack: .param = index */ \
FETCH_OP(STACKP, none), /* Stack pointer */ \
FETCH_OP(RETVAL, none), /* Return value */ \
+ FETCH_OP(BADDR, none), /* Break address */ \
FETCH_OP(IMM, imm), /* Immediate: .immediate */ \
FETCH_OP(COMM, none), /* Current comm */ \
FETCH_OP(CURRENT, none), /* Current task_struct address */\
@@ -419,6 +420,7 @@ static inline int traceprobe_get_entry_data_size(struct trace_probe *tp)
#define TPARG_FL_USER BIT(4)
#define TPARG_FL_FPROBE BIT(5)
#define TPARG_FL_TPOINT BIT(6)
+#define TPARG_FL_WPROBE BIT(7)
#define TPARG_FL_LOC_MASK GENMASK(4, 0)
static inline bool tparg_is_function_entry(unsigned int flags)
@@ -600,7 +602,11 @@ extern int traceprobe_define_arg_fields(struct trace_event_call *event_call,
C(TYPECAST_SYM_OFFSET, "@SYM+/-OFFSET with typecast needs parentheses"), \
C(TYPECAST_NOT_ALIGNED, "Typecast field option is not byte-aligned"), \
C(TYPECAST_BAD_ARROW, "Typecast field option does not support -> operator"), \
- C(NOSUP_PERCPU, "Per-cpu variable access is only for kernel probes"),
+ C(NOSUP_PERCPU, "Per-cpu variable access is only for kernel probes"), \
+ C(BAD_ACCESS_FMT, "Access memory address requires @"), \
+ C(BAD_ACCESS_TYPE, "Bad memory access type"), \
+ C(BAD_ACCESS_LEN, "This memory access length is not supported"), \
+ C(BAD_ACCESS_ADDR, "Invalid access memory address"),
#undef C
#define C(a, b) TP_ERR_##a
diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
new file mode 100644
index 000000000000..b52f3eac719f
--- /dev/null
+++ b/kernel/trace/trace_wprobe.c
@@ -0,0 +1,692 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Hardware-breakpoint-based tracing events
+ *
+ * Copyright (C) 2023, Masami Hiramatsu <mhiramat@kernel.org>
+ */
+#define pr_fmt(fmt) "trace_wprobe: " fmt
+
+#include <linux/hw_breakpoint.h>
+#include <linux/kallsyms.h>
+#include <linux/list.h>
+#include <linux/module.h>
+#include <linux/mutex.h>
+#include <linux/perf_event.h>
+#include <linux/rculist.h>
+#include <linux/security.h>
+#include <linux/tracepoint.h>
+#include <linux/uaccess.h>
+
+#include <asm/ptrace.h>
+
+#include "trace_dynevent.h"
+#include "trace_probe.h"
+#include "trace_probe_kernel.h"
+#include "trace_probe_tmpl.h"
+
+#define WPROBE_EVENT_SYSTEM "wprobes"
+
+static int trace_wprobe_create(const char *raw_command);
+static int trace_wprobe_show(struct seq_file *m, struct dyn_event *ev);
+static int trace_wprobe_release(struct dyn_event *ev);
+static bool trace_wprobe_is_busy(struct dyn_event *ev);
+static bool trace_wprobe_match(const char *system, const char *event,
+ int argc, const char **argv, struct dyn_event *ev);
+
+static struct dyn_event_operations trace_wprobe_ops = {
+ .create = trace_wprobe_create,
+ .show = trace_wprobe_show,
+ .is_busy = trace_wprobe_is_busy,
+ .free = trace_wprobe_release,
+ .match = trace_wprobe_match,
+};
+
+struct trace_wprobe {
+ struct dyn_event devent;
+ struct perf_event * __percpu *bp_event;
+ unsigned long addr;
+ int len;
+ int type;
+ const char *symbol;
+ struct trace_probe tp;
+};
+
+static bool is_trace_wprobe(struct dyn_event *ev)
+{
+ return ev->ops == &trace_wprobe_ops;
+}
+
+static struct trace_wprobe *to_trace_wprobe(struct dyn_event *ev)
+{
+ return container_of(ev, struct trace_wprobe, devent);
+}
+
+#define for_each_trace_wprobe(pos, dpos) \
+ for_each_dyn_event(dpos) \
+ if (is_trace_wprobe(dpos) && (pos = to_trace_wprobe(dpos)))
+
+static bool trace_wprobe_is_busy(struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+
+ return trace_probe_is_enabled(&tw->tp);
+}
+
+static bool trace_wprobe_match(const char *system, const char *event,
+ int argc, const char **argv, struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+
+ if (event[0] != '\0' && strcmp(trace_probe_name(&tw->tp), event))
+ return false;
+
+ if (system && strcmp(trace_probe_group_name(&tw->tp), system))
+ return false;
+
+ /* TODO: match arguments */
+ return true;
+}
+
+/*
+ * Note that we don't verify the fetch_insn code, since it does not come
+ * from user space.
+ */
+static int
+process_fetch_insn(struct fetch_insn *code, void *rec, void *edata,
+ void *dest, void *base)
+{
+ void *baddr = rec;
+ unsigned long val;
+ int ret;
+
+retry:
+ /* 1st stage: get value from context */
+ switch (code->op) {
+ case FETCH_OP_BADDR:
+ val = (unsigned long)baddr;
+ break;
+ case FETCH_NOP_SYMBOL: /* Ignore a place holder */
+ code++;
+ goto retry;
+ default:
+ ret = process_common_fetch_insn(code, &val);
+ if (ret < 0)
+ return ret;
+ }
+ code++;
+
+ return process_fetch_insn_bottom(code, val, dest, base);
+}
+NOKPROBE_SYMBOL(process_fetch_insn)
+
+static void wprobe_trace_handler(struct trace_wprobe *tw,
+ unsigned long addr,
+ struct pt_regs *regs,
+ struct trace_event_file *trace_file)
+{
+ struct wprobe_trace_entry_head *entry;
+ struct trace_event_call *call = trace_probe_event_call(&tw->tp);
+ struct trace_event_buffer fbuffer;
+ int dsize;
+
+ if (WARN_ON_ONCE(call != trace_file->event_call))
+ return;
+
+ if (trace_trigger_soft_disabled(trace_file))
+ return;
+
+ if (tw->addr != addr)
+ return;
+
+ dsize = __get_data_size(&tw->tp, (void *)addr, NULL);
+
+ entry = trace_event_buffer_reserve(&fbuffer, trace_file,
+ sizeof(*entry) + tw->tp.size + dsize);
+ if (!entry)
+ return;
+
+ entry->ip = instruction_pointer(regs);
+ store_trace_args(&entry[1], &tw->tp, (void *)addr, NULL, sizeof(*entry), dsize);
+
+ fbuffer.regs = regs;
+ trace_event_buffer_commit(&fbuffer);
+}
+
+static void wprobe_perf_handler(struct perf_event *bp,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct trace_wprobe *tw = bp->overflow_handler_context;
+ struct event_file_link *link;
+ unsigned long addr = bp->attr.bp_addr;
+
+ trace_probe_for_each_link_rcu(link, &tw->tp)
+ wprobe_trace_handler(tw, addr, regs, link->file);
+}
+
+static int __register_trace_wprobe(struct trace_wprobe *tw)
+{
+ struct perf_event_attr attr;
+
+ if (tw->bp_event)
+ return -EINVAL;
+
+ hw_breakpoint_init(&attr);
+ attr.bp_addr = tw->addr;
+ attr.bp_len = tw->len;
+ attr.bp_type = tw->type;
+
+ tw->bp_event = register_wide_hw_breakpoint(&attr, wprobe_perf_handler, tw);
+ if (IS_ERR_PCPU(tw->bp_event)) {
+ int ret = PTR_ERR_PCPU(tw->bp_event);
+
+ tw->bp_event = NULL;
+ return ret;
+ }
+
+ return 0;
+}
+
+static void __unregister_trace_wprobe(struct trace_wprobe *tw)
+{
+ if (tw->bp_event) {
+ unregister_wide_hw_breakpoint(tw->bp_event);
+ tw->bp_event = NULL;
+ }
+}
+
+static void free_trace_wprobe(struct trace_wprobe *tw)
+{
+ if (tw) {
+ trace_probe_cleanup(&tw->tp);
+ kfree(tw->symbol);
+ kfree(tw);
+ }
+}
+DEFINE_FREE(free_trace_wprobe, struct trace_wprobe *,
+ if (!IS_ERR_OR_NULL(_T))
+ free_trace_wprobe(_T))
+
+
+static struct trace_wprobe *alloc_trace_wprobe(const char *group,
+ const char *event,
+ const char *symbol,
+ unsigned long addr,
+ int len, int type, int nargs)
+{
+ struct trace_wprobe *tw __free(free_trace_wprobe) = NULL;
+ int ret;
+
+ tw = kzalloc(struct_size(tw, tp.args, nargs), GFP_KERNEL);
+ if (!tw)
+ return ERR_PTR(-ENOMEM);
+
+ if (symbol) {
+ tw->symbol = kstrdup(symbol, GFP_KERNEL);
+ if (!tw->symbol)
+ return ERR_PTR(-ENOMEM);
+ }
+ tw->addr = addr;
+ tw->len = len;
+ tw->type = type;
+
+ ret = trace_probe_init(&tw->tp, event, group, false, nargs);
+ if (ret < 0)
+ return ERR_PTR(ret);
+
+ dyn_event_init(&tw->devent, &trace_wprobe_ops);
+ return_ptr(tw);
+}
+
+static struct trace_wprobe *find_trace_wprobe(const char *event,
+ const char *group)
+{
+ struct dyn_event *pos;
+ struct trace_wprobe *tw;
+
+ for_each_trace_wprobe(tw, pos)
+ if (strcmp(trace_probe_name(&tw->tp), event) == 0 &&
+ strcmp(trace_probe_group_name(&tw->tp), group) == 0)
+ return tw;
+ return NULL;
+}
+
+static enum print_line_t
+print_wprobe_event(struct trace_iterator *iter, int flags,
+ struct trace_event *event)
+{
+ struct wprobe_trace_entry_head *field;
+ struct trace_seq *s = &iter->seq;
+ struct trace_probe *tp;
+
+ field = (struct wprobe_trace_entry_head *)iter->ent;
+ tp = trace_probe_primary_from_call(
+ container_of(event, struct trace_event_call, event));
+ if (WARN_ON_ONCE(!tp))
+ goto out;
+
+ trace_seq_printf(s, "%s: (", trace_probe_name(tp));
+
+ if (!seq_print_ip_sym_offset(s, field->ip, flags))
+ goto out;
+
+ trace_seq_putc(s, ')');
+
+ if (trace_probe_print_args(s, tp->args, tp->nr_args,
+ (u8 *)&field[1], field) < 0)
+ goto out;
+
+ trace_seq_putc(s, '\n');
+out:
+ return trace_handle_return(s);
+}
+
+static int wprobe_event_define_fields(struct trace_event_call *event_call)
+{
+ int ret;
+ struct wprobe_trace_entry_head field;
+ struct trace_probe *tp;
+
+ tp = trace_probe_primary_from_call(event_call);
+ if (WARN_ON_ONCE(!tp))
+ return -ENOENT;
+
+ DEFINE_FIELD(unsigned long, ip, FIELD_STRING_IP, 0);
+
+ return traceprobe_define_arg_fields(event_call, sizeof(field), tp);
+}
+
+static struct trace_event_functions wprobe_funcs = {
+ .trace = print_wprobe_event
+};
+
+static struct trace_event_fields wprobe_fields_array[] = {
+ { .type = TRACE_FUNCTION_TYPE,
+ .define_fields = wprobe_event_define_fields },
+ {}
+};
+
+static int wprobe_register(struct trace_event_call *event,
+ enum trace_reg type, void *data);
+
+static inline void init_trace_event_call(struct trace_wprobe *tw)
+{
+ struct trace_event_call *call = trace_probe_event_call(&tw->tp);
+
+ call->event.funcs = &wprobe_funcs;
+ call->class->fields_array = wprobe_fields_array;
+ call->flags = TRACE_EVENT_FL_WPROBE;
+ call->class->reg = wprobe_register;
+}
+
+static int register_wprobe_event(struct trace_wprobe *tw)
+{
+ init_trace_event_call(tw);
+ return trace_probe_register_event_call(&tw->tp);
+}
+
+static int register_trace_wprobe_event(struct trace_wprobe *tw)
+{
+ struct trace_wprobe *old_tb;
+ int ret;
+
+ guard(mutex)(&event_mutex);
+
+ old_tb = find_trace_wprobe(trace_probe_name(&tw->tp),
+ trace_probe_group_name(&tw->tp));
+ if (old_tb)
+ return -EBUSY;
+
+ ret = register_wprobe_event(tw);
+ if (ret)
+ return ret;
+
+ dyn_event_add(&tw->devent, trace_probe_event_call(&tw->tp));
+ return 0;
+}
+static int unregister_wprobe_event(struct trace_wprobe *tw)
+{
+ return trace_probe_unregister_event_call(&tw->tp);
+}
+
+static int unregister_trace_wprobe(struct trace_wprobe *tw)
+{
+ if (trace_probe_has_sibling(&tw->tp))
+ goto unreg;
+
+ if (trace_probe_is_enabled(&tw->tp))
+ return -EBUSY;
+
+ if (trace_event_dyn_busy(trace_probe_event_call(&tw->tp)))
+ return -EBUSY;
+
+ if (unregister_wprobe_event(tw))
+ return -EBUSY;
+
+unreg:
+ __unregister_trace_wprobe(tw);
+ dyn_event_remove(&tw->devent);
+ trace_probe_unlink(&tw->tp);
+
+ return 0;
+}
+
+static int enable_trace_wprobe(struct trace_event_call *call,
+ struct trace_event_file *file)
+{
+ struct trace_probe *tp;
+ struct trace_wprobe *tw;
+ bool enabled;
+ int ret = 0;
+
+ tp = trace_probe_primary_from_call(call);
+ if (WARN_ON_ONCE(!tp))
+ return -ENODEV;
+ enabled = trace_probe_is_enabled(tp);
+
+ if (file) {
+ ret = trace_probe_add_file(tp, file);
+ if (ret)
+ return ret;
+ } else {
+ trace_probe_set_flag(tp, TP_FLAG_PROFILE);
+ }
+
+ if (!enabled) {
+ list_for_each_entry(tw, trace_probe_probe_list(tp), tp.list) {
+ ret = __register_trace_wprobe(tw);
+ if (ret < 0) {
+ /* TODO: rollback */
+ return ret;
+ }
+ }
+ }
+
+ return 0;
+}
+
+static int disable_trace_wprobe(struct trace_event_call *call,
+ struct trace_event_file *file)
+{
+ struct trace_wprobe *tw;
+ struct trace_probe *tp;
+
+ tp = trace_probe_primary_from_call(call);
+ if (WARN_ON_ONCE(!tp))
+ return -ENODEV;
+
+ if (file) {
+ if (!trace_probe_get_file_link(tp, file))
+ return -ENOENT;
+ if (!trace_probe_has_single_file(tp))
+ goto out;
+ trace_probe_clear_flag(tp, TP_FLAG_TRACE);
+ } else {
+ trace_probe_clear_flag(tp, TP_FLAG_PROFILE);
+ }
+
+ if (!trace_probe_is_enabled(tp)) {
+ list_for_each_entry(tw, trace_probe_probe_list(tp), tp.list) {
+ __unregister_trace_wprobe(tw);
+ }
+ }
+
+out:
+ if (file)
+ trace_probe_remove_file(tp, file);
+
+ return 0;
+}
+
+static int wprobe_register(struct trace_event_call *event,
+ enum trace_reg type, void *data)
+{
+ struct trace_event_file *file = data;
+
+ switch (type) {
+ case TRACE_REG_REGISTER:
+ return enable_trace_wprobe(event, file);
+ case TRACE_REG_UNREGISTER:
+ return disable_trace_wprobe(event, file);
+
+#ifdef CONFIG_PERF_EVENTS
+ case TRACE_REG_PERF_REGISTER:
+ return enable_trace_wprobe(event, NULL);
+ case TRACE_REG_PERF_UNREGISTER:
+ return disable_trace_wprobe(event, NULL);
+ case TRACE_REG_PERF_OPEN:
+ case TRACE_REG_PERF_CLOSE:
+ case TRACE_REG_PERF_ADD:
+ case TRACE_REG_PERF_DEL:
+ return 0;
+#endif
+ }
+ return 0;
+}
+
+static int parse_address_spec(const char *spec, unsigned long *addr, int *type,
+ int *len, char **symbol)
+{
+ char *_spec __free(kfree) = NULL;
+ int _len = HW_BREAKPOINT_LEN_4;
+ int _type = HW_BREAKPOINT_RW;
+ unsigned long _addr = 0;
+ char *at, *col;
+
+ _spec = kstrdup(spec, GFP_KERNEL);
+ if (!_spec)
+ return -ENOMEM;
+
+ at = strchr(_spec, '@');
+ col = strchr(_spec, ':');
+
+ if (!at) {
+ trace_probe_log_err(0, BAD_ACCESS_FMT);
+ return -EINVAL;
+ }
+
+ if (at != _spec) {
+ *at = '\0';
+
+ if (strcmp(_spec, "r") == 0)
+ _type = HW_BREAKPOINT_R;
+ else if (strcmp(_spec, "w") == 0)
+ _type = HW_BREAKPOINT_W;
+ else if (strcmp(_spec, "rw") == 0)
+ _type = HW_BREAKPOINT_RW;
+ else {
+ trace_probe_log_err(0, BAD_ACCESS_TYPE);
+ return -EINVAL;
+ }
+ }
+
+ if (col) {
+ *col = '\0';
+ if (kstrtoint(col + 1, 0, &_len)) {
+ trace_probe_log_err(col + 1 - _spec, BAD_ACCESS_LEN);
+ return -EINVAL;
+ }
+
+ switch (_len) {
+ case 1:
+ _len = HW_BREAKPOINT_LEN_1;
+ break;
+ case 2:
+ _len = HW_BREAKPOINT_LEN_2;
+ break;
+ case 4:
+ _len = HW_BREAKPOINT_LEN_4;
+ break;
+ case 8:
+ _len = HW_BREAKPOINT_LEN_8;
+ break;
+ default:
+ trace_probe_log_err(col + 1 - _spec, BAD_ACCESS_LEN);
+ return -EINVAL;
+ }
+ }
+
+ if (kstrtoul(at + 1, 0, &_addr) != 0) {
+ char *off_str = strpbrk(at + 1, "+-");
+ int offset = 0;
+
+ if (off_str) {
+ if (kstrtoint(off_str, 0, &offset) != 0) {
+ trace_probe_log_err(off_str - _spec, BAD_PROBE_ADDR);
+ return -EINVAL;
+ }
+ *off_str = '\0';
+ }
+ _addr = kallsyms_lookup_name(at + 1);
+ if (!_addr) {
+ trace_probe_log_err(at + 1 - _spec, BAD_ACCESS_ADDR);
+ return -ENOENT;
+ }
+ _addr += offset;
+ *symbol = kstrdup(at + 1, GFP_KERNEL);
+ if (!*symbol)
+ return -ENOMEM;
+ }
+
+ *addr = _addr;
+ *type = _type;
+ *len = _len;
+ return 0;
+}
+
+static int __trace_wprobe_create(int argc, const char *argv[])
+{
+ /*
+ * Argument syntax:
+ * b[:[GRP/][EVENT]] SPEC
+ *
+ * SPEC:
+ * [r|w|rw]@[ADDR|SYMBOL[+OFFS]][:LEN]
+ */
+ struct traceprobe_parse_context *ctx __free(traceprobe_parse_context) = NULL;
+ struct trace_wprobe *tw __free(free_trace_wprobe) = NULL;
+ const char *event = NULL, *group = WPROBE_EVENT_SYSTEM;
+ const char *tplog __free(trace_probe_log_clear) = NULL;
+ char *symbol = NULL;
+ unsigned long addr;
+ int len, type, i;
+ int ret = 0;
+
+ if (argv[0][0] != 'w')
+ return -ECANCELED;
+
+ if (argc < 2)
+ return -EINVAL;
+
+ tplog = trace_probe_log_init("wprobe", argc, argv);
+
+ if (argv[0][1] != '\0') {
+ if (argv[0][1] != ':') {
+ trace_probe_log_set_index(0);
+ trace_probe_log_err(1, BAD_MAXACT_TYPE);
+ /* Invalid format */
+ return -EINVAL;
+ }
+ event = &argv[0][2];
+ }
+
+ trace_probe_log_set_index(1);
+ ret = parse_address_spec(argv[1], &addr, &type, &len, &symbol);
+ if (ret < 0)
+ return ret;
+
+ if (!event)
+ event = symbol ? symbol : "wprobe";
+
+ argc -= 2; argv += 2;
+ tw = alloc_trace_wprobe(group, event, symbol, addr, len, type, argc);
+ if (IS_ERR(tw))
+ return PTR_ERR(tw);
+
+ ctx = kzalloc_obj(*ctx);
+ if (!ctx)
+ return -ENOMEM;
+
+ ctx->flags = TPARG_FL_KERNEL | TPARG_FL_WPROBE;
+
+ /* parse arguments */
+ for (i = 0; i < argc; i++) {
+ trace_probe_log_set_index(i + 2);
+ ctx->offset = 0;
+ ret = traceprobe_parse_probe_arg(&tw->tp, i, argv[i], ctx);
+ if (ret)
+ return ret; /* This can be -ENOMEM */
+ }
+
+ ret = traceprobe_set_print_fmt(&tw->tp, PROBE_PRINT_NORMAL);
+ if (ret < 0)
+ return ret;
+
+ ret = register_trace_wprobe_event(tw);
+ if (!ret)
+ tw = NULL; /* To avoid free */
+
+ return ret;
+}
+
+static int trace_wprobe_create(const char *raw_command)
+{
+ return trace_probe_create(raw_command, __trace_wprobe_create);
+}
+
+static int trace_wprobe_release(struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+ int ret = unregister_trace_wprobe(tw);
+
+ if (!ret)
+ free_trace_wprobe(tw);
+ return ret;
+}
+
+static int trace_wprobe_show(struct seq_file *m, struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+ int i;
+
+ seq_printf(m, "w:%s/%s", trace_probe_group_name(&tw->tp),
+ trace_probe_name(&tw->tp));
+
+ char type_char;
+
+ if (tw->type == HW_BREAKPOINT_R)
+ type_char = 'r';
+ else if (tw->type == HW_BREAKPOINT_W)
+ type_char = 'w';
+ else
+ type_char = 'x'; /* Should be rw */
+
+ int len;
+
+ if (tw->len == HW_BREAKPOINT_LEN_1)
+ len = 1;
+ else if (tw->len == HW_BREAKPOINT_LEN_2)
+ len = 2;
+ else if (tw->len == HW_BREAKPOINT_LEN_4)
+ len = 4;
+ else
+ len = 8;
+
+ if (tw->symbol)
+ seq_printf(m, " %c@%s:%d", type_char, tw->symbol, len);
+ else
+ seq_printf(m, " %c@0x%lx:%d", type_char, tw->addr, len);
+
+ for (i = 0; i < tw->tp.nr_args; i++)
+ seq_printf(m, " %s=%s", tw->tp.args[i].name, tw->tp.args[i].comm);
+ seq_putc(m, '\n');
+
+ return 0;
+}
+
+static __init int init_wprobe_trace(void)
+{
+ return dyn_event_register(&trace_wprobe_ops);
+}
+fs_initcall(init_wprobe_trace);
+
^ permalink raw reply related
* [PATCH v7 00/10] tracing: wprobe: x86: Add wprobe for watchpoint
From: Masami Hiramatsu (Google) @ 2026-07-15 1:43 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
Hi,
Here is the 7th version of the series for adding new wprobe (watch probe)
which provides memory access tracing event. Moreover, this can be used
via event trigger. Thus it can trace memory access on a dynamically
allocated objects too.
The previous version is here:
https://lore.kernel.org/all/176960933881.182525.11984731584313026309.stgit@devnote2/
This version is rebased on top of probes/for-next branch, fold IS_ERR_PCPU
fix and fix some checkpatch.pl related issue (not all, some of them seems
wrong warnings).
Other updates:
- Fix a race on wprobe handler using bp->attr.bp_addr instead of tw->addr.
- Fix modify_wide_hw_breakpoint_local() to update bp->attr.bp_addr.
- Return -EOPNOTSUPP instead of -ENOSYS. (that is not a syscall function)
- Update test script to use dentry_kill instead of __dentry_kill.
To support arm64, we need to avoid major pagefault on single-stepping
issue. I think we can solve it with checking whether the address is the
kernel (which must not cause a major page fault), or not.
Usage
-----
The basic usage of this wprobe is similar to other probes;
w:[GRP/][EVENT] [r|w|rw]@<ADDRESS|SYMBOL[+OFFS]> [FETCHARGS]
This defines a new wprobe event. For example, to trace jiffies update,
you can do;
echo 'w:my_jiffies w@jiffies:8 value=+0($addr)' >> dynamic_events
echo 1 > events/wprobes/my_jiffies/enable
Moreover, this can be combined with event trigger to trace the memory
accecss on slab objects. The trigger syntax is;
set_wprobe:WPROBE_EVENT:FIELD[+OFFSET] [if FILTER]
clear_wprobe:WPROBE_EVENT[:FIELD[+OFFSET]] [if FILTER]
set_wprobe sets WPROBE_EVENT's watch address on FIELD[+OFFSET].
clear_wprobe clears WPROBE_EVENT's watch address if it is set to
FIELD[+OFFSET]. If FIELD is omitted, forcibly clear the watch address
when trigger event is hit.
For example, trace the first 8 byte of the dentry data structure passed
to do_truncate() until it is deleted by __dentry_kill().
(Note: all tracefs setup uses '>>' so that it does not kick do_truncate())
# echo 'w:watch rw@0:8 address=$addr value=+0($addr)' > dynamic_events
# echo 'f:truncate do_truncate dentry=$arg2' >> dynamic_events
# echo 'set_wprobe:watch:dentry' >> events/fprobes/truncate/trigger
# echo 'f:dentry_kill __dentry_kill dentry=$arg1' >> dynamic_events
# echo 'clear_wprobe:watch:dentry' >> events/fprobes/dentry_kill/trigger
# echo 1 >> events/fprobes/truncate/enable
# echo 1 >> events/fprobes/dentry_kill/enable
# echo aaa > /tmp/hoge
# echo bbb > /tmp/hoge
# echo ccc > /tmp/hoge
# rm /tmp/hoge
Then, the trace data will show;
# tracer: nop
#
# entries-in-buffer/entries-written: 16/16 #P:8
#
# _-----=> irqs-off/BH-disabled
# / _----=> need-resched
# | / _---=> hardirq/softirq
# || / _--=> preempt-depth
# ||| / _-=> migrate-disable
# |||| / delay
# TASK-PID CPU# ||||| TIMESTAMP FUNCTION
# | | | ||||| | |
sh-113 [004] ..... 6.467444: truncate: (do_truncate+0x4/0x120) dentry=0xffff8880044f0fd8
sh-113 [004] ..Zff 6.468534: watch: (lookup_fast+0xaa/0x150) address=0xffff8880044f0fd8 value=0x200080
sh-113 [004] ..Zff 6.468542: watch: (step_into+0x82/0x360) address=0xffff8880044f0fd8 value=0x200080
sh-113 [004] ..Zff 6.468547: watch: (step_into+0x9f/0x360) address=0xffff8880044f0fd8 value=0x200080
sh-113 [004] ..Zff 6.468553: watch: (path_openat+0xb3a/0xe70) address=0xffff8880044f0fd8 value=0x200080
sh-113 [004] ..Zff 6.468557: watch: (path_openat+0xb9a/0xe70) address=0xffff8880044f0fd8 value=0x200080
sh-113 [004] ..... 6.468563: truncate: (do_truncate+0x4/0x120) dentry=0xffff8880044f0fd8
sh-113 [004] ...1. 6.469826: dentry_kill: (__dentry_kill+0x0/0x220) dentry=0xffff8880044f0ea0
sh-113 [004] ...1. 6.469859: dentry_kill: (__dentry_kill+0x0/0x220) dentry=0xffff8880044f0d68
rm-118 [001] ..Zff 6.472360: watch: (lookup_fast+0xaa/0x150) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472366: watch: (step_into+0x82/0x360) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472370: watch: (step_into+0x9f/0x360) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472386: watch: (lookup_fast+0xaa/0x150) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472390: watch: (step_into+0x82/0x360) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472394: watch: (step_into+0x9f/0x360) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472415: watch: (lookup_one_qstr_excl+0x2c/0x150) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472419: watch: (lookup_one_qstr_excl+0xd5/0x150) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472424: watch: (may_delete+0x18/0x200) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472428: watch: (may_delete+0x194/0x200) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] ..Zff 6.472446: watch: (vfs_unlink+0x63/0x1c0) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] d.Z.. 6.472528: watch: (dont_mount+0x19/0x30) address=0xffff8880044f0fd8 value=0x200180
rm-118 [001] ..Zff 6.472533: watch: (vfs_unlink+0x11a/0x1c0) address=0xffff8880044f0fd8 value=0x200180
rm-118 [001] ..Zff 6.472538: watch: (vfs_unlink+0x12e/0x1c0) address=0xffff8880044f0fd8 value=0x200180
rm-118 [001] d.Z1. 6.472543: watch: (d_delete+0x61/0xa0) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] d.Z1. 6.472547: watch: (dentry_unlink_inode+0x14/0x110) address=0xffff8880044f0fd8 value=0x200080
rm-118 [001] d.Z1. 6.472551: watch: (dentry_unlink_inode+0x1e/0x110) address=0xffff8880044f0fd8 value=0x80
rm-118 [001] d.Z.. 6.472563: watch: (fast_dput+0x8d/0x120) address=0xffff8880044f0fd8 value=0x80
rm-118 [001] ...1. 6.472567: dentry_kill: (__dentry_kill+0x0/0x220) dentry=0xffff8880044f0fd8
sh-113 [004] ...2. 6.473049: dentry_kill: (__dentry_kill+0x0/0x220) dentry=0xffff888006e383a8
Thank you,
---
Jinchao Wang (2):
x86/hw_breakpoint: Unify breakpoint install/uninstall
x86/hw_breakpoint: Add arch_reinstall_hw_breakpoint
Masami Hiramatsu (Google) (8):
tracing: wprobe: Add watchpoint probe event based on hardware breakpoint
x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires
selftests: tracing: Add a basic testcase for wprobe
selftests: tracing: Add syntax testcase for wprobe
tracing: wprobe: Use a new seq_print_ip_sym_offset() wrapper
HWBP: Add modify_wide_hw_breakpoint_local() API
tracing: wprobe: Add wprobe event trigger
selftests: ftrace: Add wprobe trigger testcase
Documentation/trace/index.rst | 1
Documentation/trace/wprobetrace.rst | 158 +++
arch/Kconfig | 20
arch/x86/Kconfig | 2
arch/x86/include/asm/hw_breakpoint.h | 8
arch/x86/kernel/hw_breakpoint.c | 148 ++-
include/linux/hw_breakpoint.h | 6
include/linux/trace_events.h | 3
kernel/events/hw_breakpoint.c | 39 +
kernel/trace/Kconfig | 24
kernel/trace/Makefile | 1
kernel/trace/trace.c | 9
kernel/trace/trace.h | 5
kernel/trace/trace_probe.c | 22
kernel/trace/trace_probe.h | 8
kernel/trace/trace_wprobe.c | 1108 ++++++++++++++++++++
tools/testing/selftests/ftrace/config | 2
.../ftrace/test.d/dynevent/add_remove_wprobe.tc | 68 +
.../test.d/dynevent/wprobes_syntax_errors.tc | 20
.../ftrace/test.d/trigger/trigger-wprobe.tc | 48 +
20 files changed, 1635 insertions(+), 65 deletions(-)
create mode 100644 Documentation/trace/wprobetrace.rst
create mode 100644 kernel/trace/trace_wprobe.c
create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc
--
Masami Hiramatsu (Google) <mhiramat@kernel.org>
^ permalink raw reply
* Re: [PATCH v2 3/4] tracing/remotes: Add REMOTE_EVENT_CUSTOM_PRINTK() helper
From: Steven Rostedt @ 2026-07-14 22:40 UTC (permalink / raw)
To: Vincent Donnefort
Cc: maz, oliver.upton, joey.gouly, suzuki.poulose, yuzenghui,
catalin.marinas, will, linux-arm-kernel, kvmarm, kernel-team,
qperret, tabba, linux-trace-kernel
In-Reply-To: <20260708075435.47419-4-vdonnefort@google.com>
On Wed, 8 Jul 2026 08:54:34 +0100
Vincent Donnefort <vdonnefort@google.com> wrote:
> The current REMOTE_EVENT() takes as a __printk argument a string format
> and a list of arguments, such as RE_STRUCT("foo=%d bar=%d", foo, bar).
> Add a REMOTE_EVENT_CUSTOM_PRINTK() where the __printk argument can be a
> function. This intends to support the creation of a "printk" event for
> the arm64 nVHE/pKVM hypervisor with a dynamic prototype and by extension
> a dynamic print format.
>
> Reviewed-by: Fuad Tabba <tabba@google.com>
> Tested-by: Fuad Tabba <tabba@google.com>
> Signed-off-by: Vincent Donnefort <vdonnefort@google.com>
I'm assuming this goes in via the arm64 or KVM trees?
Acked-by: Steven Rostedt <rostedt@goodmis.org>
-- Steve
^ permalink raw reply
* Re: [PATCH] tracing: ring-buffer: allowlist clang-generated symbols
From: Steven Rostedt @ 2026-07-14 21:56 UTC (permalink / raw)
To: Vincent Donnefort
Cc: Arnd Bergmann, Masami Hiramatsu, Nathan Chancellor, Arnd Bergmann,
Mathieu Desnoyers, Nick Desaulniers, Bill Wendling, Justin Stitt,
Marc Zyngier, Thomas Weißschuh, Paolo Bonzini, linux-kernel,
linux-trace-kernel, llvm
In-Reply-To: <ajKgjNuHUlteSziZ@google.com>
On Wed, 17 Jun 2026 14:26:36 +0100
Vincent Donnefort <vdonnefort@google.com> wrote:
> On Tue, Jun 16, 2026 at 06:42:03PM +0200, Arnd Bergmann wrote:
> > From: Arnd Bergmann <arnd@arndb.de>
> >
> > In randconfig build testing using clang-22, I came across two
> > sets of extra symbols in the ring buffer code that may get
> > inserted by the compiler:
> >
> > Unexpected symbols in kernel/trace/simple_ring_buffer.o:
> > U memset
> >
> > Unexpected symbols in kernel/trace/simple_ring_buffer.o:
> > U llvm_gcda_emit_arcs
> > U llvm_gcda_emit_function
> > U llvm_gcda_end_file
> > U llvm_gcda_start_file
> > U llvm_gcda_summary_info
> > U llvm_gcov_init
> >
> > Add all of these to the allowlist.
> >
> > Signed-off-by: Arnd Bergmann <arnd@arndb.de>
> > ---
> > kernel/trace/Makefile | 1 +
> > 1 file changed, 1 insertion(+)
> >
> > diff --git a/kernel/trace/Makefile b/kernel/trace/Makefile
> > index f934ff586bd4..aa8564fb8ff4 100644
> > --- a/kernel/trace/Makefile
> > +++ b/kernel/trace/Makefile
> > @@ -146,6 +146,7 @@ KASAN_SANITIZE_undefsyms_base.o := y
>
> Would "GCOV_PROFILE_undefsyms_base.o := y" work?
Arnd?
-- Steve
>
> >
> > UNDEFINED_ALLOWLIST = __asan __gcov __kasan __kcsan __hwasan __sancov __sanitizer __tsan __ubsan __msan \
> > __aeabi_unwind_cpp __s390_indirect_jump __x86_indirect_thunk simple_ring_buffer \
> > + memset llvm_gcda llvm_gcov \
> > $(shell $(NM) -u $(obj)/undefsyms_base.o 2>/dev/null | awk '{print $$2}')
> >
> > quiet_cmd_check_undefined = NM $<
> > --
> > 2.39.5
> >
^ permalink raw reply
* Re: [RFC PATCH v4 2/3] trace: integrate stackmap into ftrace stack recording path
From: Steven Rostedt @ 2026-07-14 21:53 UTC (permalink / raw)
To: Li Pengfei
Cc: Masami Hiramatsu, Mathieu Desnoyers, Mark Rutland,
Jonathan Corbet, Shuah Khan, linux-kernel, linux-trace-kernel,
linux-doc, linux-kselftest, lipengfei28, zhangbo56
In-Reply-To: <20260616064119.438063-3-lipengfei28@xiaomi.com>
On Tue, 16 Jun 2026 14:41:18 +0800
Li Pengfei <ljdlns1987@gmail.com> wrote:
> int set_tracer_flag(struct trace_array *tr, u64 mask, int enabled)
> {
> switch (mask) {
> @@ -3993,6 +4091,33 @@ int set_tracer_flag(struct trace_array *tr, u64 mask, int enabled)
> if (!!(tr->trace_flags & mask) == !!enabled)
> return 0;
>
> +#ifdef CONFIG_FTRACE_STACKMAP
> + /*
> + * STACKMAP is intentionally global-instance-only: the dedup map,
> + * its tracefs files (stack_map / stack_map_stat / stack_map_bin)
> + * and the lifetime/reset semantics are tied to the global trace
> + * array. options/stackmap is hidden on secondary instances via
> + * TOP_LEVEL_TRACE_FLAGS, but writes still reach set_tracer_flag()
> + * through the aggregate trace_options file. Reject the enable on
> + * a secondary instance so it cannot be silently accepted and then
> + * become a no-op in the hot path (where tr->stackmap is NULL and
> + * the code falls back to a full stack trace).
> + *
> + * On the global instance, allow the enable while init is still
> + * pending (boot-time trace_options=stackmap is applied before the
> + * tracefs init work creates the map; the hot path falls back
> + * until the map is published). Only reject once init has
> + * permanently failed, so options/stackmap never reports an
> + * enabled no-op. READ_ONCE() suffices: this only inspects the
> + * init state, it does not dereference the map (the hot path uses
> + * smp_load_acquire(&tr->stackmap) for that).
> + */
> + if (mask == TRACE_ITER(STACKMAP) && enabled &&
> + (tr != &global_trace ||
> + READ_ONCE(stackmap_init_state) == STACKMAP_INIT_FAILED))
> + return -EINVAL;
> +#endif
> +
> /* Give the tracer a chance to approve the change */
> if (tr->current_trace->flag_changed)
> if (tr->current_trace->flag_changed(tr, mask, !!enabled))
> @@ -9222,6 +9347,91 @@ static __init void tracer_init_tracefs_work_func(struct work_struct *work)
> NULL, &tracing_dyn_info_fops);
> #endif
>
> +#ifdef CONFIG_FTRACE_STACKMAP
> + {
> + struct ftrace_stackmap *smap;
> + struct dentry *map_file;
> +
> + smap = ftrace_stackmap_create(&global_trace);
> + if (!IS_ERR(smap)) {
> + /*
> + * Failure-atomic init: stack_map is the single
> + * required tracefs file (it doubles as the reset
> + * interface and the human-readable resolver). If
> + * we cannot create it, the hot path must not be
> + * able to emit <stack_id N> events that no one can
> + * resolve or clear, so refuse to publish the map
> + * and tear it down.
> + *
> + * Create stack_map BEFORE smp_store_release() so an
> + * observed non-NULL global_trace.stackmap implies
> + * its resolver/reset file exists.
> + */
> + map_file = trace_create_file("stack_map",
> + TRACE_MODE_WRITE, NULL,
> + smap,
> + &ftrace_stackmap_fops);
> + if (!map_file) {
> + pr_warn("ftrace stackmap init: stack_map create failed, dedup disabled\n");
> + ftrace_stackmap_destroy(smap);
> + /*
> + * Permanent failure. Record it and clear a
> + * STACKMAP flag that a boot-time
> + * trace_options=stackmap may have set, so
> + * options/stackmap does not report an
> + * enabled no-op and later userspace enables
> + * return -EINVAL.
> + */
> + WRITE_ONCE(stackmap_init_state,
> + STACKMAP_INIT_FAILED);
> + global_trace.trace_flags &=
> + ~TRACE_ITER(STACKMAP);
80 columns is no longer a hard requirement. 100 is more the default, so the
above should be:
WRITE_ONCE(stackmap_init_state, STACKMAP_INIT_FAILED);
global_trace.trace_flags &= ~TRACE_ITER(STACKMAP);
> + } else {
> + /*
> + * smp_store_release pairs with the
> + * smp_load_acquire() in
> + * __ftrace_trace_stack(). Publishing only
> + * after the required file exists keeps
> + * "smap visible" => "resolver/reset
> + * available".
> + */
> + smp_store_release(&global_trace.stackmap,
> + smap);
> + WRITE_ONCE(stackmap_init_state,
> + STACKMAP_INIT_DONE);
Same with the above two.
> + /*
> + * stat and bin are auxiliary observability
> + * surfaces. If they fail to be created we
> + * keep dedup enabled (the kernel side still
> + * works, and stack_map alone is enough to
> + * resolve and reset); trace_create_file()
> + * already pr_warn()s on failure.
> + */
> + trace_create_file("stack_map_stat",
> + TRACE_MODE_READ, NULL,
> + smap,
> + &ftrace_stackmap_stat_fops);
> + trace_create_file("stack_map_bin",
> + TRACE_MODE_READ, NULL,
> + smap,
> + &ftrace_stackmap_bin_fops);
> + }
-- Steve
^ permalink raw reply
* Re: [RFC PATCH 08/13] mm/kwatch: add hardware breakpoint backend
From: Steven Rostedt @ 2026-07-14 21:14 UTC (permalink / raw)
To: Jinchao Wang
Cc: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Masami Hiramatsu,
Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc
In-Reply-To: <20260714183206.12688-1-wangjinchao600@gmail.com>
On Wed, 15 Jul 2026 02:32:06 +0800
Jinchao Wang <wangjinchao600@gmail.com> wrote:
> --- /dev/null
> +++ b/include/trace/events/kwatch.h
> @@ -0,0 +1,57 @@
> +/* SPDX-License-Identifier: GPL-2.0 */
> +#undef TRACE_SYSTEM
> +#define TRACE_SYSTEM kwatch
> +
> +#if !defined(_TRACE_KWATCH_H) || defined(TRACE_HEADER_MULTI_READ)
> +#define _TRACE_KWATCH_H
> +
> +#include <linux/tracepoint.h>
> +#include <linux/ptrace.h>
> +
> +#define KWATCH_STACK_DEPTH 8
> +
> +struct trace_seq;
> +const char *kwatch_trace_print_stack(struct trace_seq *p,
> + const unsigned long *stack,
> + unsigned int nr);
> +
> +TRACE_EVENT(kwatch_hit,
> + TP_PROTO(unsigned long ip, unsigned long sp, unsigned long addr,
> + u64 time_ns,
> + unsigned long *stack_entries, unsigned int stack_nr),
> + TP_ARGS(ip, sp, addr, time_ns, stack_entries, stack_nr),
> +
> + TP_STRUCT__entry(
> + __field(unsigned long, ip)
> + __field(unsigned long, sp)
> + __field(unsigned long, addr)
> + __field(u64, time_ns)
Move the time_ns to the first field, as unsigned long on 32 bit
architectures is 4 bytes, and this will make 4 byte "hole" in the event.
> + __field(unsigned int, stack_nr)
Make stack_nr the last element for the same reason.
> + __array(unsigned long, stack, KWATCH_STACK_DEPTH)
Make the above a dynamic array based on stack entries.
__dynamic_array(unsigned long, stack, min_t(unsigned int, stack_nr,
KWATCH_STACK_DEPTH);
> + ),
> +
> + TP_fast_assign(
> + unsigned int i;
unsigned long *stack = __get_dynamic_array(stack);
> +
> + __entry->ip = ip;
> + __entry->sp = sp;
> + __entry->addr = addr;
> + __entry->time_ns = time_ns;
> + __entry->stack_nr = min_t(unsigned int, stack_nr,
> + KWATCH_STACK_DEPTH);
> + for (i = 0; i < __entry->stack_nr; i++)
> + __entry->stack[i] = stack_entries[i];
stack[i] = stack_entries[i];
> + ),
> +
> + TP_printk("KWatch HIT: time=%llu.%06lu ip=%pS addr=0x%lx%s",
> + __entry->time_ns / 1000000000ULL,
> + (unsigned long)((__entry->time_ns / 1000ULL) % 1000000ULL),
> + (void *)__entry->ip, __entry->addr,
> + kwatch_trace_print_stack(p, __entry->stack,
kwatch_trace_print_stack(p, __get_dynamic_array(stack),
> + __entry->stack_nr))
> +);
> +
-- Steve
^ permalink raw reply
* Re: [RFC PATCH v4 1/3] trace: add lock-free stackmap for stack trace deduplication
From: Steven Rostedt @ 2026-07-14 21:11 UTC (permalink / raw)
To: Li Pengfei
Cc: Masami Hiramatsu, Mathieu Desnoyers, Mark Rutland,
Jonathan Corbet, Shuah Khan, linux-kernel, linux-trace-kernel,
linux-doc, linux-kselftest, lipengfei28, zhangbo56
In-Reply-To: <20260616064119.438063-2-lipengfei28@xiaomi.com>
On Tue, 16 Jun 2026 14:41:17 +0800
Li Pengfei <ljdlns1987@gmail.com> wrote:
> --- /dev/null
> +++ b/kernel/trace/trace_stackmap.c
> @@ -0,0 +1,889 @@
> +// SPDX-License-Identifier: GPL-2.0
> +/*
> + * Ftrace Stack Map - Lock-free stack trace deduplication for ftrace
> + *
> + * Modeled after tracing_map.c (used by hist triggers), this provides
> + * a lock-free hash map optimized for the ftrace hot path. The design
> + * is based on Dr. Cliff Click's non-blocking hash table algorithm.
> + *
> + * Key properties:
> + * - Lock-free insert via cmpxchg, safe in NMI/IRQ/any context
> + * - Pre-allocated element pool (zero allocation on hot path)
> + * - Linear probing with 2x over-provisioned table; probe length
> + * bounded by FTRACE_STACKMAP_MAX_PROBE to keep worst-case lookup
> + * cost constant even when the table is heavily loaded
> + * - Single global instance (initialized for the global trace array)
> + *
> + * Reset is a control-path operation, only allowed when tracing is
> + * stopped on the owning trace_array. The protocol is:
> + *
> + * - atomic_cmpxchg(&resetting, 0, 1) atomically claims reset rights
> + * and blocks new get_id() callers (they observe resetting=1 and
> + * return -EINVAL).
> + * - trace_types_lock serializes the tracer_tracing_is_on() check and
> + * the destructive ring-buffer reset against tracefs writes to
> + * tracing_on.
> + * - synchronize_rcu() drains in-flight get_id() callers from the
> + * ftrace callback path, which runs with preemption disabled.
> + *
> + * Online reset (with tracing active) is intentionally not supported
> + * to keep the design simple and the proof obligations small.
> + *
> + * The 32-bit jhash of the stack IPs is the hash table key. On hash
> + * collision, linear probing finds the next slot and full memcmp
> + * confirms the match.
> + *
> + * Concurrent userspace readers (cat stack_map / stack_map_bin) get
> + * a best-effort snapshot. They are coherent with the hot path
> + * (smp_load_acquire on entry->val); they are also serialized
> + * against reset via smap->reader_sem (readers take it in shared
> + * mode, reset in exclusive mode), so a reset cannot tear an
> + * iteration in progress -- it waits for active readers to drop
> + * the rwsem before clearing the map. The hot path is coordinated
> + * with reset separately, via acquire/release on smap->resetting.
> + */
> +
> +#include <linux/kernel.h>
> +#include <linux/slab.h>
> +#include <linux/jhash.h>
> +#include <linux/seq_file.h>
> +#include <linux/kallsyms.h>
> +#include <linux/vmalloc.h>
> +#include <linux/atomic.h>
> +#include <linux/local_lock.h>
> +#include <linux/percpu.h>
> +#include <linux/random.h>
> +#include <linux/rcupdate.h>
> +#include <linux/log2.h>
> +#include <asm/local.h>
> +
> +#include "trace.h"
> +#include "trace_stackmap.h"
> +
> +/*
> + * Bound the linear-probe scan length. With a 2x over-provisioned table,
> + * a well-distributed hash gives very short probe chains. Capping at 64
> + * keeps worst-case lookup O(1) even when the table is heavily loaded
> + * with claimed-but-empty slots from pool exhaustion.
> + */
> +#define FTRACE_STACKMAP_MAX_PROBE 64
> +
> +/*
> + * Memory ordering of entry->val: published with smp_store_release()
> + * by the inserter; consumed with smp_load_acquire() by every reader
> + * that dereferences the elt (get_id, seq_show, bin_open). This pairs
> + * the writes to elt->{nr,ips,ref_count} (initialized BEFORE the
> + * publish) with the reads of those fields (which happen AFTER the
> + * load). seq_start / seq_next only test val for NULL and use the
> + * acquire load purely to keep memory ordering symmetric.
> + */
> +
> +/*
> + * Each pre-allocated element holds one unique stack trace.
> + * Fixed size: MAX_DEPTH entries regardless of actual depth.
> + */
> +struct stackmap_elt {
> + u32 nr; /* actual number of IPs */
> + atomic_t ref_count;
> + unsigned long ips[FTRACE_STACKMAP_MAX_DEPTH];
> +};
> +
> +/*
> + * Hash table entry: a 32-bit key (jhash of stack) + pointer to elt.
> + * key == 0 means the slot is free.
> + */
> +struct stackmap_entry {
> + u32 key; /* 0 = free, non-zero = jhash */
> + struct stackmap_elt *val; /* NULL until fully published */
> +};
> +
> +static struct stackmap_elt *stackmap_load_elt(struct stackmap_entry *entry)
> +{
> + /*
> + * Pairs with the smp_store_release() that publishes entry->val
> + * after fully initializing the element payload.
> + */
> + return smp_load_acquire(&entry->val);
> +}
> +
> +struct ftrace_stackmap {
> + struct trace_array *tr; /* owning trace_array */
> + unsigned int map_bits;
> + unsigned int map_size; /* 1 << (map_bits + 1) */
> + unsigned int max_elts; /* 1 << map_bits */
> + u32 hash_seed; /* per-instance jhash seed */
> + atomic_t next_elt; /* index into elts pool */
> + struct stackmap_entry *entries; /* hash table */
> + struct stackmap_elt *elts; /* flat element pool */
> + atomic_t resetting;
> + /*
> + * Reader/reset serialization. Held in shared mode (read lock)
> + * across seq_file iteration and binary snapshot construction;
> + * held in exclusive mode (write lock) by reset's clearing
> + * phase. The hot path (get_id) does not take this lock — it
> + * uses smp_load_acquire/smp_store_release on entry->val and
> + * the resetting flag for the lock-free protocol.
> + */
> + struct rw_semaphore reader_sem;
> + /*
> + * Per-CPU counters using local_t. local_t increments are NMI-
> + * safe on all architectures (single-instruction or interrupt-
> + * masked) and avoid the raw_spinlock_t fallback that
> + * atomic64_t uses on 32-bit GENERIC_ATOMIC64 — which would
> + * deadlock if an NMI hit while the spinlock was held.
> + */
> + local_t __percpu *successes; /* events served (hits + new inserts) */
> + local_t __percpu *drops;
> +};
> +
> +/*
> + * Cap the bits parameter to keep worst-case allocations bounded:
> + * bits=18 → 256K elts, 512K slots, ~130 MB elt pool, ~130 MB bin
> + * export.
> + * Smaller workloads should use the default (14) which gives 16K elts
> + * (~8 MB pool); bump bits via the ftrace_stackmap.bits= kernel
> + * parameter for higher unique-stack capacity.
> + */
> +#define FTRACE_STACKMAP_BITS_MIN 10
> +#define FTRACE_STACKMAP_BITS_MAX 18
> +#define FTRACE_STACKMAP_BITS_DEFAULT 14
> +
> +static unsigned int stackmap_map_bits = FTRACE_STACKMAP_BITS_DEFAULT;
> +static int __init stackmap_bits_setup(char *str)
> +{
> + unsigned long val;
> +
> + if (kstrtoul(str, 0, &val))
> + return -EINVAL;
> + val = clamp_val(val, FTRACE_STACKMAP_BITS_MIN, FTRACE_STACKMAP_BITS_MAX);
> + stackmap_map_bits = val;
> + return 0;
> +}
> +early_param("ftrace_stackmap.bits", stackmap_bits_setup);
> +
> +/* --- Element pool --- */
> +
> +static struct stackmap_elt *stackmap_get_elt(struct ftrace_stackmap *smap)
> +{
> + int idx;
> +
> + /*
> + * Fast-path early-out once the pool is fully consumed. Avoids
> + * the contended atomic RMW on next_elt for every traced event
> + * after the pool is exhausted.
> + */
> + if (atomic_read(&smap->next_elt) >= smap->max_elts)
> + return NULL;
> +
> + idx = atomic_fetch_add_unless(&smap->next_elt, 1, smap->max_elts);
> + if (idx < smap->max_elts)
> + return &smap->elts[idx];
> + return NULL;
> +}
> +
> +/* --- Create / Destroy / Reset --- */
> +
> +struct ftrace_stackmap *ftrace_stackmap_create(struct trace_array *tr)
> +{
> + struct ftrace_stackmap *smap;
> + unsigned int bits;
> +
> + smap = kzalloc_obj(*smap, GFP_KERNEL);
> + if (!smap)
> + return ERR_PTR(-ENOMEM);
> +
> + /* Defensive clamp: reject bogus bits even if early_param is bypassed. */
> + bits = clamp_val(stackmap_map_bits,
> + FTRACE_STACKMAP_BITS_MIN,
> + FTRACE_STACKMAP_BITS_MAX);
> +
> + smap->tr = tr;
> + smap->map_bits = bits;
> + smap->max_elts = 1U << bits;
> + smap->map_size = 1U << (bits + 1); /* 2x over-provision */
> +
> + smap->entries = vzalloc(sizeof(*smap->entries) * smap->map_size);
Why not:
smap->entries = vcalloc(smap->map_size, sizeof(*smap->entries));
?
> + if (!smap->entries) {
> + kfree(smap);
> + return ERR_PTR(-ENOMEM);
> + }
Make the error paths have:
if (!smap->entries)
goto fail;
> +
> + /*
> + * Single large vmalloc of the element pool, indexed flat.
> + * At bits=18 this is 256K * sizeof(struct stackmap_elt). The
> + * struct is ~520 B (8 + 4 + 4 + 64*8), so total ~135 MB.
> + */
> + smap->elts = vzalloc(sizeof(*smap->elts) * (size_t)smap->max_elts);
vcalloc()?
> + if (!smap->elts) {
goto fail;
> + vfree(smap->entries);
> + kfree(smap);
> + return ERR_PTR(-ENOMEM);
> + }
> +
> + smap->successes = alloc_percpu(local_t);
> + if (!smap->successes) {
goto fail;
> + vfree(smap->elts);
> + vfree(smap->entries);
> + kfree(smap);
> + return ERR_PTR(-ENOMEM);
> + }
> + smap->drops = alloc_percpu(local_t);
> + if (!smap->drops) {
goto fail;
> + free_percpu(smap->successes);
> + vfree(smap->elts);
> + vfree(smap->entries);
> + kfree(smap);
> + return ERR_PTR(-ENOMEM);
> + }
> +
> + smap->hash_seed = get_random_u32();
> + atomic_set(&smap->next_elt, 0);
> + atomic_set(&smap->resetting, 0);
> + init_rwsem(&smap->reader_sem);
> +
> + return smap;
fail:
if (smap) {
free_percpu(smap->successes);
vfree(smap->elts);
vfree(smap->entries);
kfree(smap);
// As all the above handle passing in NULL just fine.
}
return ERR_PTR(-ENOMEM);
> +}
> +
> +void ftrace_stackmap_destroy(struct ftrace_stackmap *smap)
> +{
> + if (!smap || IS_ERR(smap))
> + return;
> + free_percpu(smap->drops);
> + free_percpu(smap->successes);
> + vfree(smap->elts);
> + vfree(smap->entries);
> + kfree(smap);
> +}
> +
> +/**
> + * ftrace_stackmap_reset - clear all entries in the stackmap
> + * @smap: the stackmap to reset
> + *
> + * Returns 0 on success, -EBUSY if another reset is already in
> + * progress, or if tracing is currently active on the owning
> + * trace_array.
> + *
> + * Online reset (with tracing active) is not supported. Caller must
> + * stop tracing first (echo 0 > tracing_on).
> + *
> + * Caller is process context (typically sysfs write handler).
> + *
> + * Protocol:
> + * 1. Atomically claim reset rights via cmpxchg on @resetting.
> + * 2. Take trace_types_lock to serialize against tracefs writes to
> + * tracing_on.
> + * 3. Verify tracing is stopped on @smap->tr; if not, release the
> + * claim and return -EBUSY. The resetting flag itself blocks
> + * any subsequent get_id() callers.
> + * 4. synchronize_rcu() drains in-flight get_id() callers from the
> + * ftrace callback path (which runs preempt-disabled).
> + * 5. Reset the ring buffer(s), then memset entries, elts, and
> + * counters.
> + * 6. Release the resetting flag with release semantics so any new
> + * get_id() observes a fully cleared map.
> + */
> +int ftrace_stackmap_reset(struct ftrace_stackmap *smap)
> +{
> + struct trace_array *tr;
> + int ret = 0;
> +
> + if (!smap)
> + return 0;
> +
> + if (atomic_cmpxchg(&smap->resetting, 0, 1) != 0)
> + return -EBUSY;
> +
> + mutex_lock(&trace_types_lock);
> +
> + tr = smap->tr;
> + if (tr && tracer_tracing_is_on(tr)) {
> + ret = -EBUSY;
> + goto out_unlock;
> + }
> +
> + /*
> + * synchronize_rcu() itself is a full barrier; no extra smp_mb()
> + * is needed before it. It drains in-flight ftrace callbacks that
> + * may have already passed the resetting check with the old value.
> + */
> + synchronize_rcu();
> +
> + /*
> + * Take the reader_sem in exclusive mode. This serializes the
> + * memset against any tracefs reader (seq_file iteration or
> + * stack_map_bin snapshot) that may currently hold the rwsem
> + * for read. synchronize_rcu() already drained the hot path;
> + * this rwsem covers process-context readers that aren't
> + * preempt-disabled.
> + */
> + down_write(&smap->reader_sem);
> +
> + /*
> + * Clear the ring buffer(s) BEFORE the map, both under the write
> + * lock. The ring buffer may still hold TRACE_STACK_ID events
> + * whose stack_id points at slots we are about to free/reuse.
> + * Resetting the buffer first guarantees an external observer
> + * never sees the inconsistent "trace still has <stack_id N> but
> + * the map is already empty" window: it sees either (old buffer,
> + * old map) or (cleared buffer, old map) or (cleared buffer,
> + * cleared map) -- never (old buffer, cleared map).
> + *
> + * Use tracing_reset_all_cpus() (not _online_cpus) so per-CPU
> + * buffers belonging to currently offline CPUs are also cleared.
> + * The ring buffer is allocated per-possible-CPU; an offline CPU's
> + * buffer can still hold a TRACE_STACK_ID event written before
> + * the CPU went offline. tracing_reset_online_cpus() iterates
> + * for_each_online_buffer_cpu() and would leave that data behind
> + * to be observed once the CPU comes back online (or by the
> + * trace reader, which iterates all allocated CPU buffers),
> + * recreating the stale-stack_id window we are trying to close.
> + *
> + * Since reset requires tracing to be stopped, this makes "reset"
> + * an explicitly destructive operation on the owning trace_array,
> + * keeping ring-buffer stack_ids and the map coherent.
> + */
> + if (tr) {
> + tracing_reset_all_cpus(&tr->array_buffer);
> +#ifdef CONFIG_TRACER_SNAPSHOT
> + if (tr->allocated_snapshot)
> + tracing_reset_all_cpus(&tr->snapshot_buffer);
> +#endif
> + }
> +
> + memset(smap->entries, 0, sizeof(*smap->entries) * smap->map_size);
> + memset(smap->elts, 0, sizeof(*smap->elts) * (size_t)smap->max_elts);
> +
> + atomic_set(&smap->next_elt, 0);
> + {
Do not add anonymous blocks in functions.
> + int cpu;
Just declare cpu at the beginning of the function.
> +
> + for_each_possible_cpu(cpu) {
> + local_set(per_cpu_ptr(smap->successes, cpu), 0);
> + local_set(per_cpu_ptr(smap->drops, cpu), 0);
> + }
> + }
> +
> + up_write(&smap->reader_sem);
> +
> +out_unlock:
> + mutex_unlock(&trace_types_lock);
> +
> + /* Release resetting=0 so new get_id() observes a cleared map. */
> + atomic_set_release(&smap->resetting, 0);
> + return ret;
> +}
> +
> +/* --- Core: get_id (lock-free, NMI-safe) --- */
> +
> +int ftrace_stackmap_get_id(struct ftrace_stackmap *smap,
> + unsigned long *ips, unsigned int nr_entries)
> +{
> + u32 key_hash, idx, test_key, trace_len;
> + struct stackmap_entry *entry;
> + struct stackmap_elt *val;
> + int probes = 0;
> +
> + /*
> + * atomic_read_acquire() pairs with atomic_set_release() in the
> + * reset path. This ensures that subsequent reads of entry->key
> + * and entry->val are ordered after this check; without acquire,
> + * the CPU would only have a control dependency, which orders
> + * subsequent stores but not loads (per LKMM).
> + */
> + if (!smap || !nr_entries || atomic_read_acquire(&smap->resetting))
> + return -EINVAL;
> + /*
> + * Never truncate: a stack deeper than the map can hold must not be
> + * silently shortened, or two distinct traces sharing their first
> + * FTRACE_STACKMAP_MAX_DEPTH frames would be merged into one
> + * stack_id. The caller is expected to fall back to a full stack
> + * trace for such events. Reject defensively in case of a future
> + * caller that forgets this contract.
> + */
> + if (nr_entries > FTRACE_STACKMAP_MAX_DEPTH)
> + return -E2BIG;
> +
> + trace_len = nr_entries * sizeof(unsigned long);
> + /*
> + * jhash2() requires the length in u32 units and the data to be
> + * u32-aligned. On 64-bit kernels sizeof(unsigned long)==8, so
> + * trace_len is always a multiple of 8 (hence of 4). Use jhash2
> + * directly; the cast to u32* is safe because ips[] is naturally
> + * aligned to sizeof(unsigned long) >= 4.
> + */
> + key_hash = jhash2((const u32 *)ips, trace_len / sizeof(u32),
> + smap->hash_seed);
> + if (key_hash == 0)
> + key_hash = 1; /* 0 means free slot */
> +
> + idx = key_hash >> (32 - (smap->map_bits + 1));
> +
> + while (probes < FTRACE_STACKMAP_MAX_PROBE) {
> + idx &= (smap->map_size - 1);
> + entry = &smap->entries[idx];
> + /*
> + * READ_ONCE() to avoid LKMM data race with concurrent
> + * cmpxchg(&entry->key, 0, key_hash) on this slot.
> + */
> + test_key = READ_ONCE(entry->key);
> +
> + if (test_key == key_hash) {
> + val = stackmap_load_elt(entry);
> + /*
> + * READ_ONCE(val->nr) keeps style consistent with
> + * the seq_show / bin_open readers. nr is write-once
> + * (set before publish, never modified afterwards),
> + * so the load is data-race-free, but READ_ONCE
> + * silences any analysis tool that flags a plain
> + * read of a field that is also read under acquire
> + * elsewhere.
> + */
> + if (val && READ_ONCE(val->nr) == nr_entries &&
> + memcmp(val->ips, ips, trace_len) == 0) {
> + /*
> + * ref_count is a best-effort popularity
> + * counter. On a long (from-boot, multi-hour)
> + * trace a hot stack can be hit billions of
> + * times. atomic_add_unless() gives true
> + * saturation at INT_MAX even under concurrent
> + * hits on multiple CPUs (a plain
> + * check-then-inc could let several CPUs past
> + * the check near the cap and still wrap).
> + */
> + atomic_add_unless(&val->ref_count, 1, INT_MAX);
> + /*
> + * successes/drops are best-effort throughput
> + * counters. Saturate at LONG_MAX so they do
> + * not wrap on long runs (notably where local_t
> + * is 32-bit), matching ref_count's behaviour.
> + */
> + local_add_unless(this_cpu_ptr(smap->successes),
> + 1, LONG_MAX);
> + return (int)idx;
> + }
> + /*
> + * val == NULL: another CPU is mid-insert, or this
> + * slot is "claimed but empty" (pool exhausted).
> + * val != NULL but mismatch: 32-bit hash collision
> + * with a different stack. In both cases, advance.
> + */
> + } else if (!test_key) {
> + /*
> + * Free slot: try to claim it.
> + *
> + * If two CPUs race here with the same key_hash
> + * (same stack), one loses the cmpxchg, advances,
> + * and may insert the same stack at a later slot.
> + * This can produce a small number of duplicate
> + * entries under heavy contention. The trade-off
> + * is accepted to keep the hot path lock-free;
> + * ref_count is split across the duplicates and
> + * total memory cost is bounded by the element
> + * pool size.
> + */
> + if (cmpxchg(&entry->key, 0, key_hash) == 0) {
> + struct stackmap_elt *elt;
> +
> + elt = stackmap_get_elt(smap);
> + if (!elt) {
> + /*
> + * Pool exhausted. We claimed this
> + * slot with cmpxchg but cannot fill
> + * it. Leave key set so the slot
> + * stays "claimed but empty" — future
> + * lookups treat val==NULL as a miss
> + * and probe past it. Cannot revert
> + * key=0 without racing other CPUs.
> + */
> + local_add_unless(this_cpu_ptr(smap->drops),
> + 1, LONG_MAX);
> + return -ENOSPC;
> + }
> +
> + elt->nr = nr_entries;
> + atomic_set(&elt->ref_count, 1);
> + memcpy(elt->ips, ips, trace_len);
> +
> + /*
> + * Publish elt with release semantics so the
> + * reader's smp_load_acquire can safely
> + * dereference val->nr / val->ips.
> + */
> + smp_store_release(&entry->val, elt);
> + local_add_unless(this_cpu_ptr(smap->successes),
> + 1, LONG_MAX);
> + return (int)idx;
> + }
> + /* cmpxchg failed; another CPU claimed this slot. */
> + }
> +
> + idx++;
> + probes++;
> + }
> +
> + local_add_unless(this_cpu_ptr(smap->drops), 1, LONG_MAX);
> + return -ENOSPC;
> +}
> +
> +/* --- Text export: /sys/kernel/debug/tracing/stack_map --- */
> +
> +struct stackmap_seq_private {
> + struct ftrace_stackmap *smap;
> +};
> +
> +static void *stackmap_seq_start(struct seq_file *m, loff_t *pos)
> +{
> + struct stackmap_seq_private *priv = m->private;
> + struct ftrace_stackmap *smap = priv->smap;
> + u32 i;
> +
> + if (!smap)
> + return NULL;
> + /*
> + * Take the reader_sem to serialize against ftrace_stackmap_reset(),
> + * which holds it for write while clearing the table. Released in
> + * stackmap_seq_stop(), which seq_file calls regardless of whether
> + * start() returned an element or NULL (per Documentation/filesystems
> + * /seq_file.rst: "the iterator value returned by start() or next()
> + * is guaranteed to be passed to a subsequent next() or stop()").
> + */
> + down_read(&smap->reader_sem);
> + for (i = *pos; i < smap->map_size; i++) {
> + if (READ_ONCE(smap->entries[i].key) &&
> + stackmap_load_elt(&smap->entries[i])) {
> + *pos = i;
> + return &smap->entries[i];
> + }
> + }
> + return NULL;
> +}
> +
> +static void *stackmap_seq_next(struct seq_file *m, void *v, loff_t *pos)
> +{
> + struct stackmap_seq_private *priv = m->private;
> + struct ftrace_stackmap *smap = priv->smap;
> + u32 i;
> +
> + if (!smap)
> + return NULL;
> + for (i = *pos + 1; i < smap->map_size; i++) {
> + if (READ_ONCE(smap->entries[i].key) &&
> + stackmap_load_elt(&smap->entries[i])) {
> + *pos = i;
> + return &smap->entries[i];
> + }
> + }
> + /*
> + * Advance *pos past the end so that on the next read() the
> + * subsequent stackmap_seq_start() call returns NULL and the
> + * iteration terminates. Without this, seq_read() would loop
> + * on the last element.
> + */
> + *pos = smap->map_size;
> + return NULL;
> +}
> +
> +static void stackmap_seq_stop(struct seq_file *m, void *v)
> +{
> + struct stackmap_seq_private *priv = m->private;
> + struct ftrace_stackmap *smap = priv->smap;
> +
> + /*
> + * seq_file invokes stop() unconditionally after each iteration
> + * pass (see seq_read_iter / traverse), even when start() returned
> + * NULL. Always release here, balanced against the down_read in
> + * stackmap_seq_start().
> + */
> + if (smap)
> + up_read(&smap->reader_sem);
> +}
> +
> +static int stackmap_seq_show(struct seq_file *m, void *v)
> +{
> + struct stackmap_entry *entry = v;
> + struct stackmap_seq_private *priv = m->private;
> + struct stackmap_elt *elt;
> + u32 idx = entry - priv->smap->entries;
> + u32 i, nr;
> +
> + elt = stackmap_load_elt(entry);
> + if (!elt)
> + return 0;
> +
> + nr = READ_ONCE(elt->nr);
> + if (nr > FTRACE_STACKMAP_MAX_DEPTH)
> + nr = FTRACE_STACKMAP_MAX_DEPTH;
> +
> + seq_printf(m, "stack_id %u [ref %u, depth %u]\n",
> + idx, atomic_read(&elt->ref_count), nr);
> + for (i = 0; i < nr; i++) {
> + unsigned long ip = elt->ips[i];
> +
> + /*
> + * Mirror trace_stack_print(): __ftrace_trace_stack()
> + * may replace trampoline addresses with
> + * FTRACE_TRAMPOLINE_MARKER before the stack reaches the
> + * map, and normal addresses must go through
> + * trace_adjust_address() (KASLR / module text delta)
> + * before symbolization. Without this the export would
> + * print a bogus symbol for the marker and unadjusted
> + * addresses for everything else.
> + */
> + if (ip == FTRACE_TRAMPOLINE_MARKER) {
> + seq_printf(m, " [%u] [FTRACE TRAMPOLINE]\n", i);
> + continue;
> + }
> + seq_printf(m, " [%u] %pS\n", i,
> + (void *)trace_adjust_address(priv->smap->tr, ip));
> + }
> + seq_putc(m, '\n');
> + return 0;
> +}
> +
> +static const struct seq_operations stackmap_seq_ops = {
> + .start = stackmap_seq_start,
> + .next = stackmap_seq_next,
> + .stop = stackmap_seq_stop,
> + .show = stackmap_seq_show,
> +};
> +
> +static int stackmap_open(struct inode *inode, struct file *file)
> +{
> + struct stackmap_seq_private *priv;
> + struct seq_file *m;
> + int ret;
> +
> + ret = seq_open_private(file, &stackmap_seq_ops,
> + sizeof(struct stackmap_seq_private));
> + if (ret)
> + return ret;
> + m = file->private_data;
> + priv = m->private;
> + priv->smap = inode->i_private;
> + return 0;
> +}
> +
> +/*
> + * Accept exactly "0" or "reset" (optionally followed by a single newline).
> + */
> +static bool stackmap_write_is_reset(const char *buf, size_t n)
> +{
> + if (n > 0 && buf[n - 1] == '\n')
> + n--;
> + return (n == 1 && buf[0] == '0') ||
> + (n == 5 && memcmp(buf, "reset", 5) == 0);
> +}
> +
> +static ssize_t stackmap_write(struct file *file, const char __user *ubuf,
> + size_t count, loff_t *ppos)
> +{
> + struct seq_file *m = file->private_data;
> + struct stackmap_seq_private *priv = m->private;
> + char buf[8];
> + size_t n = min(count, sizeof(buf) - 1);
> + int ret;
> +
> + if (n == 0)
> + return -EINVAL;
> + if (copy_from_user(buf, ubuf, n))
> + return -EFAULT;
> + buf[n] = '\0';
> +
> + if (!stackmap_write_is_reset(buf, n))
> + return -EINVAL;
> +
> + /*
> + * ftrace_stackmap_reset() atomically claims reset rights via
> + * cmpxchg and returns -EBUSY if another reset is in progress
> + * or if tracing is active.
> + */
> + ret = ftrace_stackmap_reset(priv->smap);
> + if (ret)
> + return ret;
> + return count;
> +}
> +
> +const struct file_operations ftrace_stackmap_fops = {
> + .open = stackmap_open,
> + .read = seq_read,
> + .write = stackmap_write,
> + .llseek = seq_lseek,
> + .release = seq_release_private,
> +};
> +
> +/* --- Stats --- */
> +
> +static int stackmap_stat_show(struct seq_file *m, void *v)
> +{
> + struct ftrace_stackmap *smap = m->private;
> + u64 successes = 0, drops = 0;
> + u32 entries;
> + int cpu;
> +
> + if (!smap) {
> + seq_puts(m, "stackmap not initialized\n");
> + return 0;
> + }
> +
> + entries = atomic_read(&smap->next_elt);
> + for_each_possible_cpu(cpu) {
> + successes += local_read(per_cpu_ptr(smap->successes, cpu));
> + drops += local_read(per_cpu_ptr(smap->drops, cpu));
> + }
> +
> + seq_printf(m, "entries: %u / %u\n", entries, smap->max_elts);
> + seq_printf(m, "table_size: %u\n", smap->map_size);
> + seq_printf(m, "successes: %llu\n", successes);
> + seq_printf(m, "drops: %llu\n", drops);
> + if (successes + drops > 0)
> + seq_printf(m, "success_rate: %llu%%\n",
> + successes * 100 / (successes + drops));
> + return 0;
> +}
> +
> +static int stackmap_stat_open(struct inode *inode, struct file *file)
> +{
> + return single_open(file, stackmap_stat_show, inode->i_private);
> +}
> +
> +const struct file_operations ftrace_stackmap_stat_fops = {
> + .open = stackmap_stat_open,
> + .read = seq_read,
> + .llseek = seq_lseek,
> + .release = single_release,
> +};
> +
> +/* --- Binary export --- */
> +
> +struct stackmap_bin_snapshot {
> + /*
> + * Use u64 (not size_t) so data[] is 8-byte aligned on both
> + * 32-bit and 64-bit architectures. The IP array within data[]
> + * is accessed as u64*, which would alignment-fault on strict
> + * architectures (e.g. older ARM, SPARC) if data[] started at
> + * a 4-byte boundary.
> + */
> + u64 size;
> + char data[];
> +};
> +
> +static int stackmap_bin_open(struct inode *inode, struct file *file)
> +{
> + struct ftrace_stackmap *smap = inode->i_private;
> + struct stackmap_bin_snapshot *snap;
> + struct ftrace_stackmap_bin_header *hdr;
> + size_t alloc_size, off;
> + u32 nr_entries, i, nr_stacks;
> +
> + if (!smap)
> + return -ENODEV;
> +
> + /*
> + * Worst-case allocation size: every populated entry uses a
> + * full-depth stack. The (+1) gives one slack slot in case a
> + * concurrent insert lands between this snapshot and iteration.
> + * The loop below performs an explicit bounds check anyway.
> + *
> + * At bits=18 this caps at ~135 MB. The file is mode 0440
> + * (TRACE_MODE_READ), so only privileged users can open it.
> + */
> + nr_entries = atomic_read(&smap->next_elt);
> + alloc_size = sizeof(*hdr) + (nr_entries + 1) *
> + (sizeof(struct ftrace_stackmap_bin_entry) +
> + FTRACE_STACKMAP_MAX_DEPTH * sizeof(u64));
Really should have ftrace_stackmap_bin_entry have a flexible array:
(move struct ftrace_stackmap_bin_entry *e to top)
alloc_size = sizeof(*hdr) + (nr_entries + 1) *
struct_size(e, ips, FTRACE_STACKMAP_MAX_DEPTH);
> +
> + snap = vmalloc(sizeof(*snap) + alloc_size);
> + if (!snap)
> + return -ENOMEM;
> +
> + hdr = (struct ftrace_stackmap_bin_header *)snap->data;
> + hdr->magic = FTRACE_STACKMAP_BIN_MAGIC;
> + hdr->version = FTRACE_STACKMAP_BIN_VERSION;
> + hdr->reserved = 0;
> + off = sizeof(*hdr);
> + nr_stacks = 0;
> +
> + /*
> + * Take reader_sem to serialize against ftrace_stackmap_reset(),
> + * which clears the table and elt pool under the write lock.
> + */
> + down_read(&smap->reader_sem);
> +
> + for (i = 0; i < smap->map_size; i++) {
> + struct stackmap_entry *entry = &smap->entries[i];
> + struct stackmap_elt *elt;
> + struct ftrace_stackmap_bin_entry *e;
move to top of function.
> + u64 *ips_out;
> + u32 k, nr;
> +
> + if (!READ_ONCE(entry->key))
> + continue;
> + elt = stackmap_load_elt(entry);
> + if (!elt)
> + continue;
> +
> + nr = READ_ONCE(elt->nr);
> + if (nr > FTRACE_STACKMAP_MAX_DEPTH)
> + nr = FTRACE_STACKMAP_MAX_DEPTH;
> +
> + /* Bounds check: stop if we would overflow the allocation. */
> + if (off + sizeof(*e) + nr * sizeof(u64) > alloc_size)
if (off + struct_size(e, ips, nr) > alloc_size)
> + break;
> +
> + e = (struct ftrace_stackmap_bin_entry *)(snap->data + off);
> + e->stack_id = i;
> + e->nr = nr;
> + e->ref_count = atomic_read(&elt->ref_count);
> + e->reserved = 0;
> + off += sizeof(*e);
delete the above.
> +
> + ips_out = (u64 *)(snap->data + off);
ips_out = e->ips;
> + for (k = 0; k < nr; k++) {
> + unsigned long ip = elt->ips[k];
> +
> + /*
> + * Emit the trampoline marker verbatim so userspace
> + * can render it as [FTRACE TRAMPOLINE]; pass every
> + * other address through trace_adjust_address() so the
> + * binary export follows the same address-adjustment
> + * rules as the text export.
> + */
> + if (ip == FTRACE_TRAMPOLINE_MARKER)
> + ips_out[k] = (u64)FTRACE_TRAMPOLINE_MARKER;
> + else
> + ips_out[k] = (u64)trace_adjust_address(smap->tr, ip);
> + }
> + off += nr * sizeof(u64);
off += struct_size(e, ips, nr);
> + nr_stacks++;
> + }
> +
> + up_read(&smap->reader_sem);
> +
> + hdr->nr_stacks = nr_stacks;
> + snap->size = off;
> + file->private_data = snap;
> + return 0;
> +}
> +
> +static ssize_t stackmap_bin_read(struct file *file, char __user *ubuf,
> + size_t count, loff_t *ppos)
> +{
> + struct stackmap_bin_snapshot *snap = file->private_data;
> +
> + if (!snap)
> + return -EINVAL;
> + return simple_read_from_buffer(ubuf, count, ppos, snap->data, snap->size);
> +}
> +
> +static int stackmap_bin_release(struct inode *inode, struct file *file)
> +{
> + vfree(file->private_data);
> + return 0;
> +}
> +
> +const struct file_operations ftrace_stackmap_bin_fops = {
> + .open = stackmap_bin_open,
> + .read = stackmap_bin_read,
> + .llseek = default_llseek,
> + .release = stackmap_bin_release,
> +};
> diff --git a/kernel/trace/trace_stackmap.h b/kernel/trace/trace_stackmap.h
> new file mode 100644
> index 000000000000..7c2e5ab9d36d
> --- /dev/null
> +++ b/kernel/trace/trace_stackmap.h
> @@ -0,0 +1,57 @@
> +/* SPDX-License-Identifier: GPL-2.0 */
> +#ifndef _TRACE_STACKMAP_H
> +#define _TRACE_STACKMAP_H
> +
> +#include <linux/types.h>
> +#include <linux/atomic.h>
> +
> +#define FTRACE_STACKMAP_MAX_DEPTH 64
> +
> +/* Binary export format */
> +#define FTRACE_STACKMAP_BIN_MAGIC 0x46534D42 /* 'FSMB' */
> +#define FTRACE_STACKMAP_BIN_VERSION 1
> +
> +struct ftrace_stackmap_bin_header {
> + u32 magic;
> + u32 version;
> + u32 nr_stacks;
> + u32 reserved;
> +};
> +
> +struct ftrace_stackmap_bin_entry {
> + u32 stack_id;
> + u32 nr;
> + u32 ref_count;
> + u32 reserved;
> + /* followed by u64 ips[nr] */
Why not make this a flexible array?
u64 ips[];
Then the code can be simpler as described above.
-- Steve
> +};
> +
> +struct trace_array;
> +
> +#ifdef CONFIG_FTRACE_STACKMAP
> +
> +struct ftrace_stackmap;
> +
> +struct ftrace_stackmap *ftrace_stackmap_create(struct trace_array *tr);
> +void ftrace_stackmap_destroy(struct ftrace_stackmap *smap);
> +int ftrace_stackmap_get_id(struct ftrace_stackmap *smap,
> + unsigned long *ips, unsigned int nr_entries);
> +int ftrace_stackmap_reset(struct ftrace_stackmap *smap);
> +
> +extern const struct file_operations ftrace_stackmap_fops;
> +extern const struct file_operations ftrace_stackmap_stat_fops;
> +extern const struct file_operations ftrace_stackmap_bin_fops;
> +
> +#else
> +
> +struct ftrace_stackmap;
> +static inline struct ftrace_stackmap *
> +ftrace_stackmap_create(struct trace_array *tr) { return NULL; }
> +static inline void ftrace_stackmap_destroy(struct ftrace_stackmap *s) { }
> +static inline int ftrace_stackmap_get_id(struct ftrace_stackmap *s,
> + unsigned long *ips, unsigned int n)
> +{ return -EOPNOTSUPP; }
> +static inline int ftrace_stackmap_reset(struct ftrace_stackmap *s) { return 0; }
> +
> +#endif
> +#endif /* _TRACE_STACKMAP_H */
^ permalink raw reply
* Re: [PATCH v1 06/11] rcu: Enable RCU callbacks to benefit from expedited grace periods
From: Paul E. McKenney @ 2026-07-14 18:48 UTC (permalink / raw)
To: Frederic Weisbecker
Cc: Puranjay Mohan, rcu, linux-kernel, linux-trace-kernel,
Neeraj Upadhyay, Joel Fernandes, Josh Triplett, Boqun Feng,
Uladzislau Rezki, Steven Rostedt, Mathieu Desnoyers,
Lai Jiangshan, Zqiang, Masami Hiramatsu, Davidlohr Bueso,
Breno Leitao
In-Reply-To: <alD6o01ukyliCS37@localhost.localdomain>
On Fri, Jul 10, 2026 at 03:58:59PM +0200, Frederic Weisbecker wrote:
> Le Wed, Jun 24, 2026 at 06:23:48AM -0700, Puranjay Mohan a écrit :
> > Currently, RCU callbacks only track normal grace-period sequence
> > numbers. This means callbacks must wait for normal grace periods to
> > complete even when expedited grace periods have already elapsed.
> >
> > Use the full struct rcu_gp_seq (which tracks both the normal and
> > expedited grace-period sequences) throughout the callback
> > infrastructure.
> >
> > rcu_segcblist_advance() now checks both normal and expedited GP
> > completion via poll_state_synchronize_rcu_full(), and becomes
> > parameterless since it reads the grace-period state internally.
> > rcu_segcblist_accelerate() stores the full state (both sequences)
> > instead of just the normal one. rcu_accelerate_cbs() and
> > rcu_accelerate_cbs_unlocked() use get_state_synchronize_rcu_full() to
> > capture both sequences, and the NOCB advance checks use
> > poll_state_synchronize_rcu_full() instead of comparing only the normal
> > sequence.
> >
> > srcu_segcblist_advance() becomes a standalone implementation because it
> > compares SRCU sequences directly and cannot use
> > poll_state_synchronize_rcu_full(), which reads RCU-specific globals.
> > srcu_segcblist_accelerate() sets the ->exp field to
> > RCU_GET_STATE_NOT_TRACKED so that poll_state_synchronize_rcu_full()
> > compares only ->norm and ignores ->exp.
> >
> > Reviewed-by: Paul E. McKenney <paulmck@kernel.org>
> > Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
> > ---
> > kernel/rcu/rcu_segcblist.c | 30 +++++++++++++++++++++++-------
> > kernel/rcu/rcu_segcblist.h | 2 +-
> > kernel/rcu/tree.c | 9 +++------
> > kernel/rcu/tree_nocb.h | 33 +++++++++++++++++++++++----------
> > 4 files changed, 50 insertions(+), 24 deletions(-)
> >
> > diff --git a/kernel/rcu/rcu_segcblist.c b/kernel/rcu/rcu_segcblist.c
> > index 4e3dfe42bc097..cf8951d33e767 100644
> > --- a/kernel/rcu/rcu_segcblist.c
> > +++ b/kernel/rcu/rcu_segcblist.c
> > @@ -12,6 +12,7 @@
> > #include <linux/kernel.h>
> > #include <linux/types.h>
> >
> > +#include "rcu.h"
> > #include "rcu_segcblist.h"
> >
> > /* Initialize simple callback list. */
> > @@ -494,9 +495,9 @@ static void rcu_segcblist_advance_compact(struct rcu_segcblist *rsclp, int i)
> >
> > /*
> > * Advance the callbacks in the specified rcu_segcblist structure based
> > - * on the current value passed in for the grace-period counter.
> > + * on the current value of the grace-period counter.
> > */
> > -void rcu_segcblist_advance(struct rcu_segcblist *rsclp, struct rcu_gp_seq *gsp)
> > +void rcu_segcblist_advance(struct rcu_segcblist *rsclp)
> > {
> > int i;
> >
> > @@ -509,7 +510,7 @@ void rcu_segcblist_advance(struct rcu_segcblist *rsclp, struct rcu_gp_seq *gsp)
> > * are ready to invoke, and put them into the RCU_DONE_TAIL segment.
> > */
> > for (i = RCU_WAIT_TAIL; i < RCU_NEXT_TAIL; i++) {
> > - if (ULONG_CMP_LT(gsp->norm, rsclp->gp_seq[i].norm))
> > + if (!poll_state_synchronize_rcu_full(&rsclp->gp_seq[i]))
>
> So after more careful review, the smp_mb() at the end of a successful
> poll_state_synchronize_rcu_full() is necessary here because the current locking
> is not enough to make sure we synchronize against the end of the grace period.
>
> But what about the smp_mb() at the beginning? Paul what is the point of this one
> already? It advertizes to pair with the smp_mb() on root cleanup but what
> exactly is to be ordered here? Why does gp cleanup need to synchronize with
> failing poll_state_synchronize_rcu_full() ? The smp_mb() before rcu_seq_snap()
> in get_state_synchronize_rcu_full() should already synchronize the accesses
> before that call against the beginning of the grace period.
>
> If we keep all these barriers around and both RCU_WAIT_TAIL and RCU_NEXT_READY
> need to be advanced, that makes 4 smp_mb() calls.
Apologies for the delay, I missed this one.
You are right that if poll_state_synchronize_rcu_full() fails we don't
need ordering. Especially given that poll_state_synchronize_rcu()
doesn't have this first memory barrier.
This fits in with get_state_synchronize_full() emulating a call to
synchronize_rcu() and poll_state_synchronize_rcu_full() emulating the
return from synchronize_rcu(). So get_state_synchronize_full() has
its smp_mb() at the beginning and poll_state_synchronize_rcu_full()
at the end.
The only rationale I can give for that initial smp_mb() is that in
the comment, but it makes no sense because the code prior to the call
to poll_state_synchronize_rcu_full() cannot know whether or not this
function will return false.
Puranjay, are you going to remove this smp_mb(), or would you prefer
that I do so?
> > @@ -637,14 +638,29 @@ void rcu_segcblist_merge(struct rcu_segcblist *dst_rsclp,
> >
> > void srcu_segcblist_advance(struct rcu_segcblist *rsclp, unsigned long seq)
> > {
> > - struct rcu_gp_seq gs = { .norm = seq };
> > + int i;
> > +
> > + WARN_ON_ONCE(!rcu_segcblist_is_enabled(rsclp));
> > + if (rcu_segcblist_restempty(rsclp, RCU_DONE_TAIL))
> > + return;
> > +
> > + for (i = RCU_WAIT_TAIL; i < RCU_NEXT_TAIL; i++) {
> > + if (ULONG_CMP_LT(seq, rsclp->gp_seq[i].norm))
> > + break;
>
> Why not use the same API here and consolidate the code? ->exp is RCU_GET_STATE_NOT_TRACKED so it's
> harmless?
>
> > @@ -1164,7 +1164,7 @@ static bool rcu_accelerate_cbs(struct rcu_node *rnp, struct rcu_data *rdp)
> > * accelerating callback invocation to an earlier grace-period
> > * number.
> > */
> > - gs.norm = rcu_seq_snap(&rcu_state.gp_seq);
> > + get_state_synchronize_rcu_full(&gs);
>
> I have similar concerns about the three smp_mb() in
> get_state_synchronize_rcu_full(). It could be just two (rcu_seq_snap()
> has a barrier that could be just one). Not sure if that matters but,
> just wanted to point that.
We need the one at the beginning of get_state_synchronize_rcu_full(),
but from what I can see, not the ones in the calls to rcu_seq_snap().
I blame laziness. We could make an rcu_seq_snap_no_ordering() that
didn't have the smp_mb(), but I didn't believe that the overhead would
be visible at the system level.
Thanx, Paul
> Thanks.
>
> --
> Frederic Weisbecker
> SUSE Labs
^ permalink raw reply
* [PATCH v5 0/5] Enable perf tracing for unprivileged users
From: Anubhav Shelat @ 2026-07-14 18:39 UTC (permalink / raw)
To: rostedt, mhiramat, mathieu.desnoyers, peterz, mingo, acme,
namhyung, mark.rutland, alexander.shishkin, jolsa, irogers,
adrian.hunter
Cc: linux-kernel, linux-trace-kernel, linux-perf-users,
Anubhav Shelat
Enable users to use perf-trace to trace their own processes, like strace
but without the overhead of ptrace(). Ensure that users cannot access
other users' or systemwide tracing data.
Changes in v5:
- Move event_define_fields() before directory creation. If
event_define_fields() fails then we don't need to cleanup whatever
dirs were created.
- New read-only eventfs file system with the same structure as
/sys/kernel/tracing/events/ to handle files read by unprivileged
users.
- Allow unprivileged users to fall back to /sys/kernel/events/ when they
cannot access /sys/kernel/tracing/events/.
- Factor out reused code into helper function that checks if a
tracepoint should be restricted in commit 5.
Changes in v4:
- Preserve security_perf_event_open(PERF_SECURITY_KERNEL) LSM hook in
the tp_bypass path.
- Lift the PERF_SAMPLE_IP check out of the tp_bypass path above the
PERF_SAMPLE_RAW branch so it applies to counting and sampling. This
also allows us to ensure PERF_SAMPLE_IP is set for uprobes.
- Block counting path for TRACE_EVENT_FL_CAP_ANY for unprivileged users
with sysctl_perf_event_paranoid > 1.
Changes in v3:
- Don't set PERF_SAMPLE_IP for unprivileged tracepoints. This allows us
to exclude PERF_SAMPLE_IP from kaddr_leak without weakening KASLR.
- Mount tracefs as world-traversable so users can access eventfs
directories.
Anubhav Shelat (5):
eventfs: define event fields before directory creation
tracefs: add read-only eventfs filesystem at /sys/kernel/events
perf tools: fall back to eventfs for unprivileged event discovery
perf evsel: don't set PERF_SAMPLE_IP for unprivileged tracepoints
perf: enable unprivileged syscall tracing with perf trace
fs/tracefs/event_inode.c | 61 ++++++++++++++++++
fs/tracefs/inode.c | 95 ++++++++++++++++++++++++++-
fs/tracefs/internal.h | 3 +
include/linux/trace_events.h | 1 +
include/linux/tracefs.h | 4 ++
include/uapi/linux/magic.h | 1 +
kernel/events/core.c | 28 +++++++-
kernel/trace/trace.h | 2 +
kernel/trace/trace_event_perf.c | 28 +++++++-
kernel/trace/trace_events.c | 100 +++++++++++++++++++++++++++--
tools/lib/api/fs/fs.c | 10 +++
tools/lib/api/fs/fs.h | 1 +
tools/lib/api/fs/tracing_path.c | 52 +++++++++++++--
tools/lib/api/fs/tracing_path.h | 1 +
tools/perf/util/evsel.c | 14 +++-
tools/perf/util/tp_pmu.c | 5 +-
tools/perf/util/trace-event-info.c | 19 +++---
17 files changed, 395 insertions(+), 30 deletions(-)
--
2.54.0
^ permalink raw reply
* [RFC PATCH 13/13] Documentation/dev-tools: document KWatch
From: Jinchao Wang @ 2026-07-14 18:33 UTC (permalink / raw)
To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
Masami Hiramatsu
Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260714182243.10687-1-wangjinchao600@gmail.com>
Describe what KWatch is for, how it compares with KASAN and KFENCE,
the debugfs configuration interface, the watch expression syntax,
how to read hits from the trace buffer (including after a crash),
and the current limitations.
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
Documentation/dev-tools/index.rst | 1 +
Documentation/dev-tools/kwatch.rst | 193 +++++++++++++++++++++++++++++
2 files changed, 194 insertions(+)
create mode 100644 Documentation/dev-tools/kwatch.rst
diff --git a/Documentation/dev-tools/index.rst b/Documentation/dev-tools/index.rst
index 59cbb77b33ff..f4c748da63db 100644
--- a/Documentation/dev-tools/index.rst
+++ b/Documentation/dev-tools/index.rst
@@ -30,6 +30,7 @@ Documentation/process/debugging/index.rst
ubsan
kmemleak
kcsan
+ kwatch
lkmm/index
kfence
kselftest
diff --git a/Documentation/dev-tools/kwatch.rst b/Documentation/dev-tools/kwatch.rst
new file mode 100644
index 000000000000..8ead0beb06b6
--- /dev/null
+++ b/Documentation/dev-tools/kwatch.rst
@@ -0,0 +1,193 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+======================================
+KWatch - Kernel Memory Watchpoint Tool
+======================================
+
+Overview
+========
+
+KWatch is a runtime-configurable debugging tool for locating kernel memory
+corruption. It arms hardware breakpoints (watchpoints) on a target address
+while a chosen function is executing, and reports the exact instruction that
+touches the watched memory, together with a stack trace, through a
+tracepoint.
+
+Unlike shadow-memory sanitizers, KWatch does not detect invalid accesses in
+general; it answers a narrower but common question during corruption hunts:
+"who writes to this address?". This includes in-bounds logical overwrites
+that KASAN cannot see, because the rogue writer modifies valid memory
+through a valid pointer, just at the wrong time or with the wrong data.
+
+Comparison with other tools:
+
+* KASAN detects out-of-bounds and use-after-free accesses, but reports the
+ symptom (the invalid access), not the writer that corrupted the data
+ earlier. It requires a rebuild and has significant CPU and memory
+ overhead, and its redzones perturb memory layout, which can hide
+ timing-sensitive bugs.
+* KFENCE is a low-overhead sampling detector for slab objects; it cannot be
+ pointed at one specific address.
+* Hardware breakpoints via kgdb or perf can watch an address, but only a
+ fixed one, system-wide, for the whole run. KWatch resolves the address
+ dynamically at function entry (for example "argument 2 of this function,
+ plus offset 8, dereferenced once") and disarms it again at function exit,
+ so short-lived and per-invocation objects can be watched too.
+
+KWatch has near-zero overhead while armed: the watched function pays for
+one kprobe/kretprobe pair plus programming of the debug registers; the rest
+of the system runs at full speed.
+
+Requirements
+============
+
+* ``CONFIG_KWATCH=y`` or ``m``. The Kconfig symbol depends on
+ ``CONFIG_PERF_EVENTS``, ``CONFIG_DEBUG_FS`` and an architecture that
+ provides ``HAVE_REINSTALL_HW_BREAKPOINT`` (currently x86 only).
+* Resolving symbol names in watch expressions requires ``CONFIG_KWATCH=y``
+ (built-in); a module can only watch absolute hexadecimal addresses.
+
+Usage
+=====
+
+KWatch is configured through a single debugfs file::
+
+ /sys/kernel/debug/kwatch/config
+
+Writing a configuration string starts a watch session (stopping any previous
+one); reading the file shows the active configuration and hit-rejection
+counters. The configuration is a whitespace-separated list of ``key=value``
+tokens:
+
+=================== ==========================================================
+Key Meaning
+=================== ==========================================================
+``func_name`` Function whose execution opens the watch window.
+``func_offset`` Instruction offset inside ``func_name`` at which the
+ watchpoint is armed (default 0 = function entry).
+``watch_expr`` Expression describing the address to watch (see below).
+``watch_len`` Watched length in bytes: 1, 2, 4 or 8 (default 8).
+``access_type`` 0 = write (default), 1 = read, 2 = read/write,
+ 3 = execute.
+``depth`` Recursion depth at which the window opens (default 0).
+``max_watch`` Number of hardware watchpoints to preallocate
+ (default 4).
+``max_concurrency`` Maximum number of tasks concurrently inside the watch
+ window (default 256).
+``duration`` For global watches: seconds until automatic stop.
+=================== ==========================================================
+
+Watch expressions
+-----------------
+
+The address to watch is computed at function entry from::
+
+ watch_expr={base}[+-offset][->[+-]offset]...
+
+* ``base`` is one of:
+
+ - ``arg1`` ... ``arg6``: a function argument (register calling
+ convention),
+ - ``stack``: the kernel stack pointer at the probe point,
+ - an absolute hexadecimal address, e.g. ``0xffffffff81234567``,
+ - a global symbol name (built-in KWatch only).
+
+* ``+offset`` / ``-offset`` adjusts the current address.
+* ``->offset`` loads the pointer stored at the current address (via
+ ``get_kernel_nofault()``) and then applies the offset. Up to four chain
+ elements are supported; offsets must be explicit (``->`` alone is
+ rejected).
+
+Given::
+
+ struct some_struct {
+ struct some_struct *ptr; /* offset 0 */
+ int num; /* offset 8 */
+ };
+
+ void target_function(struct some_struct *arg1);
+
+typical expressions are:
+
+=========================== ==============================================
+Expression Watches
+=========================== ==============================================
+``watch_expr=arg1`` ``&arg1->ptr`` (the pointer field itself)
+``watch_expr=arg1+8`` ``&arg1->num``
+``watch_expr=arg1->0`` ``&arg1->ptr->ptr`` (one dereference)
+``watch_expr=arg1->8`` ``&arg1->ptr->num``
+``watch_expr=0xffff...+8`` absolute address plus 8
+=========================== ==============================================
+
+Example: catch whoever overwrites ``arg1->num`` of a function while that
+function runs::
+
+ echo "func_name=target_function watch_expr=arg1+8 watch_len=4" \
+ > /sys/kernel/debug/kwatch/config
+
+Watching global variables
+-------------------------
+
+A global variable has no natural function window. When ``duration`` is
+given without ``func_name``, KWatch starts an internal anchor kernel thread
+that sleeps inside a dummy function, and uses that function as the window::
+
+ echo "watch_expr=jiffies_wobble duration=60 watch_len=8" \
+ > /sys/kernel/debug/kwatch/config
+
+The session tears itself down when the duration expires.
+
+Reading hits
+------------
+
+Hits are emitted as the ``kwatch:kwatch_hit`` tracepoint, which is safe in
+NMI-like contexts where printk is not. Each event carries the timestamp,
+the instruction pointer, the watched address and a short stack trace::
+
+ echo 1 > /sys/kernel/debug/tracing/events/kwatch/kwatch_hit/enable
+ cat /sys/kernel/debug/tracing/trace_pipe
+
+If the corruption crashes the machine, the ring buffer can still be
+recovered:
+
+* ``echo 1 > /proc/sys/kernel/ftrace_dump_on_oops`` (or the
+ ``ftrace_dump_on_oops`` boot parameter) dumps the buffer to the console
+ on an oops.
+* With kdump, the buffer is present in the vmcore and can be read with
+ ``crash> trace``.
+* ``CONFIG_PSTORE_FTRACE`` persists it across reboots on supported
+ platforms.
+
+Limitations
+===========
+
+* Functions that run in a genuine NMI(-like) context are rejected at
+ function entry; rejected invocations never open a watch window and are
+ counted in the ``nmi_rejected`` field of the config file. Watching
+ functions reachable from NMI handlers is out of scope.
+* The number of concurrent watchpoints is bounded by the CPU's debug
+ registers (typically 4).
+* If the target address cannot be resolved at arming time (for example a
+ ``get_kernel_nofault()`` failure on a swapped or unmapped page), the
+ watchpoint is not armed for that invocation.
+* Offsets in watch expressions are static; dynamic indexing such as
+ ``arg1->ptr[arg2]`` is not supported.
+* arm64 is not yet supported: stepping over a hit that has a custom
+ overflow handler needs a generic mechanism in the arch code, which is
+ planned as a follow-up series.
+
+Implementation notes
+====================
+
+The implementation lives in ``mm/kwatch/`` and is split into a control
+plane (``core.c``, the debugfs interface), an execution plane (``probe.c``
+and ``deref.c``: kprobe/kretprobe window management and address
+resolution), and a resource plane (``hwbp.c`` and ``task_ctx.c``).
+
+Hardware watchpoints are preallocated as perf events on every CPU and
+re-pointed at hit time with ``modify_wide_hw_breakpoint_local()``, a new
+hw_breakpoint API that updates the breakpoint on the local CPU without
+releasing its slot; other CPUs are updated by asynchronous IPIs. Per-task
+window state is kept in a fixed-size, lockless open-addressing array
+claimed with ``cmpxchg()``, so the hit path performs no allocation and
+takes no locks, which keeps it safe in atomic and NMI-like contexts.
--
2.53.0
^ permalink raw reply related
* [RFC PATCH 12/13] mm/kwatch: add KUnit tests for the watch expression parser
From: Jinchao Wang @ 2026-07-14 18:33 UTC (permalink / raw)
To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
Masami Hiramatsu
Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260714182243.10687-1-wangjinchao600@gmail.com>
Cover base anchors (stack, argN, absolute address), positive and
negative offsets, dereference chains, and rejection of malformed
expressions (missing offsets, bad argument index, junk offsets).
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
mm/kwatch/.kunitconfig | 9 +++
mm/kwatch/Kconfig | 10 +++
mm/kwatch/Makefile | 1 +
mm/kwatch/deref_test.c | 137 +++++++++++++++++++++++++++++++++++++++++
4 files changed, 157 insertions(+)
create mode 100644 mm/kwatch/.kunitconfig
create mode 100644 mm/kwatch/deref_test.c
diff --git a/mm/kwatch/.kunitconfig b/mm/kwatch/.kunitconfig
new file mode 100644
index 000000000000..7e977ddf0da1
--- /dev/null
+++ b/mm/kwatch/.kunitconfig
@@ -0,0 +1,9 @@
+CONFIG_KUNIT=y
+CONFIG_KWATCH=y
+CONFIG_KWATCH_KUNIT_TEST=y
+CONFIG_PERF_EVENTS=y
+CONFIG_HAVE_HW_BREAKPOINT=y
+CONFIG_HAVE_REINSTALL_HW_BREAKPOINT=y
+CONFIG_KPROBES=y
+CONFIG_KRETPROBES=y
+CONFIG_PRINTK=y
diff --git a/mm/kwatch/Kconfig b/mm/kwatch/Kconfig
index b1c37a829dd5..74083040a1a3 100644
--- a/mm/kwatch/Kconfig
+++ b/mm/kwatch/Kconfig
@@ -15,3 +15,13 @@ config KWATCH
exact instruction causing the illegal access.
If unsure, say N.
+
+config KWATCH_KUNIT_TEST
+ bool "KUnit tests for KWatch" if !KUNIT_ALL_TESTS
+ depends on KWATCH && KUNIT
+ default KUNIT_ALL_TESTS
+ help
+ Enable KUnit tests for the KWatch kernel module.
+ This suite tests the core parsing logic, the pointer-chasing
+ finite state machine, and edge cases involving complex watchpoint
+ expressions. If unsure, say N.
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index 02d7917602f1..1d223d73b461 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,4 @@
obj-$(CONFIG_KWATCH) += kwatch.o
kwatch-y := core.o deref.o task_ctx.o hwbp.o probe.o anchor.o
+kwatch-$(CONFIG_KWATCH_KUNIT_TEST) += deref_test.o
diff --git a/mm/kwatch/deref_test.c b/mm/kwatch/deref_test.c
new file mode 100644
index 000000000000..094b7afeb235
--- /dev/null
+++ b/mm/kwatch/deref_test.c
@@ -0,0 +1,137 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <kunit/test.h>
+#include "kwatch.h"
+#include <linux/string.h>
+
+static void kwatch_test_parse_deref_chain(struct kunit *test)
+{
+ struct kwatch_config cfg;
+ int ret;
+
+ // Test 1: stack
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "stack");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_STACK);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+
+ // Test 2: arg1
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg1");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG1);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+
+ // Test 3: arg6+8
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg6+8");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG6);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 8);
+
+ // Test 4: arg2-16
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg2-16");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG2);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], -16);
+
+ // Test 5: arg3->8
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg3->8");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG3);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[1], 8);
+
+ // Test 6: arg4+8->16
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg4+8->16");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG4);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 8);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[1], 16);
+
+ // Test 7: arg5-8->-16
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg5-8->-16");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG5);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], -8);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[1], -16);
+
+ // Test 8: stack->0->8
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "stack->0->8");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_STACK);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 3);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[1], 0);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[2], 8);
+
+ // Test 9: arg1->+8
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg1->+8");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ARG1);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 2);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 0);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[1], 8);
+
+ // Test 9.1: arg1-> (implicit 0 should fail)
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg1->");
+ KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+ // Test 9.2: stack->->8 (implicit 0 should fail)
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "stack->->8");
+ KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+ // Test 10: Invalid base
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "invalid_base");
+ KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+ // Test 11: Invalid offset
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg1+abc");
+ KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+ // Test 12: Invalid arg
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "arg7");
+ KUNIT_EXPECT_EQ(test, ret, -EINVAL);
+
+ // Test 13: Absolute address
+ memset(&cfg, 0, sizeof(cfg));
+ ret = kwatch_deref_parse(&cfg, "0xffffffff81000000+8");
+ KUNIT_EXPECT_EQ(test, ret, 0);
+ KUNIT_EXPECT_EQ(test, cfg.base, KWATCH_BASE_ABS_ADDR);
+ KUNIT_EXPECT_EQ(test, cfg.sym_addr, 0xffffffff81000000UL);
+ KUNIT_EXPECT_EQ(test, cfg.offset_count, 1);
+ KUNIT_EXPECT_EQ(test, cfg.offsets[0], 8);
+}
+
+static struct kunit_case kwatch_deref_test_cases[] = {
+ KUNIT_CASE(kwatch_test_parse_deref_chain),
+ {}
+};
+
+static struct kunit_suite kwatch_deref_test_suite = {
+ .name = "kwatch_deref",
+ .test_cases = kwatch_deref_test_cases,
+};
+
+kunit_test_suite(kwatch_deref_test_suite);
+
+MODULE_DESCRIPTION("KUnit tests for the KWatch watch expression parser");
+MODULE_LICENSE("GPL");
--
2.53.0
^ permalink raw reply related
* [RFC PATCH 11/13] mm/kwatch: add debugfs control plane
From: Jinchao Wang @ 2026-07-14 18:33 UTC (permalink / raw)
To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
Masami Hiramatsu
Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260714182243.10687-1-wangjinchao600@gmail.com>
Wire the pieces together behind a single debugfs file,
/sys/kernel/debug/kwatch/config. Writing a key=value configuration
string stops any active session and starts a new one; reading shows
the active configuration and the nmi_rejected counter. An open-count
guard keeps the file single-open and a mutex serializes
start/stop/auto-stop against each other.
Add the Kconfig entry and hook mm/kwatch into the mm build. KWatch
can be built in or as a module; symbol-name watch expressions need
the built-in flavour (kallsyms_lookup_name is not exported).
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
MAINTAINERS | 8 ++
mm/Kconfig | 1 +
mm/Makefile | 1 +
mm/kwatch/Kconfig | 17 +++
mm/kwatch/Makefile | 2 +-
mm/kwatch/core.c | 325 +++++++++++++++++++++++++++++++++++++++++++++
6 files changed, 353 insertions(+), 1 deletion(-)
create mode 100644 mm/kwatch/Kconfig
create mode 100644 mm/kwatch/core.c
diff --git a/MAINTAINERS b/MAINTAINERS
index 7cc4bca5a2c5..b6371f92fe5c 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -14578,6 +14578,14 @@ S: Supported
T: git git://git.kernel.org/pub/scm/virt/kvm/kvm.git
F: arch/x86/kvm/xen.*
+KWATCH
+M: Jinchao Wang <wangjinchao600@gmail.com>
+L: linux-mm@kvack.org
+S: Maintained
+F: Documentation/dev-tools/kwatch.rst
+F: include/trace/events/kwatch.h
+F: mm/kwatch/
+
L3MDEV
M: David Ahern <dsahern@kernel.org>
L: netdev@vger.kernel.org
diff --git a/mm/Kconfig b/mm/Kconfig
index 9e0ca4824905..cac75a46e21a 100644
--- a/mm/Kconfig
+++ b/mm/Kconfig
@@ -1510,5 +1510,6 @@ config LAZY_MMU_MODE_KUNIT_TEST
If unsure, say N.
source "mm/damon/Kconfig"
+source "mm/kwatch/Kconfig"
endmenu
diff --git a/mm/Makefile b/mm/Makefile
index eff9f9e7e061..80c688330358 100644
--- a/mm/Makefile
+++ b/mm/Makefile
@@ -92,6 +92,7 @@ obj-$(CONFIG_PAGE_POISONING) += page_poison.o
obj-$(CONFIG_KASAN) += kasan/
obj-$(CONFIG_KFENCE) += kfence/
obj-$(CONFIG_KMSAN) += kmsan/
+obj-$(CONFIG_KWATCH) += kwatch/
obj-$(CONFIG_FAILSLAB) += failslab.o
obj-$(CONFIG_FAIL_PAGE_ALLOC) += fail_page_alloc.o
obj-$(CONFIG_MEMTEST) += memtest.o
diff --git a/mm/kwatch/Kconfig b/mm/kwatch/Kconfig
new file mode 100644
index 000000000000..b1c37a829dd5
--- /dev/null
+++ b/mm/kwatch/Kconfig
@@ -0,0 +1,17 @@
+config KWATCH
+ tristate "Kernel Watch Framework"
+ depends on PERF_EVENTS && HAVE_HW_BREAKPOINT && DEBUG_FS
+ depends on HAVE_REINSTALL_HW_BREAKPOINT
+ select KPROBES
+ select KRETPROBES
+ select STACKTRACE
+ help
+ A generalized hardware-assisted memory monitor utility.
+ It provides a low-overhead, real-time trigger mechanism to monitor
+ kernel memory safely in atomic contexts using hardware breakpoints.
+
+ KWatch is designed to catch silent memory corruptions, stack
+ overwrites, and complex Heisenbugs by synchronously trapping the
+ exact instruction causing the illegal access.
+
+ If unsure, say N.
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index b196c794619a..02d7917602f1 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
obj-$(CONFIG_KWATCH) += kwatch.o
-kwatch-y := deref.o task_ctx.o hwbp.o probe.o anchor.o
+kwatch-y := core.o deref.o task_ctx.o hwbp.o probe.o anchor.o
diff --git a/mm/kwatch/core.c b/mm/kwatch/core.c
new file mode 100644
index 000000000000..548d0cdd0812
--- /dev/null
+++ b/mm/kwatch/core.c
@@ -0,0 +1,325 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/kstrtox.h>
+#include <linux/module.h>
+#include <linux/slab.h>
+#include <linux/string.h>
+#include <linux/types.h>
+#include <linux/atomic.h>
+#include <linux/debugfs.h>
+#include <linux/mutex.h>
+#include "kwatch.h"
+
+static struct kwatch_config kwatch_config;
+static bool watching_active;
+
+static struct dentry *dbgfs_dir;
+static struct dentry *dbgfs_config;
+static DEFINE_MUTEX(kwatch_dbgfs_mutex);
+static atomic_t dbgfs_config_busy = ATOMIC_INIT(0);
+
+static int kwatch_start_watching(void)
+{
+ int ret;
+
+ if (!strlen(kwatch_config.func_name)) {
+ if (kwatch_config.duration > 0) {
+ strscpy(kwatch_config.func_name, "kwatch_global_anchor",
+ sizeof(kwatch_config.func_name));
+ } else {
+ pr_err("func_name or duration is required\n");
+ return -EINVAL;
+ }
+ } else if (kwatch_config.duration > 0 &&
+ strcmp(kwatch_config.func_name, "kwatch_global_anchor")) {
+ pr_warn("duration is ignored when watching a specific function\n");
+ }
+
+ if (kwatch_config.access_type > 3) {
+ pr_err("Invalid access_type (must be 0-3)\n");
+ return -EINVAL;
+ }
+
+ ret = kwatch_hwbp_prealloc(kwatch_config.max_watch,
+ kwatch_config.access_type);
+ if (ret) {
+ pr_err("kwatch_hwbp_prealloc ret: %d\n", ret);
+ return ret;
+ }
+
+ ret = kwatch_tsk_ctx_prealloc(kwatch_config.max_concurrency);
+ if (ret) {
+ kwatch_hwbp_free();
+ return ret;
+ }
+
+ ret = kwatch_probe_start(&kwatch_config);
+ if (ret) {
+ pr_err("kwatch_probe_start ret: %d\n", ret);
+ kwatch_tsk_ctx_free();
+ kwatch_hwbp_free();
+ return ret;
+ }
+
+ if (!strcmp(kwatch_config.func_name, "kwatch_global_anchor")) {
+ ret = kwatch_anchor_start(kwatch_config.duration);
+ if (ret) {
+ kwatch_probe_stop();
+ synchronize_rcu();
+ kwatch_tsk_ctx_release_wps();
+ kwatch_hwbp_free();
+ kwatch_tsk_ctx_free();
+ return ret;
+ }
+ }
+
+ watching_active = true;
+ return 0;
+}
+
+static void kwatch_stop_watching(void)
+{
+ watching_active = false;
+
+ kwatch_anchor_stop();
+ /* after kthread_stop: the dead thread cannot re-mark expiry */
+ kwatch_anchor_clear_expired();
+
+ kwatch_probe_stop();
+ synchronize_rcu();
+ kwatch_tsk_ctx_release_wps();
+ /*
+ * Waits for disarm IPIs and unregisters breakpoints: no #DB can
+ * reach the ctx pool once this returns.
+ */
+ kwatch_hwbp_free();
+ kwatch_tsk_ctx_free();
+}
+
+void kwatch_auto_stop(void)
+{
+ mutex_lock(&kwatch_dbgfs_mutex);
+ /* the expired check neutralizes work items from torn-down sessions */
+ if (watching_active && kwatch_anchor_has_expired()) {
+ kwatch_stop_watching();
+ pr_info("watch duration expired, stopped watching\n");
+ }
+ mutex_unlock(&kwatch_dbgfs_mutex);
+}
+
+static int kwatch_config_parse(char *buf, struct kwatch_config *cfg)
+{
+ char *token, *key, *val;
+ int ret = 0;
+
+ memset(cfg, 0, sizeof(*cfg));
+ cfg->max_concurrency = 256;
+ cfg->max_watch = 4;
+ cfg->watch_len = 8;
+ cfg->access_type = 0;
+
+ while ((token = strsep(&buf, " \t\n")) != NULL) {
+ if (!*token)
+ continue;
+ key = strsep(&token, "=");
+ val = token;
+ if (!key || !val)
+ return -EINVAL;
+
+ if (!strcmp(key, "func_name")) {
+ strscpy(cfg->func_name, val, sizeof(cfg->func_name));
+ } else if (!strcmp(key, "func_offset")) {
+ ret = kstrtou16(val, 0, &cfg->func_offset);
+ } else if (!strcmp(key, "depth")) {
+ ret = kstrtou16(val, 0, &cfg->depth);
+ } else if (!strcmp(key, "max_concurrency")) {
+ ret = kstrtou16(val, 0, &cfg->max_concurrency);
+ } else if (!strcmp(key, "max_watch")) {
+ ret = kstrtou16(val, 0, &cfg->max_watch);
+ } else if (!strcmp(key, "access_type")) {
+ ret = kstrtouint(val, 0, &cfg->access_type);
+ } else if (!strcmp(key, "watch_len")) {
+ ret = kstrtou16(val, 0, &cfg->watch_len);
+ if (!ret && cfg->watch_len != 1 &&
+ cfg->watch_len != 2 && cfg->watch_len != 4 &&
+ cfg->watch_len != 8)
+ ret = -EINVAL;
+ } else if (!strcmp(key, "duration")) {
+ ret = kstrtou16(val, 0, &cfg->duration);
+ } else if (!strcmp(key, "watch_expr")) {
+ strscpy(cfg->watch_expr, val, sizeof(cfg->watch_expr));
+ ret = kwatch_deref_parse(cfg, val);
+ }
+
+ if (ret)
+ return ret;
+ }
+ return 0;
+}
+
+static int kwatch_dbgfs_open(struct inode *inode, struct file *file)
+{
+ if (atomic_cmpxchg(&dbgfs_config_busy, 0, 1))
+ return -EBUSY;
+ return 0;
+}
+
+static int kwatch_dbgfs_release(struct inode *inode, struct file *file)
+{
+ atomic_set(&dbgfs_config_busy, 0);
+ return 0;
+}
+
+static ssize_t kwatch_dbgfs_read(struct file *file, char __user *user_buf,
+ size_t count, loff_t *ppos)
+{
+ char *out_buf;
+ size_t len = 0;
+ ssize_t ret;
+
+ out_buf = kzalloc(MAX_CONFIG_STR_LEN, GFP_KERNEL);
+ if (!out_buf)
+ return -ENOMEM;
+
+ if (watching_active) {
+ len += scnprintf(out_buf + len, MAX_CONFIG_STR_LEN - len,
+ "func_name=%s\n"
+ "func_offset=%u\n"
+ "depth=%u\n"
+ "duration=%u\n"
+ "max_concurrency=%u\n"
+ "max_watch=%u\n"
+ "access_type=%u\n"
+ "watch_len=%u\n",
+ kwatch_config.func_name,
+ kwatch_config.func_offset, kwatch_config.depth,
+ kwatch_config.duration,
+ kwatch_config.max_concurrency,
+ kwatch_config.max_watch,
+ kwatch_config.access_type,
+ kwatch_config.watch_len);
+
+ if (kwatch_config.base == KWATCH_BASE_GLOBAL_SYM) {
+ len += scnprintf(out_buf + len, MAX_CONFIG_STR_LEN - len,
+ "sym_addr=0x%lx\n", kwatch_config.sym_addr);
+ }
+
+ len += scnprintf(out_buf + len, MAX_CONFIG_STR_LEN - len,
+ "watch_expr=%s\n"
+ "nmi_rejected=%lu\n",
+ kwatch_config.watch_expr,
+ kwatch_probe_nmi_rejected());
+ } else {
+ len = scnprintf(out_buf, MAX_CONFIG_STR_LEN, "not watching\n");
+ }
+
+ ret = simple_read_from_buffer(user_buf, count, ppos, out_buf, len);
+ kfree(out_buf);
+ return ret;
+}
+
+static ssize_t kwatch_dbgfs_write(struct file *file, const char __user *buffer,
+ size_t count, loff_t *ppos)
+{
+ char *input_alloc;
+ char *parse_str;
+ int ret;
+
+ if (count == 0 || count >= MAX_CONFIG_STR_LEN)
+ return -EINVAL;
+
+ input_alloc = memdup_user_nul(buffer, count);
+ if (IS_ERR(input_alloc))
+ return PTR_ERR(input_alloc);
+
+ mutex_lock(&kwatch_dbgfs_mutex);
+
+ if (watching_active)
+ kwatch_stop_watching();
+
+ parse_str = strim(input_alloc);
+
+ if (!strlen(parse_str)) {
+ ret = -EINVAL;
+ goto out;
+ }
+
+ ret = kwatch_config_parse(parse_str, &kwatch_config);
+ if (ret) {
+ pr_err("Failed to parse config %d\n", ret);
+ goto out;
+ }
+
+ ret = kwatch_start_watching();
+ if (ret) {
+ pr_err("Failed to start watching with %d\n", ret);
+ goto out;
+ }
+
+ ret = count;
+
+out:
+ mutex_unlock(&kwatch_dbgfs_mutex);
+ kfree(input_alloc);
+ return ret;
+}
+
+static const struct file_operations kwatch_fops = {
+ .owner = THIS_MODULE,
+ .open = kwatch_dbgfs_open,
+ .release = kwatch_dbgfs_release,
+ .read = kwatch_dbgfs_read,
+ .write = kwatch_dbgfs_write,
+};
+
+static int __init kwatch_init(void)
+{
+ int ret = 0;
+
+ memset(&kwatch_config, 0, sizeof(kwatch_config));
+
+ dbgfs_dir = debugfs_create_dir("kwatch", NULL);
+ if (IS_ERR(dbgfs_dir)) {
+ ret = PTR_ERR(dbgfs_dir);
+ goto err_dir;
+ }
+
+ dbgfs_config = debugfs_create_file("config", 0600, dbgfs_dir, NULL,
+ &kwatch_fops);
+ if (IS_ERR(dbgfs_config)) {
+ ret = PTR_ERR(dbgfs_config);
+ goto err_file;
+ }
+
+ pr_info("module loaded\n");
+ return 0;
+
+err_file:
+ debugfs_remove_recursive(dbgfs_dir);
+ dbgfs_dir = NULL;
+err_dir:
+ return ret;
+}
+module_init(kwatch_init);
+
+static void __exit kwatch_exit(void)
+{
+ mutex_lock(&kwatch_dbgfs_mutex);
+ if (watching_active)
+ kwatch_stop_watching();
+ mutex_unlock(&kwatch_dbgfs_mutex);
+
+ /* the anchor thread is dead: nothing can schedule new work now */
+ kwatch_anchor_cancel_work();
+
+ debugfs_remove_recursive(dbgfs_dir);
+ dbgfs_dir = NULL;
+
+ pr_info("kwatch unloaded\n");
+}
+module_exit(kwatch_exit);
+
+MODULE_AUTHOR("Jinchao Wang <wangjinchao600@gmail.com>");
+MODULE_DESCRIPTION("Kernel watchpoint");
+MODULE_LICENSE("GPL");
--
2.53.0
^ permalink raw reply related
* [RFC PATCH 10/13] mm/kwatch: add anchor thread for global watchpoints
From: Jinchao Wang @ 2026-07-14 18:32 UTC (permalink / raw)
To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
Masami Hiramatsu
Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260714182243.10687-1-wangjinchao600@gmail.com>
Global variables have no function whose execution can bound the
watch window. Provide one: a kernel thread sleeps for the configured
duration inside a dedicated noinline function,
kwatch_global_anchor(), and the probe runtime hooks that function
like any other target.
When the duration expires the thread schedules a work item that
tears the session down; the expired flag is cleared under the
control-plane mutex so a stale work item from a previous session
cannot stop a new one.
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
mm/kwatch/Makefile | 2 +-
mm/kwatch/anchor.c | 82 ++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 83 insertions(+), 1 deletion(-)
create mode 100644 mm/kwatch/anchor.c
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index f04673cc5b1c..b196c794619a 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
obj-$(CONFIG_KWATCH) += kwatch.o
-kwatch-y := deref.o task_ctx.o hwbp.o probe.o
+kwatch-y := deref.o task_ctx.o hwbp.o probe.o anchor.o
diff --git a/mm/kwatch/anchor.c b/mm/kwatch/anchor.c
new file mode 100644
index 000000000000..11da6aff9413
--- /dev/null
+++ b/mm/kwatch/anchor.c
@@ -0,0 +1,82 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/kthread.h>
+#include <linux/wait.h>
+#include <linux/jiffies.h>
+#include <linux/err.h>
+#include <linux/workqueue.h>
+
+#include "kwatch.h"
+
+static DECLARE_WAIT_QUEUE_HEAD(kwatch_anchor_wq);
+static struct task_struct *kwatch_anchor_tsk;
+static bool kwatch_anchor_expired;
+
+bool kwatch_anchor_has_expired(void)
+{
+ return READ_ONCE(kwatch_anchor_expired);
+}
+
+void kwatch_anchor_clear_expired(void)
+{
+ WRITE_ONCE(kwatch_anchor_expired, false);
+}
+
+static void kwatch_auto_stop_handler(struct work_struct *work)
+{
+ kwatch_auto_stop();
+}
+
+static DECLARE_WORK(kwatch_auto_stop_work, kwatch_auto_stop_handler);
+
+noinline void kwatch_global_anchor(unsigned long duration_sec)
+{
+ wait_event_timeout(kwatch_anchor_wq, kthread_should_stop(),
+ duration_sec * HZ);
+}
+
+static int kwatch_anchor_thread_fn(void *data)
+{
+ unsigned long duration = (unsigned long)data;
+
+ kwatch_global_anchor(duration);
+
+ if (!kthread_should_stop()) {
+ /* mark before scheduling; cleared under the control mutex */
+ WRITE_ONCE(kwatch_anchor_expired, true);
+ schedule_work(&kwatch_auto_stop_work);
+ }
+
+ while (!kthread_should_stop())
+ schedule_timeout_uninterruptible(HZ);
+
+ return 0;
+}
+
+int kwatch_anchor_start(u16 duration)
+{
+ kwatch_anchor_tsk = kthread_run(kwatch_anchor_thread_fn,
+ (void *)(unsigned long)duration,
+ "kwatch_anchor");
+ if (IS_ERR(kwatch_anchor_tsk)) {
+ int ret = PTR_ERR(kwatch_anchor_tsk);
+
+ kwatch_anchor_tsk = NULL;
+ return ret;
+ }
+ return 0;
+}
+
+void kwatch_anchor_stop(void)
+{
+ if (kwatch_anchor_tsk) {
+ kthread_stop(kwatch_anchor_tsk);
+ kwatch_anchor_tsk = NULL;
+ }
+}
+
+void kwatch_anchor_cancel_work(void)
+{
+ cancel_work_sync(&kwatch_auto_stop_work);
+}
--
2.53.0
^ permalink raw reply related
* [RFC PATCH 09/13] mm/kwatch: add probe lifecycle runtime
From: Jinchao Wang @ 2026-07-14 18:32 UTC (permalink / raw)
To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
Masami Hiramatsu
Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260714182243.10687-1-wangjinchao600@gmail.com>
Open and close the watch window with a kretprobe on the target
function: the entry handler tracks per-task nesting depth and, when
the configured depth is reached, resolves the watch expression and
arms a watchpoint; the exit handler disarms it. An optional kprobe
at func_offset arms mid-function instead of at entry.
Functions running in a real NMI(-like) context are rejected once, at
function entry, by comparing the NMI nesting count against the one
NMI-like layer that int3-based kprobe delivery itself adds; a
companion kprobe with a post_handler pins the probe point so jump
optimization cannot change the delivery mechanism after it is
sampled. Rejections are counted and exposed to the control plane.
A global epoch versioning scheme invalidates stale per-task state
across reconfigurations, and a per-CPU mute flag keeps window
management quiet while a CPU rewrites its own debug registers.
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
mm/kwatch/Makefile | 2 +-
mm/kwatch/probe.c | 263 +++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 264 insertions(+), 1 deletion(-)
create mode 100644 mm/kwatch/probe.c
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index b2bc3003c89b..f04673cc5b1c 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
obj-$(CONFIG_KWATCH) += kwatch.o
-kwatch-y := deref.o task_ctx.o hwbp.o
+kwatch-y := deref.o task_ctx.o hwbp.o probe.o
diff --git a/mm/kwatch/probe.c b/mm/kwatch/probe.c
new file mode 100644
index 000000000000..af6e0af45c10
--- /dev/null
+++ b/mm/kwatch/probe.c
@@ -0,0 +1,263 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/atomic.h>
+#include <linux/kprobes.h>
+#include <linux/kallsyms.h>
+#include <linux/percpu.h>
+#include <linux/preempt.h>
+#include <linux/sched.h>
+
+#include "kwatch.h"
+#define TRAMPOLINE_CHECK_DEPTH 16
+static DEFINE_PER_CPU(bool, kwatch_probe_cpu_muted);
+
+struct kwatch_probe_ctx {
+ struct kprobe kp;
+ struct kretprobe rp;
+ struct kprobe pin_kp;
+ const struct kwatch_config *cfg;
+ bool rp_via_int3;
+
+ u32 epoch;
+};
+
+static struct kwatch_probe_ctx kwatch_probe_ctx;
+static atomic_long_t kwatch_nmi_rejected;
+
+unsigned long kwatch_probe_nmi_rejected(void)
+{
+ return atomic_long_read(&kwatch_nmi_rejected);
+}
+
+/*
+ * True if the probed function itself runs in an NMI-like context.
+ * int3-based kprobe delivery adds one NMI-like layer of its own;
+ * delivery is pinned at registration so the subtraction stays exact.
+ */
+static bool kwatch_probed_ctx_in_nmi(bool via_int3)
+{
+ return (preempt_count() & NMI_MASK) > (via_int3 ? NMI_OFFSET : 0);
+}
+
+static void kwatch_pin_post_handler(struct kprobe *p, struct pt_regs *regs,
+ unsigned long flags)
+{
+ /* a post_handler pins the probepoint: no jump optimization */
+}
+
+bool kwatch_probe_validate_hit(struct pt_regs *regs,
+ struct task_struct *arm_tsk)
+{
+ struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+
+ if (unlikely(!ctx))
+ return true;
+
+ if (arm_tsk != current ||
+ ctx->depth != kwatch_probe_ctx.cfg->depth + 1)
+ return true;
+
+ return false;
+}
+
+void kwatch_probe_mute(bool mute)
+{
+ __this_cpu_write(kwatch_probe_cpu_muted, mute);
+}
+
+static inline bool kwatch_probe_is_muted(void)
+{
+ return __this_cpu_read(kwatch_probe_cpu_muted);
+}
+
+enum kwatch_probe_position {
+ KWATCH_PROBE_POSITION_ENTRY,
+ KWATCH_PROBE_POSITION_ACTIVE,
+ KWATCH_PROBE_POSITION_EXIT
+};
+
+static bool kwatch_tsk_ctx_check(enum kwatch_probe_position pos)
+{
+ struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(true);
+ u32 epoch;
+
+ if (unlikely(!ctx))
+ return false;
+
+ /* Pairs with smp_store_release() in kwatch_probe_start/stop() */
+ epoch = smp_load_acquire(&kwatch_probe_ctx.epoch);
+
+ if (unlikely(ctx->epoch != epoch))
+ kwatch_tsk_ctx_reset(ctx, epoch);
+
+ if (unlikely(!epoch))
+ return false;
+
+ switch (pos) {
+ case KWATCH_PROBE_POSITION_ENTRY:
+ ctx->depth++;
+ return true;
+ case KWATCH_PROBE_POSITION_ACTIVE:
+ return true;
+ case KWATCH_PROBE_POSITION_EXIT:
+ if (unlikely(ctx->depth == 0)) {
+ kwatch_tsk_ctx_put();
+ return false;
+ }
+
+ ctx->depth--;
+ if (ctx->depth == 0) {
+ kwatch_tsk_ctx_put();
+ return false;
+ }
+ return true;
+ }
+ return false;
+}
+
+static int kwatch_activate_handler(struct kprobe *p, struct pt_regs *regs)
+{
+ struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+ unsigned long watch_addr;
+ u16 watch_len;
+
+ if (unlikely(!ctx))
+ return 0;
+
+ if (unlikely(kwatch_probe_is_muted()))
+ return 0;
+
+ if (unlikely(!kwatch_tsk_ctx_check(KWATCH_PROBE_POSITION_ACTIVE)))
+ return 0;
+
+ if (ctx->depth != kwatch_probe_ctx.cfg->depth + 1 || ctx->wp)
+ return 0;
+
+ if (kwatch_deref_resolve(kwatch_probe_ctx.cfg, regs, &watch_addr,
+ &watch_len))
+ return 0;
+
+ if (kwatch_hwbp_get(&ctx->wp))
+ return 0;
+
+ kwatch_hwbp_arm(ctx->wp, watch_addr, watch_len);
+ return 0;
+}
+
+static int kwatch_lifecycle_entry(struct kretprobe_instance *ri,
+ struct pt_regs *regs)
+{
+ /*
+ * Single policy point: the target function's context is judged once
+ * here. A rejected invocation never increments depth, so the offset
+ * kprobe path inherits the verdict through the depth check.
+ */
+ if (unlikely(kwatch_probed_ctx_in_nmi(kwatch_probe_ctx.rp_via_int3))) {
+ atomic_long_inc(&kwatch_nmi_rejected);
+ return 1; /* NMI context is unsupported: no window, no return hook */
+ }
+
+ if (!kwatch_tsk_ctx_check(KWATCH_PROBE_POSITION_ENTRY))
+ return 0;
+
+ if (kwatch_probe_ctx.cfg->func_offset == 0)
+ kwatch_activate_handler(NULL, regs);
+
+ return 0;
+}
+
+static int kwatch_lifecycle_exit(struct kretprobe_instance *ri,
+ struct pt_regs *regs)
+{
+ struct kwatch_tsk_ctx *ctx = kwatch_tsk_ctx_get(false);
+
+ if (unlikely(!ctx))
+ return 0;
+
+ if (!kwatch_tsk_ctx_check(KWATCH_PROBE_POSITION_EXIT))
+ return 0;
+
+ if (ctx->depth == kwatch_probe_ctx.cfg->depth) {
+ struct kwatch_watchpoint *wp = xchg(&ctx->wp, NULL);
+
+ if (wp)
+ kwatch_hwbp_put(wp);
+ }
+
+ return 0;
+}
+
+int kwatch_probe_start(struct kwatch_config *cfg)
+{
+ static u32 next_epoch;
+ u32 current_epoch;
+ int ret;
+
+ /*
+ * Lockless check to prevent concurrent starts. Strictly serialized
+ * by the control plane mutex, but serves as a sanity check.
+ */
+ if (smp_load_acquire(&kwatch_probe_ctx.epoch) != 0)
+ return -EBUSY;
+
+ memset(&kwatch_probe_ctx, 0, sizeof(kwatch_probe_ctx));
+ kwatch_probe_ctx.cfg = cfg;
+
+ /*
+ * Pin the entry probepoint before the kretprobe registers, so its
+ * delivery (int3 vs ftrace) can never change under jump optimization.
+ * register_kretprobe() clears kp.post_handler, hence the companion.
+ */
+ kwatch_probe_ctx.pin_kp.symbol_name = cfg->func_name;
+ kwatch_probe_ctx.pin_kp.post_handler = kwatch_pin_post_handler;
+ ret = register_kprobe(&kwatch_probe_ctx.pin_kp);
+ if (ret < 0)
+ return ret;
+
+ kwatch_probe_ctx.rp.entry_handler = kwatch_lifecycle_entry;
+ kwatch_probe_ctx.rp.handler = kwatch_lifecycle_exit;
+ kwatch_probe_ctx.rp.kp.symbol_name = cfg->func_name;
+
+ ret = register_kretprobe(&kwatch_probe_ctx.rp);
+ if (ret < 0) {
+ unregister_kprobe(&kwatch_probe_ctx.pin_kp);
+ return ret;
+ }
+ kwatch_probe_ctx.rp_via_int3 = !kprobe_ftrace(&kwatch_probe_ctx.rp.kp);
+
+ if (cfg->func_offset) {
+ kwatch_probe_ctx.kp.symbol_name = cfg->func_name;
+ kwatch_probe_ctx.kp.offset = cfg->func_offset;
+ kwatch_probe_ctx.kp.pre_handler = kwatch_activate_handler;
+
+ ret = register_kprobe(&kwatch_probe_ctx.kp);
+ if (ret) {
+ unregister_kretprobe(&kwatch_probe_ctx.rp);
+ unregister_kprobe(&kwatch_probe_ctx.pin_kp);
+ return ret;
+ }
+ }
+
+ current_epoch = ++next_epoch;
+ if (unlikely(!current_epoch))
+ current_epoch = ++next_epoch;
+
+ /* Pairs with smp_load_acquire() in kwatch_tsk_ctx_check() */
+ smp_store_release(&kwatch_probe_ctx.epoch, current_epoch);
+
+ return 0;
+}
+
+void kwatch_probe_stop(void)
+{
+ if (!kwatch_probe_ctx.epoch)
+ return;
+
+ /* Pairs with smp_load_acquire() in kwatch_tsk_ctx_check() */
+ smp_store_release(&kwatch_probe_ctx.epoch, 0);
+
+ if (kwatch_probe_ctx.cfg->func_offset > 0)
+ unregister_kprobe(&kwatch_probe_ctx.kp);
+
+ unregister_kretprobe(&kwatch_probe_ctx.rp);
+ unregister_kprobe(&kwatch_probe_ctx.pin_kp);
+}
--
2.53.0
^ permalink raw reply related
* [PATCH v5 5/5] perf: enable unprivileged syscall tracing with perf trace
From: Anubhav Shelat @ 2026-07-14 18:31 UTC (permalink / raw)
To: mpetlan, Peter Zijlstra, Ingo Molnar, Arnaldo Carvalho de Melo,
Namhyung Kim, Mark Rutland, Alexander Shishkin, Jiri Olsa,
Ian Rogers, Adrian Hunter, James Clark, Steven Rostedt,
Masami Hiramatsu, Mathieu Desnoyers, linux-perf-users,
linux-kernel, linux-trace-kernel
Cc: Anubhav Shelat
In-Reply-To: <20260714183150.292861-2-ashelat@redhat.com>
Allow unprivileged users to trace their own processes' syscalls using
perf trace, similar to strace without the overhead of ptrace().
Currently, perf trace requires CAP_PERFMON or paranoid level ≤ 1 even
though the kernel has existing infrastructure (TRACE_EVENT_FL_CAP_ANY)
designed to mark syscall tracepoints as safe for unprivileged access.
To fix this:
1. Loosen the condition in perf_event_open() which requires privileges
for all events with exclude_kernel=0. This allows perf_event_open() to
bypass the paranoid check for task-attached tracepoint events. Ensure
that sample types which can expose kernel addresses to unprivileged
users are blocked. Ensure the PERF_SECURITY_KERNEL LSM hook is
preserved.
2. Add a check to perf_trace_event_perm() to block PERF_SAMPLE_IP on
kernel tracepoints for unprivileged users to prevent KASLR bypass. We do
this here rather than in kaddr_leak because perf_trace_event_perm() can
distinguish between kernel tracepoints and uprobe tracepoints, where the
IP is a safe user space address and is necessary for uprobe
functionality.
3. Restrict pure counting events (no PERF_SAMPLE_RAW) to
TRACE_EVENT_FL_CAP_ANY tracepoints preventing unprivileged users from
counting internal kernel tracepoints while preserving current
behavior for exclude_kernel=1 events.
Example usage after this change:
$ perf trace ls # works as unprivileged user
$ perf trace # system-wide, still requires privileges
$ perf trace -p 1234 # requires ptrace permission on pid 1234
Assisted-by: CLAUDE:claude-opus-4 Apogee
Signed-off-by: Anubhav Shelat <ashelat@redhat.com>
---
kernel/events/core.c | 28 +++++++++++++++++++++++++---
kernel/trace/trace_event_perf.c | 28 +++++++++++++++++++++++++++-
2 files changed, 52 insertions(+), 4 deletions(-)
diff --git a/kernel/events/core.c b/kernel/events/core.c
index 954c36e28101..48bfff07ae02 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -13910,9 +13910,31 @@ SYSCALL_DEFINE5(perf_event_open,
return err;
if (!attr.exclude_kernel) {
- err = perf_allow_kernel();
- if (err)
- return err;
+ bool tp_bypass = false;
+
+ /* Check unprivileged tracepoints */
+ if (attr.type == PERF_TYPE_TRACEPOINT && pid != -1) {
+ /*
+ * Block sample types that expose kernel addresses to
+ * prevent KASLR bypass
+ */
+ u64 kaddr_leak = PERF_SAMPLE_CALLCHAIN |
+ PERF_SAMPLE_BRANCH_STACK |
+ PERF_SAMPLE_ADDR |
+ PERF_SAMPLE_REGS_INTR;
+
+ tp_bypass = !(attr.sample_type & kaddr_leak);
+ }
+
+ if (!tp_bypass) {
+ err = perf_allow_kernel();
+ if (err)
+ return err;
+ } else {
+ err = security_perf_event_open(PERF_SECURITY_KERNEL);
+ if (err)
+ return err;
+ }
}
if (attr.namespaces) {
diff --git a/kernel/trace/trace_event_perf.c b/kernel/trace/trace_event_perf.c
index 5b272856e5ab..a264154b460e 100644
--- a/kernel/trace/trace_event_perf.c
+++ b/kernel/trace/trace_event_perf.c
@@ -24,6 +24,16 @@ typedef typeof(unsigned long [PERF_MAX_TRACE_SIZE / sizeof(unsigned long)])
/* Count the events in use (per event id, not per instance) */
static int total_ref_count;
+/* Check if perf tracepoint is restricted for unprivileged users */
+static bool perf_tp_is_restricted(struct perf_event *p_event)
+{
+ if (p_event->attr.exclude_kernel)
+ return false;
+ if (sysctl_perf_event_paranoid <= 1 || perfmon_capable())
+ return false;
+ return true;
+}
+
static int perf_trace_event_perm(struct trace_event_call *tp_event,
struct perf_event *p_event)
{
@@ -72,9 +82,25 @@ static int perf_trace_event_perm(struct trace_event_call *tp_event,
return -EINVAL;
}
+ /*
+ * PERF_SAMPLE_IP on kernel tracepoints exposes a kernel text
+ * address, weakening KASLR. Block for unprivileged users unless
+ * the tracepoint is a uprobe (userspace IP, safe to expose).
+ */
+ if ((p_event->attr.sample_type & PERF_SAMPLE_IP) &&
+ !(tp_event->flags & TRACE_EVENT_FL_UPROBE) &&
+ perf_tp_is_restricted(p_event))
+ return -EACCES;
+
/* No tracing, just counting, so no obvious leak */
- if (!(p_event->attr.sample_type & PERF_SAMPLE_RAW))
+ if (!(p_event->attr.sample_type & PERF_SAMPLE_RAW)) {
+ /* Prevent unprivileged users from counting kernel tracepoints */
+ if (perf_tp_is_restricted(p_event) &&
+ !(p_event->attach_state == PERF_ATTACH_TASK &&
+ (tp_event->flags & TRACE_EVENT_FL_CAP_ANY)))
+ return -EACCES;
return 0;
+ }
/* Some events are ok to be traced by non-root users... */
if (p_event->attach_state == PERF_ATTACH_TASK) {
--
2.54.0
^ permalink raw reply related
* [PATCH v5 2/5] tracefs: add read-only eventfs filesystem at /sys/kernel/events
From: Anubhav Shelat @ 2026-07-14 18:31 UTC (permalink / raw)
To: mpetlan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
Ackerley Tng, Fuad Tabba, David Hildenbrand, Anubhav Shelat,
Christian Brauner, linux-kernel, linux-trace-kernel
In-Reply-To: <20260714183150.292861-2-ashelat@redhat.com>
Introduce a new read-only pseudo-filesystem "eventfs" mounted at
/sys/kernel/events that exposes trace event format and id files
(mode 0444) to unprivileged users. This allows tools like perf to
discover event formats without requiring access to the full
tracefs/debugfs mount.
The eventfs filesystem reuses the existing eventfs_inode lazy-lookup
infrastructure. A new set of super_operations
(eventfs_ro_super_operations) shares the tracefs inode allocator so
that eventfs_get_inode() and get_tracefs() work on the RO superblock.
The superblock is manually initialized to ensure the root inode is
allocated with tracefs_alloc_inode, allowing the root to serve
directly as the events directory without another events subdirectory.
Each qualifying event gets a subsystem directory containing format
and id files. The top-level events directory also exposes header_page
and header_event. Similar to tracefs, a change will need to be made in
systemd to mount this filesystem automatically.
Assisted-by: CLAUDE:claude-opus-4 Apogee
Signed-off-by: Anubhav Shelat <ashelat@redhat.com>
---
fs/tracefs/event_inode.c | 61 +++++++++++++++++++++++
fs/tracefs/inode.c | 95 +++++++++++++++++++++++++++++++++++-
fs/tracefs/internal.h | 3 ++
include/linux/trace_events.h | 1 +
include/linux/tracefs.h | 4 ++
include/uapi/linux/magic.h | 1 +
kernel/trace/trace.h | 2 +
kernel/trace/trace_events.c | 87 +++++++++++++++++++++++++++++++++
8 files changed, 252 insertions(+), 2 deletions(-)
diff --git a/fs/tracefs/event_inode.c b/fs/tracefs/event_inode.c
index 39c7a34531e8..fd6f63ec3ce0 100644
--- a/fs/tracefs/event_inode.c
+++ b/fs/tracefs/event_inode.c
@@ -812,6 +812,67 @@ struct eventfs_inode *eventfs_create_events_dir(const char *name, struct dentry
return ERR_PTR(-ENOMEM);
}
+/**
+ * eventfs_create_events_dir_ro - create a read-only events directory
+ * @name: The name of the top level directory to create.
+ * @entries: A list of entries that represent the files under this directory
+ * @size: The number of @entries
+ * @data: The default data to pass to the files (an entry may override it).
+ *
+ * This function configures the eventfs filesystem root as a read-only
+ * trace event directory using the existing eventfs_inode lazy-lookup
+ * infrastructure.
+ *
+ * See eventfs_create_dir() for use of @entries.
+ */
+struct eventfs_inode *eventfs_create_events_dir_ro(const char *name,
+ const struct eventfs_entry *entries,
+ int size, void *data)
+{
+ struct dentry *dentry;
+ struct eventfs_root_inode *rei;
+ struct eventfs_inode *ei;
+ struct tracefs_inode *ti;
+ struct inode *inode;
+
+ dentry = eventfs_ro_get_root();
+ if (IS_ERR(dentry))
+ return ERR_CAST(dentry);
+
+ inode = d_inode(dentry);
+
+ ei = alloc_root_ei(name);
+ if (!ei)
+ goto fail;
+
+ rei = get_root_inode(ei);
+ rei->events_dir = dentry;
+
+ ei->entries = entries;
+ ei->nr_entries = size;
+ ei->data = data;
+
+ INIT_LIST_HEAD(&ei->children);
+ INIT_LIST_HEAD(&ei->list);
+
+ ti = get_tracefs(inode);
+ ti->flags |= TRACEFS_EVENT_INODE;
+ ti->private = ei;
+
+ inode->i_op = &eventfs_dir_inode_operations;
+ inode->i_fop = &eventfs_file_operations;
+
+ dentry->d_fsdata = get_ei(ei);
+
+ return ei;
+
+ fail:
+ cleanup_ei(ei);
+ dput(dentry);
+ eventfs_ro_put_root();
+ return ERR_PTR(-ENOMEM);
+}
+
/**
* eventfs_remove_rec - remove eventfs dir or file from list
* @ei: eventfs_inode to be removed.
diff --git a/fs/tracefs/inode.c b/fs/tracefs/inode.c
index f3d6188a3b7b..fd064d79d940 100644
--- a/fs/tracefs/inode.c
+++ b/fs/tracefs/inode.c
@@ -30,6 +30,9 @@ static struct vfsmount *tracefs_mount;
static int tracefs_mount_count;
static bool tracefs_registered;
+static struct vfsmount *eventfs_ro_mount;
+static int eventfs_ro_mount_count;
+
/*
* Keep track of all tracefs_inodes in order to update their
* flags if necessary on a remount.
@@ -423,6 +426,14 @@ static const struct super_operations tracefs_super_operations = {
.show_options = tracefs_show_options,
};
+static const struct super_operations eventfs_ro_super_operations = {
+ .alloc_inode = tracefs_alloc_inode,
+ .free_inode = tracefs_free_inode,
+ .destroy_inode = tracefs_destroy_inode,
+ .drop_inode = tracefs_drop_inode,
+ .statfs = simple_statfs,
+};
+
/*
* It would be cleaner if eventfs had its own dentry ops.
*
@@ -523,6 +534,79 @@ static struct file_system_type trace_fs_type = {
};
MODULE_ALIAS_FS("tracefs");
+static int eventfs_ro_fill_super(struct super_block *sb, struct fs_context *fc)
+{
+ struct inode *inode;
+ struct dentry *root;
+
+ sb->s_blocksize = PAGE_SIZE;
+ sb->s_blocksize_bits = PAGE_SHIFT;
+ sb->s_magic = EVENTFS_SUPER_MAGIC;
+ sb->s_op = &eventfs_ro_super_operations;
+ sb->s_time_gran = 1;
+ sb->s_flags |= SB_RDONLY;
+
+ inode = new_inode(sb);
+ if (!inode)
+ return -ENOMEM;
+
+ inode->i_ino = 1;
+ inode->i_mode = S_IFDIR | 0555;
+ simple_inode_init_ts(inode);
+ inode->i_op = &simple_dir_inode_operations;
+ inode->i_fop = &simple_dir_operations;
+ set_nlink(inode, 2);
+
+ set_default_d_op(sb, &tracefs_dentry_operations);
+
+ root = d_make_root(inode);
+ if (!root)
+ return -ENOMEM;
+
+ sb->s_root = root;
+
+ return 0;
+}
+
+static int eventfs_ro_get_tree(struct fs_context *fc)
+{
+ return get_tree_single(fc, eventfs_ro_fill_super);
+}
+
+static const struct fs_context_operations eventfs_ro_context_ops = {
+ .get_tree = eventfs_ro_get_tree,
+};
+
+static int eventfs_ro_init_fs_context(struct fs_context *fc)
+{
+ fc->ops = &eventfs_ro_context_ops;
+ return 0;
+}
+
+static struct file_system_type eventfs_ro_fs_type = {
+ .owner = THIS_MODULE,
+ .name = "eventfs",
+ .init_fs_context = eventfs_ro_init_fs_context,
+ .kill_sb = kill_anon_super,
+};
+
+struct dentry *eventfs_ro_get_root(void)
+{
+ int error;
+
+ error = simple_pin_fs(&eventfs_ro_fs_type, &eventfs_ro_mount,
+ &eventfs_ro_mount_count);
+ if (error)
+ return ERR_PTR(error);
+
+ return dget(eventfs_ro_mount->mnt_root);
+}
+
+void eventfs_ro_put_root(void)
+{
+ simple_release_fs(&eventfs_ro_mount, &eventfs_ro_mount_count);
+}
+
struct dentry *tracefs_start_creating(const char *name, struct dentry *parent)
{
struct dentry *dentry;
@@ -801,8 +885,15 @@ static int __init tracefs_init(void)
return -EINVAL;
retval = register_filesystem(&trace_fs_type);
- if (!retval)
- tracefs_registered = true;
+ if (retval)
+ return retval;
+ tracefs_registered = true;
+
+ retval = sysfs_create_mount_point(kernel_kobj, "events");
+ if (retval)
+ return retval;
+
+ retval = register_filesystem(&eventfs_ro_fs_type);
return retval;
}
diff --git a/fs/tracefs/internal.h b/fs/tracefs/internal.h
index a4a7f8431aff..0440413f959b 100644
--- a/fs/tracefs/internal.h
+++ b/fs/tracefs/internal.h
@@ -73,6 +73,9 @@ struct dentry *tracefs_end_creating(struct dentry *dentry);
struct dentry *tracefs_failed_creating(struct dentry *dentry);
struct inode *tracefs_get_inode(struct super_block *sb);
+struct dentry *eventfs_ro_get_root(void);
+void eventfs_ro_put_root(void);
+
void eventfs_remount(struct tracefs_inode *ti, bool update_uid, bool update_gid);
void eventfs_d_release(struct dentry *dentry);
diff --git a/include/linux/trace_events.h b/include/linux/trace_events.h
index 308c76b57d13..957695fbb015 100644
--- a/include/linux/trace_events.h
+++ b/include/linux/trace_events.h
@@ -648,6 +648,7 @@ struct trace_event_file {
struct trace_event_call *event_call;
struct event_filter __rcu *filter;
struct eventfs_inode *ei;
+ struct eventfs_inode *ei_ro;
struct trace_array *tr;
struct trace_subsystem_dir *system;
struct list_head triggers;
diff --git a/include/linux/tracefs.h b/include/linux/tracefs.h
index bc354d340046..c175efc51d20 100644
--- a/include/linux/tracefs.h
+++ b/include/linux/tracefs.h
@@ -87,6 +87,10 @@ struct eventfs_inode *eventfs_create_dir(const char *name, struct eventfs_inode
const struct eventfs_entry *entries,
int size, void *data);
+struct eventfs_inode *eventfs_create_events_dir_ro(const char *name,
+ const struct eventfs_entry *entries,
+ int size, void *data);
+
void eventfs_remove_events_dir(struct eventfs_inode *ei);
void eventfs_remove_dir(struct eventfs_inode *ei);
diff --git a/include/uapi/linux/magic.h b/include/uapi/linux/magic.h
index 4f2da935a76c..7cf8f1a1ae38 100644
--- a/include/uapi/linux/magic.h
+++ b/include/uapi/linux/magic.h
@@ -75,6 +75,7 @@
#define STACK_END_MAGIC 0x57AC6E9D
#define TRACEFS_MAGIC 0x74726163
+#define EVENTFS_SUPER_MAGIC 0x65766673
#define V9FS_MAGIC 0x01021997
diff --git a/kernel/trace/trace.h b/kernel/trace/trace.h
index 80fe152af1dd..00c35aaa5069 100644
--- a/kernel/trace/trace.h
+++ b/kernel/trace/trace.h
@@ -416,6 +416,7 @@ struct trace_array {
struct dentry *options;
struct dentry *percpu_dir;
struct eventfs_inode *event_dir;
+ struct eventfs_inode *event_dir_ro;
struct trace_options *topts;
struct list_head systems;
struct list_head events;
@@ -1604,6 +1605,7 @@ struct trace_subsystem_dir {
struct event_subsystem *subsystem;
struct trace_array *tr;
struct eventfs_inode *ei;
+ struct eventfs_inode *ei_ro;
int ref_count;
int nr_events;
};
diff --git a/kernel/trace/trace_events.c b/kernel/trace/trace_events.c
index ddb6932a3ee7..9662cb24a92c 100644
--- a/kernel/trace/trace_events.c
+++ b/kernel/trace/trace_events.c
@@ -1279,6 +1279,7 @@ static void remove_subsystem(struct trace_subsystem_dir *dir)
if (!--dir->nr_events) {
eventfs_remove_dir(dir->ei);
+ eventfs_remove_dir(dir->ei_ro);
list_del(&dir->list);
__put_system_dir(dir);
}
@@ -1308,6 +1309,7 @@ void event_file_put(struct trace_event_file *file)
static void remove_event_file_dir(struct trace_event_file *file)
{
eventfs_remove_dir(file->ei);
+ eventfs_remove_dir(file->ei_ro);
list_del(&file->list);
remove_subsystem(file->system);
free_event_filter(file->filter);
@@ -2986,6 +2988,7 @@ event_subsystem_dir(struct trace_array *tr, const char *name,
}
dir->ei = ei;
+ dir->ei_ro = NULL;
dir->tr = tr;
dir->ref_count = 1;
dir->nr_events = 1;
@@ -3126,6 +3129,33 @@ static void event_release(const char *name, void *data)
event_file_put(file);
}
+static int event_callback_ro(const char *name, umode_t *mode, void **data,
+ const struct file_operations **fops)
+{
+ int ret = event_callback(name, mode, data, fops);
+
+ /* Skip writable entries in the read-only tree */
+ if (ret && (*mode & 0222))
+ return 0;
+ if (ret)
+ *mode = 0444;
+ return ret;
+}
+
+static struct eventfs_entry event_entries_ro[] = {
+ {
+ .name = "format",
+ .callback = event_callback_ro,
+ .release = event_release,
+ },
+#ifdef CONFIG_PERF_EVENTS
+ {
+ .name = "id",
+ .callback = event_callback_ro,
+ },
+#endif
+};
+
static int
event_create_dir(struct eventfs_inode *parent, struct trace_event_file *file)
{
@@ -3218,6 +3248,28 @@ event_create_dir(struct eventfs_inode *parent, struct trace_event_file *file)
/* Gets decremented on freeing of the "enable" file */
event_file_get(file);
+ /* Create read only eventfs directory */
+ if (!(call->flags & TRACE_EVENT_FL_DYNAMIC) &&
+ !IS_ERR_OR_NULL(tr->event_dir_ro)) {
+ struct trace_subsystem_dir *sdir = file->system;
+
+ if (!sdir->ei_ro) {
+ sdir->ei_ro = eventfs_create_dir(call->class->system,
+ tr->event_dir_ro, NULL, 0, sdir);
+ if (IS_ERR(sdir->ei_ro))
+ sdir->ei_ro = NULL;
+ }
+ if (sdir->ei_ro) {
+ file->ei_ro = eventfs_create_dir(name, sdir->ei_ro,
+ event_entries_ro,
+ ARRAY_SIZE(event_entries_ro), file);
+ if (IS_ERR(file->ei_ro))
+ file->ei_ro = NULL;
+ else
+ event_file_get(file);
+ }
+ }
+
return 0;
}
@@ -4539,10 +4591,35 @@ static int events_callback(const char *name, umode_t *mode, void **data,
return 1;
}
+static int events_callback_ro(const char *name, umode_t *mode, void **data,
+ const struct file_operations **fops)
+{
+ int ret = events_callback(name, mode, data, fops);
+
+ /* Skip writable entries in the read-only tree */
+ if (ret && (*mode & 0222))
+ return 0;
+ if (ret)
+ *mode = 0444;
+ return ret;
+}
+
+static struct eventfs_entry events_entries_ro[] = {
+ {
+ .name = "header_page",
+ .callback = events_callback_ro,
+ },
+ {
+ .name = "header_event",
+ .callback = events_callback_ro,
+ },
+};
+
/* Expects to have event_mutex held when called */
static int
create_event_toplevel_files(struct dentry *parent, struct trace_array *tr)
{
+ static bool event_dir_ro_created;
struct eventfs_inode *e_events;
struct dentry *entry;
int nr_entries;
@@ -4596,6 +4673,16 @@ create_event_toplevel_files(struct dentry *parent, struct trace_array *tr)
tr->event_dir = e_events;
+ if (!event_dir_ro_created && (tr->flags & TRACE_ARRAY_FL_GLOBAL)) {
+ tr->event_dir_ro = eventfs_create_events_dir_ro(
+ "events", events_entries_ro,
+ ARRAY_SIZE(events_entries_ro), tr);
+ if (IS_ERR(tr->event_dir_ro))
+ tr->event_dir_ro = NULL;
+ else
+ event_dir_ro_created = true;
+ }
+
return 0;
}
--
2.54.0
^ permalink raw reply related
* [PATCH v5 1/5] eventfs: define event fields before directory creation
From: Anubhav Shelat @ 2026-07-14 18:31 UTC (permalink / raw)
To: mpetlan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, linux-trace-kernel
Cc: Anubhav Shelat
In-Reply-To: <20260714183150.292861-2-ashelat@redhat.com>
Move the event_define_fields() call in event_create_dir() before the
eventfs directory creation. Previously, a failure after directory
creation wouldn't clean up eventfs_inode because the error path didn't
call eventfs_remove_dir(). This eliminates the need to clean up the
eventfs directories if event_define_fields() fails.
Signed-off-by: Anubhav Shelat <ashelat@redhat.com>
---
kernel/trace/trace_events.c | 13 +++++++------
1 file changed, 7 insertions(+), 6 deletions(-)
diff --git a/kernel/trace/trace_events.c b/kernel/trace/trace_events.c
index c46e623e7e0d..ddb6932a3ee7 100644
--- a/kernel/trace/trace_events.c
+++ b/kernel/trace/trace_events.c
@@ -3190,6 +3190,13 @@ event_create_dir(struct eventfs_inode *parent, struct trace_event_file *file)
if (WARN_ON_ONCE(strcmp(call->class->system, TRACE_SYSTEM) == 0))
return -ENODEV;
+ ret = event_define_fields(call);
+ if (ret < 0) {
+ pr_warn("Could not initialize trace point events/%s\n",
+ trace_event_name(call));
+ return ret;
+ }
+
e_events = event_subsystem_dir(tr, call->class->system, file, parent);
if (!e_events)
return -ENOMEM;
@@ -3208,12 +3215,6 @@ event_create_dir(struct eventfs_inode *parent, struct trace_event_file *file)
file->ei = ei;
- ret = event_define_fields(call);
- if (ret < 0) {
- pr_warn("Could not initialize trace point events/%s\n", name);
- return ret;
- }
-
/* Gets decremented on freeing of the "enable" file */
event_file_get(file);
--
2.54.0
^ permalink raw reply related
* [RFC PATCH 08/13] mm/kwatch: add hardware breakpoint backend
From: Jinchao Wang @ 2026-07-14 18:32 UTC (permalink / raw)
To: Andrew Morton, Peter Zijlstra, Thomas Gleixner, Steven Rostedt,
Masami Hiramatsu
Cc: Ingo Molnar, Borislav Petkov, Dave Hansen, H . Peter Anvin, x86,
Arnaldo Carvalho de Melo, Namhyung Kim, Mark Rutland,
Mathieu Desnoyers, David Hildenbrand, Jonathan Corbet,
Matthew Wilcox, linux-kernel, linux-mm, linux-trace-kernel,
linux-perf-users, linux-doc, Jinchao Wang
In-Reply-To: <20260714182243.10687-1-wangjinchao600@gmail.com>
Manage a preallocated pool of wide (per-CPU) perf hardware
breakpoints. All breakpoints are registered up front against a dummy
address; arming a watchpoint only re-points an already-registered
event, so the arm path can run from a kprobe handler.
- kwatch_hwbp_get()/put() claim and release pool entries with
per-slot cmpxchg, safe for concurrent consumers on any CPU.
- kwatch_hwbp_arm() updates the local CPU synchronously via
modify_wide_hw_breakpoint_local() and broadcasts asynchronous IPIs
to the other CPUs. Arm-side IPIs are rate-limited per CPU; disarm
IPIs are refcounted so an entry is only recycled once every CPU
has dropped it.
- Hits are reported through the kwatch:kwatch_hit tracepoint with a
short stack trace: the ftrace ring buffer is usable from NMI-like
context and survives a subsequent crash, unlike printk.
- A CPU hotplug callback creates/destroys the per-CPU events as CPUs
come and go.
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
include/trace/events/kwatch.h | 57 ++++++
mm/kwatch/Makefile | 2 +-
mm/kwatch/hwbp.c | 358 ++++++++++++++++++++++++++++++++++
3 files changed, 416 insertions(+), 1 deletion(-)
create mode 100644 include/trace/events/kwatch.h
create mode 100644 mm/kwatch/hwbp.c
diff --git a/include/trace/events/kwatch.h b/include/trace/events/kwatch.h
new file mode 100644
index 000000000000..edb95405c386
--- /dev/null
+++ b/include/trace/events/kwatch.h
@@ -0,0 +1,57 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#undef TRACE_SYSTEM
+#define TRACE_SYSTEM kwatch
+
+#if !defined(_TRACE_KWATCH_H) || defined(TRACE_HEADER_MULTI_READ)
+#define _TRACE_KWATCH_H
+
+#include <linux/tracepoint.h>
+#include <linux/ptrace.h>
+
+#define KWATCH_STACK_DEPTH 8
+
+struct trace_seq;
+const char *kwatch_trace_print_stack(struct trace_seq *p,
+ const unsigned long *stack,
+ unsigned int nr);
+
+TRACE_EVENT(kwatch_hit,
+ TP_PROTO(unsigned long ip, unsigned long sp, unsigned long addr,
+ u64 time_ns,
+ unsigned long *stack_entries, unsigned int stack_nr),
+ TP_ARGS(ip, sp, addr, time_ns, stack_entries, stack_nr),
+
+ TP_STRUCT__entry(
+ __field(unsigned long, ip)
+ __field(unsigned long, sp)
+ __field(unsigned long, addr)
+ __field(u64, time_ns)
+ __field(unsigned int, stack_nr)
+ __array(unsigned long, stack, KWATCH_STACK_DEPTH)
+ ),
+
+ TP_fast_assign(
+ unsigned int i;
+
+ __entry->ip = ip;
+ __entry->sp = sp;
+ __entry->addr = addr;
+ __entry->time_ns = time_ns;
+ __entry->stack_nr = min_t(unsigned int, stack_nr,
+ KWATCH_STACK_DEPTH);
+ for (i = 0; i < __entry->stack_nr; i++)
+ __entry->stack[i] = stack_entries[i];
+ ),
+
+ TP_printk("KWatch HIT: time=%llu.%06lu ip=%pS addr=0x%lx%s",
+ __entry->time_ns / 1000000000ULL,
+ (unsigned long)((__entry->time_ns / 1000ULL) % 1000000ULL),
+ (void *)__entry->ip, __entry->addr,
+ kwatch_trace_print_stack(p, __entry->stack,
+ __entry->stack_nr))
+);
+
+#endif /* _TRACE_KWATCH_H */
+
+/* This part must be outside protection */
+#include <trace/define_trace.h>
diff --git a/mm/kwatch/Makefile b/mm/kwatch/Makefile
index cc6574df0d68..b2bc3003c89b 100644
--- a/mm/kwatch/Makefile
+++ b/mm/kwatch/Makefile
@@ -1,3 +1,3 @@
obj-$(CONFIG_KWATCH) += kwatch.o
-kwatch-y := deref.o task_ctx.o
+kwatch-y := deref.o task_ctx.o hwbp.o
diff --git a/mm/kwatch/hwbp.c b/mm/kwatch/hwbp.c
new file mode 100644
index 000000000000..19498ba03826
--- /dev/null
+++ b/mm/kwatch/hwbp.c
@@ -0,0 +1,358 @@
+// SPDX-License-Identifier: GPL-2.0
+#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
+
+#include <linux/cpuhotplug.h>
+#include <linux/ftrace.h>
+#include <linux/hw_breakpoint.h>
+#include <linux/sched/clock.h>
+#include <linux/irqflags.h>
+#include <linux/kallsyms.h>
+#include <linux/mutex.h>
+#include <linux/printk.h>
+#include <linux/slab.h>
+#include <linux/stacktrace.h>
+#include <linux/trace_seq.h>
+#include <linux/workqueue.h>
+
+#include "kwatch.h"
+
+static LIST_HEAD(kwatch_all_wp_list);
+static struct kwatch_watchpoint **kwatch_wp_slots;
+static u16 kwatch_wp_nr;
+static DEFINE_MUTEX(kwatch_all_wp_mutex);
+static unsigned long kwatch_dummy_holder __aligned(8);
+static int kwatch_hwbp_cpuhp_state = CPUHP_INVALID;
+
+#define CREATE_TRACE_POINTS
+#include <trace/events/kwatch.h>
+
+/*
+ * Render the saved stack like the ftrace built-in stacktrace / dump_stack()
+ * style. Symbol resolution runs at trace read time, not in the hit path.
+ */
+const char *kwatch_trace_print_stack(struct trace_seq *p,
+ const unsigned long *stack,
+ unsigned int nr)
+{
+ const char *ret = trace_seq_buffer_ptr(p);
+ unsigned int i;
+
+ for (i = 0; i < nr; i++)
+ trace_seq_printf(p, "\n => %pS", (void *)stack[i]);
+ trace_seq_putc(p, 0);
+ return ret;
+}
+
+static void kwatch_hwbp_handler(struct perf_event *bp,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct kwatch_watchpoint *wp = bp->overflow_handler_context;
+ unsigned long stack_entries[KWATCH_STACK_DEPTH];
+ unsigned int stack_nr;
+
+ if (!kwatch_probe_validate_hit(regs, wp->arm_tsk))
+ return;
+
+ stack_nr = stack_trace_save_regs(regs, stack_entries, KWATCH_STACK_DEPTH, 2);
+ trace_kwatch_hit(instruction_pointer(regs), kernel_stack_pointer(regs),
+ wp->attr.bp_addr, local_clock(),
+ stack_entries, stack_nr);
+}
+
+static void kwatch_hwbp_arm_local(void *info)
+{
+ struct kwatch_watchpoint *wp = info;
+ struct perf_event *bp;
+ unsigned long flags;
+ int cpu, err;
+
+ local_irq_save(flags);
+
+ cpu = smp_processor_id();
+ bp = per_cpu(*wp->event, cpu);
+
+ if (unlikely(!bp))
+ goto out;
+
+ kwatch_probe_mute(true);
+ barrier();
+
+ err = modify_wide_hw_breakpoint_local(bp, &wp->attr);
+ if (unlikely(err)) {
+ WARN_ONCE(1,
+ "KWatch: HWBP reinstall failed on CPU%d (err=%d, addr=0x%llx, len=%llu)\n",
+ cpu, err, wp->attr.bp_addr, wp->attr.bp_len);
+ }
+
+ barrier();
+ kwatch_probe_mute(false);
+
+out:
+ local_irq_restore(flags);
+}
+
+static inline void kwatch_hwbp_try_recycle(struct kwatch_watchpoint *wp)
+{
+ if (atomic_dec_and_test(&wp->pending_ipis)) {
+ if (!READ_ONCE(wp->teardown))
+ atomic_set_release(&wp->in_use, 0);
+
+ atomic_dec(&wp->refcount);
+ }
+}
+
+static void kwatch_hwbp_disarm_local(void *info)
+{
+ struct kwatch_watchpoint *wp = info;
+
+ kwatch_hwbp_arm_local(info);
+ kwatch_hwbp_try_recycle(wp);
+}
+
+static int kwatch_hwbp_cpu_online(unsigned int cpu)
+{
+ struct perf_event_attr attr;
+ struct kwatch_watchpoint *wp;
+ struct perf_event *bp;
+
+ mutex_lock(&kwatch_all_wp_mutex);
+ list_for_each_entry(wp, &kwatch_all_wp_list, list) {
+ attr = wp->attr;
+ attr.bp_addr = (unsigned long)&kwatch_dummy_holder;
+ bp = perf_event_create_kernel_counter(&attr, cpu, NULL,
+ kwatch_hwbp_handler, wp);
+ if (IS_ERR(bp)) {
+ pr_warn("%s failed to create watch on CPU %d: %ld\n",
+ __func__, cpu, PTR_ERR(bp));
+ continue;
+ }
+ per_cpu(*wp->event, cpu) = bp;
+ }
+ mutex_unlock(&kwatch_all_wp_mutex);
+ return 0;
+}
+
+static int kwatch_hwbp_cpu_offline(unsigned int cpu)
+{
+ struct kwatch_watchpoint *wp;
+ struct perf_event *bp;
+
+ mutex_lock(&kwatch_all_wp_mutex);
+ list_for_each_entry(wp, &kwatch_all_wp_list, list) {
+ bp = per_cpu(*wp->event, cpu);
+ if (bp) {
+ unregister_hw_breakpoint(bp);
+ per_cpu(*wp->event, cpu) = NULL;
+ }
+ }
+ mutex_unlock(&kwatch_all_wp_mutex);
+ return 0;
+}
+
+int kwatch_hwbp_get(struct kwatch_watchpoint **out_wp)
+{
+ struct kwatch_watchpoint *wp;
+ int i;
+
+ /*
+ * Per-slot cmpxchg claim: safe for concurrent consumers on any CPU,
+ * unlike llist_del_first() which requires a single consumer.
+ */
+ for (i = 0; i < kwatch_wp_nr; i++) {
+ wp = kwatch_wp_slots[i];
+ if (atomic_read(&wp->in_use))
+ continue;
+ if (atomic_cmpxchg(&wp->in_use, 0, 1) == 0) {
+ atomic_inc(&wp->refcount);
+ *out_wp = wp;
+ return 0;
+ }
+ }
+ return -EBUSY;
+}
+
+void kwatch_hwbp_arm(struct kwatch_watchpoint *wp, unsigned long addr, u16 len)
+{
+ static DEFINE_PER_CPU(u64, last_ipi_time);
+ int cur_cpu;
+ call_single_data_t *csd;
+ int cpu;
+ bool is_disarm = (addr == (unsigned long)&kwatch_dummy_holder);
+
+ wp->attr.bp_addr = addr;
+ wp->attr.bp_len = len;
+
+ if (!is_disarm)
+ wp->arm_tsk = current;
+
+ /* ensure attr update visible to other cpu before sending IPI */
+ smp_wmb();
+
+ atomic_set(&wp->pending_ipis, 1);
+ cur_cpu = get_cpu();
+
+ if (!is_disarm) {
+ u64 now = local_clock();
+ u64 last = this_cpu_read(last_ipi_time);
+
+ if (now - last < 1000000ULL) {
+ put_cpu();
+ return;
+ }
+ this_cpu_write(last_ipi_time, now);
+ }
+ for_each_online_cpu(cpu) {
+ if (cpu == cur_cpu)
+ continue;
+
+ if (is_disarm)
+ atomic_inc(&wp->pending_ipis);
+
+ csd = per_cpu_ptr(is_disarm ? wp->csd_disarm : wp->csd_arm,
+ cpu);
+ if (smp_call_function_single_async(cpu, csd) && is_disarm)
+ kwatch_hwbp_try_recycle(wp);
+ }
+ put_cpu();
+
+ if (is_disarm)
+ kwatch_hwbp_disarm_local(wp);
+ else
+ kwatch_hwbp_arm_local(wp);
+}
+
+int kwatch_hwbp_put(struct kwatch_watchpoint *wp)
+{
+ kwatch_hwbp_arm(wp, (unsigned long)&kwatch_dummy_holder,
+ sizeof(unsigned long));
+
+ return 0;
+}
+
+void kwatch_hwbp_free(void)
+{
+ struct kwatch_watchpoint *wp, *tmp;
+
+ kwatch_wp_nr = 0;
+ kfree(kwatch_wp_slots);
+ kwatch_wp_slots = NULL;
+
+ if (kwatch_hwbp_cpuhp_state != CPUHP_INVALID) {
+ cpuhp_remove_state_nocalls(kwatch_hwbp_cpuhp_state);
+ kwatch_hwbp_cpuhp_state = CPUHP_INVALID;
+ }
+
+ mutex_lock(&kwatch_all_wp_mutex);
+ list_for_each_entry_safe(wp, tmp, &kwatch_all_wp_list, list) {
+ list_del(&wp->list);
+
+ WRITE_ONCE(wp->teardown, true);
+ atomic_dec(&wp->refcount);
+
+ /* Wait for all async IPIs to finish */
+ while (atomic_read(&wp->refcount) > 0)
+ cpu_relax();
+
+ unregister_wide_hw_breakpoint(wp->event);
+ free_percpu(wp->csd_arm);
+ free_percpu(wp->csd_disarm);
+ kfree(wp);
+ }
+ mutex_unlock(&kwatch_all_wp_mutex);
+}
+
+int kwatch_hwbp_prealloc(u16 max_watch, enum kwatch_access_type access_type)
+{
+ struct kwatch_watchpoint *wp;
+ int success = 0, cpu;
+ u32 bp_type;
+ int ret;
+
+ switch (access_type) {
+ case KWATCH_ACCESS_X:
+ bp_type = HW_BREAKPOINT_X;
+ break;
+ case KWATCH_ACCESS_R:
+ bp_type = HW_BREAKPOINT_R;
+ break;
+ case KWATCH_ACCESS_RW:
+ bp_type = HW_BREAKPOINT_RW;
+ break;
+ case KWATCH_ACCESS_W:
+ default:
+ bp_type = HW_BREAKPOINT_W;
+ break;
+ }
+
+ while (!max_watch || success < max_watch) {
+ wp = kzalloc_obj(*wp);
+ if (!wp)
+ break;
+
+ wp->csd_arm = alloc_percpu(call_single_data_t);
+ wp->csd_disarm = alloc_percpu(call_single_data_t);
+ if (!wp->csd_arm || !wp->csd_disarm) {
+ free_percpu(wp->csd_arm);
+ free_percpu(wp->csd_disarm);
+ kfree(wp);
+ break;
+ }
+
+ for_each_possible_cpu(cpu) {
+ INIT_CSD(per_cpu_ptr(wp->csd_arm, cpu),
+ kwatch_hwbp_arm_local, wp);
+ INIT_CSD(per_cpu_ptr(wp->csd_disarm, cpu),
+ kwatch_hwbp_disarm_local, wp);
+ }
+
+ wp->teardown = false;
+
+ hw_breakpoint_init(&wp->attr);
+ wp->attr.bp_addr = (unsigned long)&kwatch_dummy_holder;
+ wp->attr.bp_len = sizeof(unsigned long);
+ wp->attr.bp_type = bp_type;
+
+ wp->event = register_wide_hw_breakpoint(&wp->attr,
+ kwatch_hwbp_handler,
+ wp);
+ if (IS_ERR((void *)wp->event)) {
+ free_percpu(wp->csd_arm);
+ free_percpu(wp->csd_disarm);
+ kfree(wp);
+ break;
+ }
+
+ atomic_set(&wp->refcount, 1);
+
+ mutex_lock(&kwatch_all_wp_mutex);
+ list_add(&wp->list, &kwatch_all_wp_list);
+ mutex_unlock(&kwatch_all_wp_mutex);
+ success++;
+ }
+
+ if (!success)
+ return -EBUSY;
+
+ kwatch_wp_slots = kcalloc(success, sizeof(*kwatch_wp_slots),
+ GFP_KERNEL);
+ if (!kwatch_wp_slots) {
+ kwatch_hwbp_free();
+ return -ENOMEM;
+ }
+ mutex_lock(&kwatch_all_wp_mutex);
+ list_for_each_entry(wp, &kwatch_all_wp_list, list)
+ kwatch_wp_slots[kwatch_wp_nr++] = wp;
+ mutex_unlock(&kwatch_all_wp_mutex);
+
+ ret = cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN, "kwatch:online",
+ kwatch_hwbp_cpu_online,
+ kwatch_hwbp_cpu_offline);
+ if (ret < 0) {
+ kwatch_hwbp_free();
+ return ret;
+ }
+
+ kwatch_hwbp_cpuhp_state = ret;
+ return 0;
+}
--
2.53.0
^ permalink raw reply related
page: next (older) | prev (newer) | latest
- recent:[subjects (threaded)|topics (new)|topics (active)]
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox