* [PATCH v12 01/11] x86/hw_breakpoints: Make DR7 updates NMI safe
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
@ 2026-08-07 15:33 ` Masami Hiramatsu (Google)
2026-08-07 16:01 ` sashiko-bot
2026-08-07 15:33 ` [PATCH v12 02/11] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
` (9 subsequent siblings)
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:33 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Jinchao Wang <wangjinchao600@gmail.com>
Hardware breakpoint installation and removal run with IRQs disabled, but
an NMI can still enter the same code through KGDB. The interrupted
operation and the NMI can consequently claim the same slot or overwrite
each other's DR7 state.
Claim and release per-CPU slots with cmpxchg. Update cpu_dr7 with
single-instruction per-CPU operations, and preserve hardware-first
disable and hardware-last enable ordering. Add a per-CPU sequence number
so interrupted DR7 writers and restore paths detect an NMI update and
retry from the latest shadow state.
Link: https://lore.kernel.org/all/4ee0a2efc9e8387af83286b8495b7d490247e165.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v12:
- Use stack variables dr7 and dr7_seq directly in exc_nmi() and remove
unnecessary per-CPU nmi_dr7 and nmi_dr7_seq variables.
---
arch/x86/include/asm/debugreg.h | 36 ++++++++++----
arch/x86/kernel/cpu/mce/core.c | 20 +++++---
arch/x86/kernel/hw_breakpoint.c | 98 +++++++++++++++++++--------------------
arch/x86/kernel/nmi.c | 7 ++-
arch/x86/kernel/traps.c | 10 +++-
5 files changed, 97 insertions(+), 74 deletions(-)
diff --git a/arch/x86/include/asm/debugreg.h b/arch/x86/include/asm/debugreg.h
index a2c1f2d24b64..b1fe1c47978d 100644
--- a/arch/x86/include/asm/debugreg.h
+++ b/arch/x86/include/asm/debugreg.h
@@ -18,6 +18,7 @@
#define DR7_FIXED_1 0x00000400
DECLARE_PER_CPU(unsigned long, cpu_dr7);
+DECLARE_PER_CPU(unsigned int, cpu_dr7_seq);
#ifndef CONFIG_PARAVIRT_XXL
/*
@@ -125,18 +126,20 @@ static __always_inline bool hw_breakpoint_active(void)
extern void hw_breakpoint_restore(void);
-static __always_inline unsigned long local_db_save(void)
+static __always_inline void local_db_save(unsigned long *dr7,
+ unsigned int *dr7_seq)
{
- unsigned long dr7;
+ *dr7 = 0;
+ *dr7_seq = this_cpu_read(cpu_dr7_seq);
if (static_cpu_has(X86_FEATURE_HYPERVISOR) && !hw_breakpoint_active())
- return 0;
+ return;
- get_debugreg(dr7, 7);
+ get_debugreg(*dr7, 7);
/* Architecturally set bit */
- dr7 &= ~DR7_FIXED_1;
- if (dr7)
+ *dr7 &= ~DR7_FIXED_1;
+ if (*dr7)
set_debugreg(DR7_FIXED_1, 7);
/*
@@ -145,20 +148,33 @@ static __always_inline unsigned long local_db_save(void)
* be good.
*/
barrier();
-
- return dr7;
}
-static __always_inline void local_db_restore(unsigned long dr7)
+static __always_inline void local_db_restore(unsigned long dr7,
+ unsigned int dr7_seq)
{
+ unsigned int seq;
+
/*
* Ensure the compiler doesn't raise this statement into
* the critical section; enabling breakpoints early would
* not be good.
*/
barrier();
- if (dr7)
+
+ do {
+ seq = this_cpu_read(cpu_dr7_seq);
+ if (seq == dr7_seq) {
+ if (!dr7)
+ return;
+ } else {
+ dr7 = this_cpu_read(cpu_dr7);
+ if (!dr7)
+ dr7 = DR7_FIXED_1;
+ }
+
set_debugreg(dr7, 7);
+ } while (unlikely(seq != this_cpu_read(cpu_dr7_seq)));
}
#ifdef CONFIG_CPU_SUP_AMD
diff --git a/arch/x86/kernel/cpu/mce/core.c b/arch/x86/kernel/cpu/mce/core.c
index 9bba1e2f03af..8dba9cd04bfa 100644
--- a/arch/x86/kernel/cpu/mce/core.c
+++ b/arch/x86/kernel/cpu/mce/core.c
@@ -2139,20 +2139,22 @@ static __always_inline void exc_machine_check_user(struct pt_regs *regs)
DEFINE_IDTENTRY_MCE(exc_machine_check)
{
unsigned long dr7;
+ unsigned int dr7_seq;
- dr7 = local_db_save();
+ local_db_save(&dr7, &dr7_seq);
exc_machine_check_kernel(regs);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
/* The user mode variant. */
DEFINE_IDTENTRY_MCE_USER(exc_machine_check)
{
unsigned long dr7;
+ unsigned int dr7_seq;
- dr7 = local_db_save();
+ local_db_save(&dr7, &dr7_seq);
exc_machine_check_user(regs);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
#ifdef CONFIG_X86_FRED
@@ -2170,13 +2172,14 @@ DEFINE_IDTENTRY_MCE_USER(exc_machine_check)
DEFINE_FREDENTRY_MCE(exc_machine_check)
{
unsigned long dr7;
+ unsigned int dr7_seq;
- dr7 = local_db_save();
+ local_db_save(&dr7, &dr7_seq);
if (user_mode(regs))
exc_machine_check_user(regs);
else
exc_machine_check_kernel(regs);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
#endif
#else
@@ -2184,13 +2187,14 @@ DEFINE_FREDENTRY_MCE(exc_machine_check)
DEFINE_IDTENTRY_RAW(exc_machine_check)
{
unsigned long dr7;
+ unsigned int dr7_seq;
- dr7 = local_db_save();
+ local_db_save(&dr7, &dr7_seq);
if (user_mode(regs))
exc_machine_check_user(regs);
else
exc_machine_check_kernel(regs);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
#endif
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index f846c15f21ca..9ef24b55737f 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -40,6 +40,9 @@
DEFINE_PER_CPU(unsigned long, cpu_dr7);
EXPORT_PER_CPU_SYMBOL(cpu_dr7);
+/* Sequence number of the per-CPU DR7 state. */
+DEFINE_PER_CPU(unsigned int, cpu_dr7_seq);
+
/* Per cpu debug address registers values */
static DEFINE_PER_CPU(unsigned long, cpu_debugreg[HBP_NUM]);
@@ -97,38 +100,30 @@ int decode_dr7(unsigned long dr7, int bpnum, unsigned *len, unsigned *type)
int arch_install_hw_breakpoint(struct perf_event *bp)
{
struct arch_hw_breakpoint *info = counter_arch_bp(bp);
- unsigned long *dr7;
+ unsigned int seq;
int i;
lockdep_assert_irqs_disabled();
for (i = 0; i < HBP_NUM; i++) {
- struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
-
- if (!*slot) {
- *slot = bp;
+ if (!this_cpu_cmpxchg(bp_per_reg[i], NULL, bp))
break;
- }
}
if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
return -EBUSY;
- set_debugreg(info->address, i);
- __this_cpu_write(cpu_debugreg[i], info->address);
-
- dr7 = this_cpu_ptr(&cpu_dr7);
- *dr7 |= encode_dr7(i, info->len, info->type);
-
- /*
- * Ensure we first write cpu_dr7 before we set the DR7 register.
- * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
- */
- barrier();
-
- set_debugreg(*dr7, 7);
- if (info->mask)
- amd_set_dr_addr_mask(info->mask, i);
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ this_cpu_write(cpu_debugreg[i], info->address);
+ barrier();
+ set_debugreg(info->address, i);
+ if (info->mask)
+ amd_set_dr_addr_mask(info->mask, i);
+ this_cpu_or(cpu_dr7, encode_dr7(i, info->len, info->type));
+ barrier();
+ set_debugreg(this_cpu_read(cpu_dr7), 7);
+ } while (seq != this_cpu_read(cpu_dr7_seq));
return 0;
}
@@ -146,36 +141,33 @@ void arch_uninstall_hw_breakpoint(struct perf_event *bp)
{
struct arch_hw_breakpoint *info = counter_arch_bp(bp);
unsigned long dr7;
+ unsigned int seq;
int i;
lockdep_assert_irqs_disabled();
for (i = 0; i < HBP_NUM; i++) {
- struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
-
- if (*slot == bp) {
- *slot = NULL;
+ if (this_cpu_read(bp_per_reg[i]) == bp)
break;
- }
}
if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
return;
- dr7 = this_cpu_read(cpu_dr7);
- dr7 &= ~__encode_dr7(i, info->len, info->type);
-
- set_debugreg(dr7, 7);
- if (info->mask)
- amd_set_dr_addr_mask(0, i);
-
- /*
- * Ensure the write to cpu_dr7 is after we've set the DR7 register.
- * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
- */
- barrier();
-
- this_cpu_write(cpu_dr7, dr7);
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ dr7 = this_cpu_read(cpu_dr7);
+ dr7 &= ~__encode_dr7(i, info->len, info->type);
+ set_debugreg(dr7, 7);
+ if (info->mask)
+ amd_set_dr_addr_mask(0, i);
+ barrier();
+ this_cpu_and(cpu_dr7,
+ ~__encode_dr7(i, info->len, info->type));
+ } while (seq != this_cpu_read(cpu_dr7_seq));
+
+ WARN_ONCE(this_cpu_cmpxchg(bp_per_reg[i], bp, NULL) != bp,
+ "Can't release breakpoint slot");
}
static int arch_bp_generic_len(int x86_len)
@@ -309,13 +301,14 @@ static inline bool within_cpu_entry(unsigned long addr, unsigned long end)
sizeof(struct tlb_state)))
return true;
- /*
- * When in guest (X86_FEATURE_HYPERVISOR), local_db_save()
- * will read per-cpu cpu_dr7 before clear dr7 register.
- */
+ /* local_db_save() reads this state before clearing DR7. */
if (within_area(addr, end, (unsigned long)&per_cpu(cpu_dr7, cpu),
sizeof(cpu_dr7)))
return true;
+ if (within_area(addr, end,
+ (unsigned long)&per_cpu(cpu_dr7_seq, cpu),
+ sizeof(cpu_dr7_seq)))
+ return true;
}
return false;
@@ -483,12 +476,17 @@ void flush_ptrace_hw_breakpoint(struct task_struct *tsk)
void hw_breakpoint_restore(void)
{
- set_debugreg(__this_cpu_read(cpu_debugreg[0]), 0);
- set_debugreg(__this_cpu_read(cpu_debugreg[1]), 1);
- set_debugreg(__this_cpu_read(cpu_debugreg[2]), 2);
- set_debugreg(__this_cpu_read(cpu_debugreg[3]), 3);
- set_debugreg(DR6_RESERVED, 6);
- set_debugreg(__this_cpu_read(cpu_dr7), 7);
+ unsigned int seq;
+
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ set_debugreg(this_cpu_read(cpu_debugreg[0]), 0);
+ set_debugreg(this_cpu_read(cpu_debugreg[1]), 1);
+ set_debugreg(this_cpu_read(cpu_debugreg[2]), 2);
+ set_debugreg(this_cpu_read(cpu_debugreg[3]), 3);
+ set_debugreg(DR6_RESERVED, 6);
+ set_debugreg(this_cpu_read(cpu_dr7), 7);
+ } while (seq != this_cpu_read(cpu_dr7_seq));
}
EXPORT_SYMBOL_FOR_KVM(hw_breakpoint_restore);
diff --git a/arch/x86/kernel/nmi.c b/arch/x86/kernel/nmi.c
index 3c9f60d6ca5a..c3806f7bd3e8 100644
--- a/arch/x86/kernel/nmi.c
+++ b/arch/x86/kernel/nmi.c
@@ -531,11 +531,12 @@ enum nmi_states {
};
static DEFINE_PER_CPU(enum nmi_states, nmi_state);
static DEFINE_PER_CPU(unsigned long, nmi_cr2);
-static DEFINE_PER_CPU(unsigned long, nmi_dr7);
DEFINE_IDTENTRY_RAW(exc_nmi)
{
irqentry_state_t irq_state;
+ unsigned long dr7;
+ unsigned int dr7_seq;
struct nmi_stats *nsp = this_cpu_ptr(&nmi_stats);
/*
@@ -572,7 +573,7 @@ DEFINE_IDTENTRY_RAW(exc_nmi)
*/
sev_es_ist_enter(regs);
- this_cpu_write(nmi_dr7, local_db_save());
+ local_db_save(&dr7, &dr7_seq);
irq_state = irqentry_nmi_enter(regs);
@@ -594,7 +595,7 @@ DEFINE_IDTENTRY_RAW(exc_nmi)
irqentry_nmi_exit(regs, irq_state);
- local_db_restore(this_cpu_read(nmi_dr7));
+ local_db_restore(dr7, dr7_seq);
sev_es_ist_exit();
diff --git a/arch/x86/kernel/traps.c b/arch/x86/kernel/traps.c
index 30aa8369957e..018abe736285 100644
--- a/arch/x86/kernel/traps.c
+++ b/arch/x86/kernel/traps.c
@@ -1231,8 +1231,12 @@ static noinstr void exc_debug_kernel(struct pt_regs *regs, unsigned long dr6)
* it results in an endless recursion and stack overflow. Thus we stay
* with the IDT approach, i.e., save DR7 and disable #DB.
*/
- unsigned long dr7 = local_db_save();
- irqentry_state_t irq_state = irqentry_nmi_enter(regs);
+ unsigned long dr7;
+ unsigned int dr7_seq;
+ irqentry_state_t irq_state;
+
+ local_db_save(&dr7, &dr7_seq);
+ irq_state = irqentry_nmi_enter(regs);
instrumentation_begin();
/*
@@ -1289,7 +1293,7 @@ static noinstr void exc_debug_kernel(struct pt_regs *regs, unsigned long dr6)
instrumentation_end();
irqentry_nmi_exit(regs, irq_state);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
static noinstr void exc_debug_user(struct pt_regs *regs, unsigned long dr6)
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 01/11] x86/hw_breakpoints: Make DR7 updates NMI safe
2026-08-07 15:33 ` [PATCH v12 01/11] x86/hw_breakpoints: Make DR7 updates NMI safe Masami Hiramatsu (Google)
@ 2026-08-07 16:01 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 16:01 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] Missing compiler barrier before the sequence number check in per-CPU seqlock loops allows the compiler to hoist the sequence check, leading to lost hardware state updates during NMI interruption.
--
--- Patch [1]: [PATCH v12 01/11] x86/hw_breakpoints: Make DR7 updates NMI safe ---
commit 585f6ee61d495eb16d59ff031d378a9d8f727fd9
Author: Jinchao Wang <wangjinchao600@gmail.com>
x86/hw_breakpoints: Make DR7 updates NMI safe
Hardware breakpoint installation and removal run with IRQs disabled, but
an NMI can still enter the same code through KGDB. The interrupted
operation and the NMI can consequently claim the same slot or overwrite
each other's DR7 state.
Claim and release per-CPU slots with cmpxchg. Update cpu_dr7 with
single-instruction per-CPU operations, and preserve hardware-first
disable and hardware-last enable ordering. Add a per-CPU sequence number
so interrupted DR7 writers and restore paths detect an NMI update and
retry from the latest shadow state.
Link: https://lore.kernel.org/all/4ee0a2efc9e8387af83286b8495b7d490247e165.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/arch/x86/include/asm/debugreg.h b/arch/x86/include/asm/debugreg.h
> index a2c1f2d24b645..b1fe1c47978d1 100644
> --- a/arch/x86/include/asm/debugreg.h
> +++ b/arch/x86/include/asm/debugreg.h
[ ... ]
> @@ -145,20 +148,33 @@ static __always_inline unsigned long local_db_save(void)
> * be good.
> */
> barrier();
> -
> - return dr7;
> }
>
> -static __always_inline void local_db_restore(unsigned long dr7)
> +static __always_inline void local_db_restore(unsigned long dr7,
> + unsigned int dr7_seq)
> {
> + unsigned int seq;
> +
> /*
> * Ensure the compiler doesn't raise this statement into
> * the critical section; enabling breakpoints early would
> * not be good.
> */
> barrier();
> - if (dr7)
> +
> + do {
> + seq = this_cpu_read(cpu_dr7_seq);
> + if (seq == dr7_seq) {
> + if (!dr7)
> + return;
> + } else {
> + dr7 = this_cpu_read(cpu_dr7);
> + if (!dr7)
> + dr7 = DR7_FIXED_1;
> + }
> +
> set_debugreg(dr7, 7);
> + } while (unlikely(seq != this_cpu_read(cpu_dr7_seq)));
[Severity: High]
Does this sequence check need a barrier() before it?
Without a compiler barrier, could the compiler hoist the
this_cpu_read(cpu_dr7_seq) check above the set_debugreg() update? If an NMI
fires after the hoisted read but before the update completes, the loop might
validate the pre-NMI sequence number and terminate prematurely without retrying,
which could desynchronize the hardware DR7 state and per-CPU shadow state.
> }
>
> diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
> index f846c15f21ca1..9ef24b55737f8 100644
> --- a/arch/x86/kernel/hw_breakpoint.c
> +++ b/arch/x86/kernel/hw_breakpoint.c
[ ... ]
> @@ -97,38 +100,30 @@ int decode_dr7(unsigned long dr7, int bpnum, unsigned *len, unsigned *type)
> int arch_install_hw_breakpoint(struct perf_event *bp)
> {
> struct arch_hw_breakpoint *info = counter_arch_bp(bp);
> - unsigned long *dr7;
> + unsigned int seq;
> int i;
>
> lockdep_assert_irqs_disabled();
>
> for (i = 0; i < HBP_NUM; i++) {
> - struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
> -
> - if (!*slot) {
> - *slot = bp;
> + if (!this_cpu_cmpxchg(bp_per_reg[i], NULL, bp))
> break;
> - }
> }
>
> if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
> return -EBUSY;
>
> - set_debugreg(info->address, i);
> - __this_cpu_write(cpu_debugreg[i], info->address);
> -
> - dr7 = this_cpu_ptr(&cpu_dr7);
> - *dr7 |= encode_dr7(i, info->len, info->type);
> -
> - /*
> - * Ensure we first write cpu_dr7 before we set the DR7 register.
> - * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
> - */
> - barrier();
> -
> - set_debugreg(*dr7, 7);
> - if (info->mask)
> - amd_set_dr_addr_mask(info->mask, i);
> + do {
> + seq = this_cpu_inc_return(cpu_dr7_seq);
> + this_cpu_write(cpu_debugreg[i], info->address);
> + barrier();
> + set_debugreg(info->address, i);
> + if (info->mask)
> + amd_set_dr_addr_mask(info->mask, i);
> + this_cpu_or(cpu_dr7, encode_dr7(i, info->len, info->type));
> + barrier();
> + set_debugreg(this_cpu_read(cpu_dr7), 7);
> + } while (seq != this_cpu_read(cpu_dr7_seq));
[Severity: High]
Is a barrier() also missing here before the sequence check, similar to
local_db_restore(), to prevent the compiler from reordering the sequence
check before the DR7 register update?
>
> return 0;
> }
> @@ -146,36 +141,33 @@ void arch_uninstall_hw_breakpoint(struct perf_event *bp)
> {
> struct arch_hw_breakpoint *info = counter_arch_bp(bp);
> unsigned long dr7;
> + unsigned int seq;
> int i;
>
> lockdep_assert_irqs_disabled();
>
> for (i = 0; i < HBP_NUM; i++) {
> - struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
> -
> - if (*slot == bp) {
> - *slot = NULL;
> + if (this_cpu_read(bp_per_reg[i]) == bp)
> break;
> - }
> }
>
> if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
> return;
>
> - dr7 = this_cpu_read(cpu_dr7);
> - dr7 &= ~__encode_dr7(i, info->len, info->type);
> -
> - set_debugreg(dr7, 7);
> - if (info->mask)
> - amd_set_dr_addr_mask(0, i);
> -
> - /*
> - * Ensure the write to cpu_dr7 is after we've set the DR7 register.
> - * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
> - */
> - barrier();
> -
> - this_cpu_write(cpu_dr7, dr7);
> + do {
> + seq = this_cpu_inc_return(cpu_dr7_seq);
> + dr7 = this_cpu_read(cpu_dr7);
> + dr7 &= ~__encode_dr7(i, info->len, info->type);
> + set_debugreg(dr7, 7);
> + if (info->mask)
> + amd_set_dr_addr_mask(0, i);
> + barrier();
> + this_cpu_and(cpu_dr7,
> + ~__encode_dr7(i, info->len, info->type));
> + } while (seq != this_cpu_read(cpu_dr7_seq));
[Severity: High]
Could the lack of a barrier() here allow the compiler to hoist the sequence
check before the shadow state update (this_cpu_and)?
> +
> + WARN_ONCE(this_cpu_cmpxchg(bp_per_reg[i], bp, NULL) != bp,
> + "Can't release breakpoint slot");
> }
[ ... ]
> @@ -483,12 +476,17 @@ void flush_ptrace_hw_breakpoint(struct task_struct *tsk)
>
> void hw_breakpoint_restore(void)
> {
> - set_debugreg(__this_cpu_read(cpu_debugreg[0]), 0);
> - set_debugreg(__this_cpu_read(cpu_debugreg[1]), 1);
> - set_debugreg(__this_cpu_read(cpu_debugreg[2]), 2);
> - set_debugreg(__this_cpu_read(cpu_debugreg[3]), 3);
> - set_debugreg(DR6_RESERVED, 6);
> - set_debugreg(__this_cpu_read(cpu_dr7), 7);
> + unsigned int seq;
> +
> + do {
> + seq = this_cpu_inc_return(cpu_dr7_seq);
> + set_debugreg(this_cpu_read(cpu_debugreg[0]), 0);
> + set_debugreg(this_cpu_read(cpu_debugreg[1]), 1);
> + set_debugreg(this_cpu_read(cpu_debugreg[2]), 2);
> + set_debugreg(this_cpu_read(cpu_debugreg[3]), 3);
> + set_debugreg(DR6_RESERVED, 6);
> + set_debugreg(this_cpu_read(cpu_dr7), 7);
> + } while (seq != this_cpu_read(cpu_dr7_seq));
[Severity: High]
Does this loop also require a barrier() before the sequence check to prevent
premature termination if an NMI interrupts the register restores?
> }
> EXPORT_SYMBOL_FOR_KVM(hw_breakpoint_restore);
>
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=1
^ permalink raw reply [flat|nested] 19+ messages in thread
* [PATCH v12 02/11] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
2026-08-07 15:33 ` [PATCH v12 01/11] x86/hw_breakpoints: Make DR7 updates NMI safe Masami Hiramatsu (Google)
@ 2026-08-07 15:33 ` Masami Hiramatsu (Google)
2026-08-07 15:50 ` sashiko-bot
2026-08-07 15:33 ` [PATCH v12 03/11] HWBP: Add modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
` (8 subsequent siblings)
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:33 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Jinchao Wang <wangjinchao600@gmail.com>
Wprobe needs to move an active per-CPU watchpoint without releasing and
reserving its hardware slot.
Add arch_modify_local_hw_breakpoint_addr() to find the installed local
slot and update only its address shadow and hardware debug address
register. Publish the shadow first so hw_breakpoint_restore() observes
the new address if an NMI interrupts the update.
The caller must provide an installed local event and a valid address.
Slot ownership, breakpoint type, length, mask and DR7 remain unchanged.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v11:
- Return int error code instead of void.
- Validate the new address using hw_breakpoint_arch_parse() before
updating registers.
- Increment cpu_dr7_seq in a sequence loop for NMI protection.
- Return -ENOENT if the breakpoint slot is not found on the local CPU.
---
arch/x86/include/asm/hw_breakpoint.h | 2 ++
arch/x86/kernel/hw_breakpoint.c | 33 +++++++++++++++++++++++++++++++++
2 files changed, 35 insertions(+)
diff --git a/arch/x86/include/asm/hw_breakpoint.h b/arch/x86/include/asm/hw_breakpoint.h
index 0bc931cd0698..f35ec491f6dd 100644
--- a/arch/x86/include/asm/hw_breakpoint.h
+++ b/arch/x86/include/asm/hw_breakpoint.h
@@ -59,6 +59,8 @@ extern int hw_breakpoint_exceptions_notify(struct notifier_block *unused,
int arch_install_hw_breakpoint(struct perf_event *bp);
+int arch_modify_local_hw_breakpoint_addr(struct perf_event *bp,
+ unsigned long addr);
void arch_uninstall_hw_breakpoint(struct perf_event *bp);
void hw_breakpoint_pmu_read(struct perf_event *bp);
void hw_breakpoint_pmu_unthrottle(struct perf_event *bp);
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index 9ef24b55737f..c89f6e71c3a9 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -128,6 +128,39 @@ int arch_install_hw_breakpoint(struct perf_event *bp)
return 0;
}
+int arch_modify_local_hw_breakpoint_addr(struct perf_event *bp,
+ unsigned long addr)
+{
+ struct arch_hw_breakpoint hw;
+ struct perf_event_attr attr = bp->attr;
+ unsigned int seq;
+ int i, ret;
+
+ lockdep_assert_irqs_disabled();
+
+ attr.bp_addr = addr;
+ ret = hw_breakpoint_arch_parse(bp, &attr, &hw);
+ if (ret)
+ return ret;
+
+ for (i = 0; i < HBP_NUM; i++) {
+ if (this_cpu_read(bp_per_reg[i]) == bp)
+ break;
+ }
+
+ if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
+ return -ENOENT;
+
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ this_cpu_write(cpu_debugreg[i], addr);
+ barrier();
+ set_debugreg(addr, i);
+ } while (seq != this_cpu_read(cpu_dr7_seq));
+
+ return 0;
+}
+
/*
* Uninstall the breakpoint contained in the given counter.
*
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 02/11] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API
2026-08-07 15:33 ` [PATCH v12 02/11] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
@ 2026-08-07 15:50 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 15:50 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] arch_modify_local_hw_breakpoint_addr updates the hardware and shadow registers but fails to update the perf_event's canonical software state, leading to silent hardware reverts and incorrect event samples.
--
--- Patch [2]: [PATCH v12 02/11] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API ---
commit bbc277e839d1fd31bb4bfa9d695721e5b6899bf6
Author: Jinchao Wang <wangjinchao600@gmail.com>
x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API
Wprobe needs to move an active per-CPU watchpoint without releasing and
reserving its hardware slot.
Add arch_modify_local_hw_breakpoint_addr() to find the installed local
slot and update only its address shadow and hardware debug address
register. Publish the shadow first so hw_breakpoint_restore() observes
the new address if an NMI interrupts the update.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
> index 9ef24b55737f8..c89f6e71c3a91 100644
> --- a/arch/x86/kernel/hw_breakpoint.c
> +++ b/arch/x86/kernel/hw_breakpoint.c
[ ... ]
> @@ -128,6 +128,39 @@ int arch_install_hw_breakpoint(struct perf_event *bp)
> return 0;
> }
>
> +int arch_modify_local_hw_breakpoint_addr(struct perf_event *bp,
> + unsigned long addr)
> +{
> + struct arch_hw_breakpoint hw;
> + struct perf_event_attr attr = bp->attr;
> + unsigned int seq;
> + int i, ret;
> +
> + lockdep_assert_irqs_disabled();
> +
> + attr.bp_addr = addr;
> + ret = hw_breakpoint_arch_parse(bp, &attr, &hw);
> + if (ret)
> + return ret;
[Severity: High]
Does this discard the validated software state?
The new address configuration is successfully validated into the local 'hw'
struct, but neither the architecture-specific state (counter_arch_bp(bp))
nor the core attribute (bp->attr.bp_addr) are updated.
If the breakpoint is later re-installed (e.g., during CPU offline/online or
PMU scheduling), arch_install_hw_breakpoint() will read the stale address:
struct arch_hw_breakpoint *info = counter_arch_bp(bp);
...
this_cpu_write(cpu_debugreg[i], info->address);
barrier();
set_debugreg(info->address, i);
Could this silently revert the hardware register back to the old address?
Also, if a watchpoint fires, perf_bp_event() initializes the sample record
using the unmodified bp->attr.bp_addr. Could this result in reporting
incorrect addresses in the trace output?
> +
> + for (i = 0; i < HBP_NUM; i++) {
> + if (this_cpu_read(bp_per_reg[i]) == bp)
> + break;
> + }
> +
> + if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
> + return -ENOENT;
> +
> + do {
> + seq = this_cpu_inc_return(cpu_dr7_seq);
> + this_cpu_write(cpu_debugreg[i], addr);
> + barrier();
> + set_debugreg(addr, i);
> + } while (seq != this_cpu_read(cpu_dr7_seq));
> +
> + return 0;
> +}
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=2
^ permalink raw reply [flat|nested] 19+ messages in thread
* [PATCH v12 03/11] HWBP: Add modify_local_hw_breakpoint_addr() API
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
2026-08-07 15:33 ` [PATCH v12 01/11] x86/hw_breakpoints: Make DR7 updates NMI safe Masami Hiramatsu (Google)
2026-08-07 15:33 ` [PATCH v12 02/11] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
@ 2026-08-07 15:33 ` Masami Hiramatsu (Google)
2026-08-07 15:58 ` sashiko-bot
2026-08-07 15:34 ` [PATCH v12 04/11] tracing/wprobe: Add wprobe (watchpoint probe) trace event support Masami Hiramatsu (Google)
` (7 subsequent siblings)
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:33 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add modify_local_hw_breakpoint_addr() to update only the watched
address of an installed hardware breakpoint on the local CPU without
releasing and reserving its hardware slot. This is available when the
architecture selects CONFIG_HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR.
The caller must provide an installed local event and a valid address,
and update other CPUs separately.
Link: https://lore.kernel.org/all/f9c49dfa49bdc57ba8c0574bc9981c1e581acf92.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
---
Changes in v12:
- Update bp->attr.bp_addr and counter_arch_bp(bp)->address before
calling arch_modify_local_hw_breakpoint_addr() to prevent race
conditions during interrupt handling.
- Restore counter_arch_bp(bp)->address and bp->attr.bp_addr if
arch_modify_local_hw_breakpoint_addr() fails.
Changes in v11:
- Check return value of arch_modify_local_hw_breakpoint_addr() and
return error on failure.
- Update bp->attr.bp_addr and counter_arch_bp(bp)->address only after
arch modification succeeds.
---
arch/Kconfig | 8 ++++++++
arch/x86/Kconfig | 1 +
include/linux/hw_breakpoint.h | 6 ++++++
kernel/events/hw_breakpoint.c | 40 ++++++++++++++++++++++++++++++++++++++++
4 files changed, 55 insertions(+)
diff --git a/arch/Kconfig b/arch/Kconfig
index fa7507ac8e13..bea383408e32 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -457,6 +457,14 @@ config HAVE_MIXED_BREAKPOINTS_REGS
Select this option if your arch implements breakpoints under the
latter fashion.
+config HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR
+ bool
+ depends on HAVE_HW_BREAKPOINT
+ help
+ Select this if the architecture can modify the address of an
+ installed hardware breakpoint on the local CPU without releasing
+ and reserving its hardware slot.
+
config HAVE_USER_RETURN_NOTIFIER
bool
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index fd979e0be1f0..33908103d9a0 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -247,6 +247,7 @@ config X86
select HAVE_FUNCTION_TRACER
select HAVE_GCC_PLUGINS
select HAVE_HW_BREAKPOINT
+ select HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR
select HAVE_IOREMAP_PROT
select HAVE_IRQ_EXIT_ON_IRQ_STACK if X86_64
select HAVE_IRQ_TIME_ACCOUNTING
diff --git a/include/linux/hw_breakpoint.h b/include/linux/hw_breakpoint.h
index db199d653dd1..bf65c7fffd99 100644
--- a/include/linux/hw_breakpoint.h
+++ b/include/linux/hw_breakpoint.h
@@ -81,6 +81,9 @@ register_wide_hw_breakpoint(struct perf_event_attr *attr,
perf_overflow_handler_t triggered,
void *context);
+int modify_local_hw_breakpoint_addr(struct perf_event *bp,
+ unsigned long addr);
+
extern int register_perf_hw_breakpoint(struct perf_event *bp);
extern void unregister_hw_breakpoint(struct perf_event *bp);
extern void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events);
@@ -124,6 +127,9 @@ register_wide_hw_breakpoint(struct perf_event_attr *attr,
perf_overflow_handler_t triggered,
void *context) { return NULL; }
static inline int
+modify_local_hw_breakpoint_addr(struct perf_event *bp,
+ unsigned long addr) { return -EOPNOTSUPP; }
+static inline int
register_perf_hw_breakpoint(struct perf_event *bp) { return -ENOSYS; }
static inline void unregister_hw_breakpoint(struct perf_event *bp) { }
static inline void
diff --git a/kernel/events/hw_breakpoint.c b/kernel/events/hw_breakpoint.c
index 789add0c185a..7d3a7bc050a4 100644
--- a/kernel/events/hw_breakpoint.c
+++ b/kernel/events/hw_breakpoint.c
@@ -888,6 +888,46 @@ void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events)
}
EXPORT_SYMBOL_GPL(unregister_wide_hw_breakpoint);
+/**
+ * modify_local_hw_breakpoint_addr - update a local breakpoint address
+ * @bp: the hwbp perf event for this CPU
+ * @addr: the new address for @bp
+ *
+ * Update only the address of an installed breakpoint on the local CPU without
+ * releasing and reserving its hardware slot. The caller must update other CPUs.
+ * Return 0, or -EOPNOTSUPP if the architecture does not support this operation.
+ */
+#ifdef CONFIG_HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR
+int modify_local_hw_breakpoint_addr(struct perf_event *bp,
+ unsigned long addr)
+{
+ unsigned long old_addr;
+ int ret;
+
+ lockdep_assert_irqs_disabled();
+
+ old_addr = bp->attr.bp_addr;
+ counter_arch_bp(bp)->address = addr;
+ bp->attr.bp_addr = addr;
+
+ ret = arch_modify_local_hw_breakpoint_addr(bp, addr);
+ if (ret) {
+ counter_arch_bp(bp)->address = old_addr;
+ bp->attr.bp_addr = old_addr;
+ return ret;
+ }
+
+ return 0;
+}
+#else
+int modify_local_hw_breakpoint_addr(struct perf_event *bp,
+ unsigned long addr)
+{
+ return -EOPNOTSUPP;
+}
+#endif
+EXPORT_SYMBOL_GPL(modify_local_hw_breakpoint_addr);
+
/**
* hw_breakpoint_is_used - check if breakpoints are currently used
*
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 03/11] HWBP: Add modify_local_hw_breakpoint_addr() API
2026-08-07 15:33 ` [PATCH v12 03/11] HWBP: Add modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
@ 2026-08-07 15:58 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 15:58 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 3 potential issue(s) to consider:
New issues:
- [High] Modifying hardware debug address registers (DR0-DR3) while they are still enabled in DR7 violates the x86 architectural guidelines and can cause unpredictable hardware behavior.
- [Medium] Temporary invalid software state in `bp->attr.bp_addr` before validation causes incorrect perf samples if interrupted by an NMI.
Pre-existing issues:
- [High] Lockless traversal of `kprobe_blacklist` can lead to Use-After-Free during module unload.
--
--- Patch [3]: [PATCH v12 03/11] HWBP: Add modify_local_hw_breakpoint_addr() API ---
commit 894126e474a8f66ca7e12e6c727a417646ec68a9
Author: Masami Hiramatsu (Google) <mhiramat@kernel.org>
HWBP: Add modify_local_hw_breakpoint_addr() API
This adds modify_local_hw_breakpoint_addr() to update the watched address
of an installed hardware breakpoint on the local CPU without releasing and
reserving its hardware slot. It is available when architectures select
HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR.
Link: https://lore.kernel.org/all/f9c49dfa49bdc57ba8c0574bc9981c1e581acf92.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/kernel/events/hw_breakpoint.c b/kernel/events/hw_breakpoint.c
> index 789add0c185a0..7d3a7bc050a40 100644
> --- a/kernel/events/hw_breakpoint.c
> +++ b/kernel/events/hw_breakpoint.c
[ ... ]
> @@ -888,6 +888,46 @@ void unregister_wide_hw_breakpoint(struct perf_event * __percpu *cpu_events)
> }
> EXPORT_SYMBOL_GPL(unregister_wide_hw_breakpoint);
>
> +/**
> + * modify_local_hw_breakpoint_addr - update a local breakpoint address
> + * @bp: the hwbp perf event for this CPU
> + * @addr: the new address for @bp
> + *
> + * Update only the address of an installed breakpoint on the local CPU without
> + * releasing and reserving its hardware slot. The caller must update other CPUs.
> + * Return 0, or -EOPNOTSUPP if the architecture does not support this operation.
> + */
> +#ifdef CONFIG_HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR
> +int modify_local_hw_breakpoint_addr(struct perf_event *bp,
> + unsigned long addr)
> +{
> + unsigned long old_addr;
> + int ret;
> +
> + lockdep_assert_irqs_disabled();
> +
> + old_addr = bp->attr.bp_addr;
> + counter_arch_bp(bp)->address = addr;
> + bp->attr.bp_addr = addr;
[Severity: Medium]
Could this early update of bp->attr.bp_addr lead to incorrect perf samples?
If an NMI triggers the hardware breakpoint on the old address before the
hardware registers are updated and validated, the NMI handler might locklessly
read the temporarily overwritten bp->attr.bp_addr. Would this attribute the
sample to the new, potentially invalid address instead of the old one?
> +
> + ret = arch_modify_local_hw_breakpoint_addr(bp, addr);
[Severity: High]
Does modifying the hardware debug address register without clearing the DR7
enable bits violate x86 architectural guidelines?
When arch_modify_local_hw_breakpoint_addr() is called for x86:
arch/x86/kernel/hw_breakpoint.c:arch_modify_local_hw_breakpoint_addr() {
...
do {
seq = this_cpu_inc_return(cpu_dr7_seq);
this_cpu_write(cpu_debugreg[i], addr);
barrier();
set_debugreg(addr, i);
} while (seq != this_cpu_read(cpu_dr7_seq));
...
}
Could this cause unpredictable hardware behavior since the Intel manual
dictates that DR0-DR3 should only be modified when their corresponding
enable bits in DR7 are clear?
[Severity: High]
This is a pre-existing issue, but can the lockless traversal in the
blacklist check lead to a use-after-free during module unload?
When parsing the breakpoint, the call chain goes through:
arch_modify_local_hw_breakpoint_addr() -> hw_breakpoint_arch_parse() ->
arch_build_bp_info() -> within_kprobe_blacklist()
kernel/kprobes.c:__within_kprobe_blacklist() {
...
list_for_each_entry(ent, &kprobe_blacklist, list) {
if (addr >= ent->start_addr && addr < ent->end_addr)
return true;
}
...
}
If a module is unloaded concurrently, kprobe_remove_area_blacklist() deletes
entries and immediately frees them with kfree(). Would a concurrent hardware
breakpoint modification dereference freed memory here since kprobe_mutex
is not held and RCU protection is not used?
> + if (ret) {
> + counter_arch_bp(bp)->address = old_addr;
> + bp->attr.bp_addr = old_addr;
> + return ret;
> + }
> +
> + return 0;
> +}
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=3
^ permalink raw reply [flat|nested] 19+ messages in thread
* [PATCH v12 04/11] tracing/wprobe: Add wprobe (watchpoint probe) trace event support
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (2 preceding siblings ...)
2026-08-07 15:33 ` [PATCH v12 03/11] HWBP: Add modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
@ 2026-08-07 15:34 ` Masami Hiramatsu (Google)
2026-08-07 15:59 ` sashiko-bot
2026-08-07 15:34 ` [PATCH v12 05/11] x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires Masami Hiramatsu (Google)
` (6 subsequent siblings)
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:34 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add hardware-breakpoint-based dynamic trace event support (wprobe).
Wprobe creates a dynamic event on data read/write accesses using
hardware breakpoints and logs the access context and fetchargs.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v12:
- Fix syntax comment typo ('b' to 'w') and remove dead 'ret = 0'
initialization in __trace_wprobe_create().
- Fix documentation notation to SYMBOL[[+|-]OFFS].
Changes in v11:
- Add trace_wprobe_is_busy() to prevent releasing busy events.
Changes in v9:
- Use kzalloc_flex for alloc_trace_wprobe.
---
Documentation/trace/index.rst | 1
Documentation/trace/wprobetrace.rst | 70 +++
include/linux/trace_events.h | 2
kernel/trace/Kconfig | 13 +
kernel/trace/Makefile | 1
kernel/trace/trace.c | 9
kernel/trace/trace.h | 5
kernel/trace/trace_probe.c | 21 +
kernel/trace/trace_probe.h | 10
kernel/trace/trace_wprobe.c | 765 +++++++++++++++++++++++++++++++++++
10 files changed, 893 insertions(+), 4 deletions(-)
create mode 100644 Documentation/trace/wprobetrace.rst
create mode 100644 kernel/trace/trace_wprobe.c
diff --git a/Documentation/trace/index.rst b/Documentation/trace/index.rst
index 5d9bf4694d5d..2f04f32001ed 100644
--- a/Documentation/trace/index.rst
+++ b/Documentation/trace/index.rst
@@ -36,6 +36,7 @@ the Linux kernel.
kprobes
kprobetrace
fprobetrace
+ wprobetrace
eprobetrace
fprobe
ring-buffer-design
diff --git a/Documentation/trace/wprobetrace.rst b/Documentation/trace/wprobetrace.rst
new file mode 100644
index 000000000000..ad5f089b5ef5
--- /dev/null
+++ b/Documentation/trace/wprobetrace.rst
@@ -0,0 +1,70 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+=======================================
+Watchpoint probe (wprobe) Event Tracing
+=======================================
+
+.. Author: Masami Hiramatsu <mhiramat@kernel.org>
+
+Overview
+--------
+
+Wprobe event is a dynamic event based on the hardware breakpoint, which is
+similar to other probe events, but it is for watching data access. It allows
+you to trace which code accesses a specified data.
+
+As same as other dynamic events, wprobe events are defined via
+`dynamic_events` interface file on tracefs.
+
+Synopsis of wprobe-events
+-------------------------
+::
+
+ w:[GRP/][EVENT] SPEC [FETCHARGS] : Probe on data access
+
+ GRP : Group name for wprobe. If omitted, use "wprobes" for it.
+ EVENT : Event name for wprobe. If omitted, an event name is
+ generated based on the address or symbol.
+ SPEC : Breakpoint specification.
+ [r|w|rw]@<ADDRESS|SYMBOL[[+|-]OFFS]>[:LENGTH]
+
+ r|w|rw : Access type, r for read, w for write, and rw for both.
+ Default is rw if omitted.
+ ADDRESS : Address to trace (hexadecimal). MUST be in kernel space.
+ SYMBOL : Symbol name to trace.
+ LENGTH : Length of the data to trace in bytes. (1, 2, 4, or 8)
+
+ FETCHARGS : Arguments. Each probe can have up to 128 args.
+ $addr : Fetch the accessing address.
+ $value : Fetch the memory value at the accessing address (same as +0($addr)).
+ @ADDR : Fetch memory at ADDR (ADDR should be in kernel)
+ @SYM[+|-offs] : Fetch memory at SYM +|- offs (SYM should be a data symbol)
+ +|-[u]OFFS(FETCHARG) : Fetch memory at FETCHARG +|- OFFS address.(\*1)(\*2)
+ \IMM : Store an immediate value to the argument.
+ NAME=FETCHARG : Set NAME as the argument name of FETCHARG.
+ FETCHARG:TYPE : Set TYPE as the type of FETCHARG. Currently, basic types
+ (u8/u16/u32/u64/s8/s16/s32/s64), hexadecimal types
+ (x8/x16/x32/x64), "char", "string", "ustring", "symbol", "symstr"
+ and bitfield are supported.
+
+ (\*1) this is useful for fetching a field of data structures.
+ (\*2) "u" means user-space dereference.
+
+For the details of TYPE, see :ref:`kprobetrace documentation <kprobetrace_types>`.
+
+Usage examples
+--------------
+Here is an example to add a wprobe event on a variable `jiffies`.
+::
+
+ # echo 'w:my_jiffies w@jiffies' >> dynamic_events
+ # cat dynamic_events
+ w:wprobes/my_jiffies w@jiffies
+ # echo 1 > events/wprobes/enable
+ # cat trace | head
+ # TASK-PID CPU# ||||| TIMESTAMP FUNCTION
+ # | | | ||||| | |
+ <idle>-0 [000] d.Z1. 717.026259: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
+ <idle>-0 [000] d.Z1. 717.026373: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
+
+You can see the code which writes to `jiffies` is `tick_do_update_jiffies64()`.
diff --git a/include/linux/trace_events.h b/include/linux/trace_events.h
index 5cbd09c8be8d..43ffd9a76d88 100644
--- a/include/linux/trace_events.h
+++ b/include/linux/trace_events.h
@@ -337,6 +337,7 @@ enum {
TRACE_EVENT_FL_UPROBE_BIT,
TRACE_EVENT_FL_EPROBE_BIT,
TRACE_EVENT_FL_FPROBE_BIT,
+ TRACE_EVENT_FL_WPROBE_BIT,
TRACE_EVENT_FL_CUSTOM_BIT,
TRACE_EVENT_FL_TEST_STR_BIT,
};
@@ -367,6 +368,7 @@ enum {
TRACE_EVENT_FL_UPROBE = (1 << TRACE_EVENT_FL_UPROBE_BIT),
TRACE_EVENT_FL_EPROBE = (1 << TRACE_EVENT_FL_EPROBE_BIT),
TRACE_EVENT_FL_FPROBE = (1 << TRACE_EVENT_FL_FPROBE_BIT),
+ TRACE_EVENT_FL_WPROBE = (1 << TRACE_EVENT_FL_WPROBE_BIT),
TRACE_EVENT_FL_CUSTOM = (1 << TRACE_EVENT_FL_CUSTOM_BIT),
TRACE_EVENT_FL_TEST_STR = (1 << TRACE_EVENT_FL_TEST_STR_BIT),
};
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index 0ab5916575a9..b58c2565024f 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -862,6 +862,19 @@ config EPROBE_EVENTS
convert the type of an event field. For example, turn an
address into a string.
+config WPROBE_EVENTS
+ bool "Enable wprobe-based dynamic events"
+ depends on TRACING
+ depends on HAVE_HW_BREAKPOINT
+ select PROBE_EVENTS
+ select DYNAMIC_EVENTS
+ help
+ This allows the user to add watchpoint tracing events based on
+ hardware breakpoints on the fly via the ftrace interface.
+
+ Those events can be inserted wherever hardware breakpoints can be
+ set, and record accessed memory address and values.
+
config BPF_EVENTS
depends on BPF_SYSCALL
depends on (KPROBE_EVENTS || UPROBE_EVENTS) && PERF_EVENTS
diff --git a/kernel/trace/Makefile b/kernel/trace/Makefile
index f934ff586bd4..141c8323de20 100644
--- a/kernel/trace/Makefile
+++ b/kernel/trace/Makefile
@@ -126,6 +126,7 @@ obj-$(CONFIG_FTRACE_RECORD_RECURSION) += trace_recursion_record.o
obj-$(CONFIG_FPROBE) += fprobe.o
obj-$(CONFIG_RETHOOK) += rethook.o
obj-$(CONFIG_FPROBE_EVENTS) += trace_fprobe.o
+obj-$(CONFIG_WPROBE_EVENTS) += trace_wprobe.o
obj-$(CONFIG_TRACEPOINT_BENCHMARK) += trace_benchmark.o
obj-$(CONFIG_RV) += rv/
diff --git a/kernel/trace/trace.c b/kernel/trace/trace.c
index 19cc07360005..4ebece96d8b7 100644
--- a/kernel/trace/trace.c
+++ b/kernel/trace/trace.c
@@ -4294,8 +4294,12 @@ static const char readme_msg[] =
" uprobe_events\t\t- Create/append/remove/show the userspace dynamic events\n"
"\t\t\t Write into this file to define/undefine new trace events.\n"
#endif
+#ifdef CONFIG_WPROBE_EVENTS
+ " wprobe_events\t\t- Create/append/remove/show the hardware breakpoint dynamic events\n"
+ "\t\t\t Write into this file to define/undefine new trace events.\n"
+#endif
#if defined(CONFIG_KPROBE_EVENTS) || defined(CONFIG_UPROBE_EVENTS) || \
- defined(CONFIG_FPROBE_EVENTS)
+ defined(CONFIG_FPROBE_EVENTS) || defined(CONFIG_WPROBE_EVENTS)
"\t accepts: event-definitions (one definition per line)\n"
#if defined(CONFIG_KPROBE_EVENTS) || defined(CONFIG_UPROBE_EVENTS)
"\t Format: p[:[<group>/][<event>]] <place> [<args>]\n"
@@ -4305,6 +4309,9 @@ static const char readme_msg[] =
"\t f[:[<group>/][<event>]] <func-name>[%return] [<args>]\n"
"\t t[:[<group>/][<event>]] <tracepoint> [<args>]\n"
#endif
+#ifdef CONFIG_WPROBE_EVENTS
+ "\t w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]\n"
+#endif
#ifdef CONFIG_HIST_TRIGGERS
"\t s:[synthetic/]<event> <field> [<field>]\n"
#endif
diff --git a/kernel/trace/trace.h b/kernel/trace/trace.h
index bf77331f56a4..64851a8d021f 100644
--- a/kernel/trace/trace.h
+++ b/kernel/trace/trace.h
@@ -179,6 +179,11 @@ struct fexit_trace_entry_head {
unsigned long ret_ip;
};
+struct wprobe_trace_entry_head {
+ struct trace_entry ent;
+ unsigned long ip;
+};
+
#define TRACE_BUF_SIZE 1024
struct trace_array;
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 5552ce7af5a1..9f4cad18977a 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -1402,6 +1402,23 @@ static int parse_probe_vars(char *orig_arg, const struct fetch_type *t,
return 0;
}
+ /* wprobe only support "$addr" and "$value" variable */
+ if (ctx->flags & TPARG_FL_WPROBE) {
+ if (!strcmp(arg, "addr")) {
+ code->op = FETCH_OP_BADDR;
+ return 0;
+ }
+ if (!strcmp(arg, "value")) {
+ code->op = FETCH_OP_BADDR;
+ code++;
+ code->op = FETCH_OP_DEREF;
+ code->offset = 0;
+ *pcode = code;
+ return 0;
+ }
+ goto inval;
+ }
+
if (strcmp(arg, "comm") == 0 || strcmp(arg, "COMM") == 0) {
code->op = FETCH_OP_COMM;
return 0;
@@ -1461,8 +1478,8 @@ static int parse_probe_arg_register(char *arg, struct fetch_insn *code,
{
int ret;
- if (ctx->flags & (TPARG_FL_TEVENT | TPARG_FL_FPROBE)) {
- /* eprobe and fprobe do not handle registers */
+ if (ctx->flags & (TPARG_FL_TEVENT | TPARG_FL_FPROBE | TPARG_FL_WPROBE)) {
+ /* eprobe, fprobe and wprobe do not handle registers */
trace_probe_log_err(ctx->offset, BAD_VAR);
return -EINVAL;
}
diff --git a/kernel/trace/trace_probe.h b/kernel/trace/trace_probe.h
index fba1af092a9b..6543d4c2cda5 100644
--- a/kernel/trace/trace_probe.h
+++ b/kernel/trace/trace_probe.h
@@ -90,6 +90,7 @@ typedef int (*print_type_func_t)(struct trace_seq *, void *, void *);
FETCH_OP(STACK, param), /* Stack: .param = index */ \
FETCH_OP(STACKP, none), /* Stack pointer */ \
FETCH_OP(RETVAL, none), /* Return value */ \
+ FETCH_OP(BADDR, none), /* Break address */ \
FETCH_OP(IMM, imm), /* Immediate: .immediate */ \
FETCH_OP(COMM, none), /* Current comm */ \
FETCH_OP(CURRENT, none), /* Current task_struct address */\
@@ -418,6 +419,7 @@ static inline int traceprobe_get_entry_data_size(struct trace_probe *tp)
#define TPARG_FL_USER BIT(4)
#define TPARG_FL_FPROBE BIT(5)
#define TPARG_FL_TPOINT BIT(6)
+#define TPARG_FL_WPROBE BIT(7)
#define TPARG_FL_LOC_MASK GENMASK(4, 0)
static inline bool tparg_is_function_entry(unsigned int flags)
@@ -544,6 +546,10 @@ extern int traceprobe_define_arg_fields(struct trace_event_call *event_call,
C(ARG_TOO_LONG, "Argument expression is too long"), \
C(ARRAY_NO_CLOSE, "Array is not closed"), \
C(ARRAY_TOO_BIG, "Array number is too big"), \
+ C(BAD_ACCESS_ADDR, "Invalid access memory address"), \
+ C(BAD_ACCESS_FMT, "Access memory address requires @"), \
+ C(BAD_ACCESS_LEN, "This memory access length is not supported"), \
+ C(BAD_ACCESS_TYPE, "Bad memory access type"), \
C(BAD_ADDR_SUFFIX, "Invalid probed address suffix"), \
C(BAD_ARG_NAME, "Argument name must follow the same rules as C identifiers"), \
C(BAD_ARG_NUM, "Invalid argument number"), \
@@ -626,7 +632,9 @@ extern int traceprobe_define_arg_fields(struct trace_event_call *event_call,
C(TYPECAST_NOT_EVENT, "Typecasts are only for eprobe fields"), \
C(TYPECAST_REQ_FIELD, "Typecast requires a field access"), \
C(TYPECAST_SYM_OFFSET, "@SYM+/-OFFSET with typecast needs parentheses"), \
- C(USED_ARG_NAME, "This argument name is already used"),
+ C(USED_ARG_NAME, "This argument name is already used"), \
+ C(WPROBE_NO_MAXACT, "Watchpoint probe does not support maxactive"), \
+ C(WPROBE_NO_SIBLING, "Watchpoint probe does not support sibling probes"),
#undef C
#define C(a, b) TP_ERR_##a
diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
new file mode 100644
index 000000000000..df592d9280a4
--- /dev/null
+++ b/kernel/trace/trace_wprobe.c
@@ -0,0 +1,765 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Hardware-breakpoint-based tracing events
+ *
+ * Copyright (C) 2023, Masami Hiramatsu <mhiramat@kernel.org>
+ */
+#define pr_fmt(fmt) "trace_wprobe: " fmt
+
+#include <linux/compiler.h>
+#include <linux/hw_breakpoint.h>
+#include <linux/kallsyms.h>
+#include <linux/list.h>
+#include <linux/module.h>
+#include <linux/mutex.h>
+#include <linux/perf_event.h>
+#include <linux/rculist.h>
+#include <linux/security.h>
+#include <linux/tracepoint.h>
+#include <linux/uaccess.h>
+
+#include <asm/ptrace.h>
+
+#include "trace_dynevent.h"
+#include "trace_probe.h"
+#include "trace_probe_kernel.h"
+#include "trace_probe_tmpl.h"
+#include "trace_output.h"
+
+#define WPROBE_EVENT_SYSTEM "wprobes"
+
+static int trace_wprobe_create(const char *raw_command);
+static int trace_wprobe_show(struct seq_file *m, struct dyn_event *ev);
+static int trace_wprobe_release(struct dyn_event *ev);
+static bool trace_wprobe_is_busy(struct dyn_event *ev);
+static bool trace_wprobe_match(const char *system, const char *event,
+ int argc, const char **argv, struct dyn_event *ev);
+
+static struct dyn_event_operations trace_wprobe_ops = {
+ .create = trace_wprobe_create,
+ .show = trace_wprobe_show,
+ .is_busy = trace_wprobe_is_busy,
+ .free = trace_wprobe_release,
+ .match = trace_wprobe_match,
+};
+
+struct trace_wprobe {
+ struct dyn_event devent;
+ struct perf_event * __percpu *bp_event;
+ unsigned long addr;
+ int len;
+ int type;
+ const char *symbol;
+ struct trace_probe tp;
+};
+
+static bool is_trace_wprobe(struct dyn_event *ev)
+{
+ return ev->ops == &trace_wprobe_ops;
+}
+
+static struct trace_wprobe *to_trace_wprobe(struct dyn_event *ev)
+{
+ return container_of(ev, struct trace_wprobe, devent);
+}
+
+#define for_each_trace_wprobe(pos, dpos) \
+ for_each_dyn_event(dpos) \
+ if (is_trace_wprobe(dpos) && (pos = to_trace_wprobe(dpos)))
+
+static bool trace_wprobe_is_busy(struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+
+ return trace_probe_is_enabled(&tw->tp);
+}
+
+static bool trace_wprobe_match(const char *system, const char *event,
+ int argc, const char **argv, struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+
+ if (event[0] != '\0' && strcmp(trace_probe_name(&tw->tp), event))
+ return false;
+
+ if (system && strcmp(trace_probe_group_name(&tw->tp), system))
+ return false;
+
+ return trace_probe_match_command_args(&tw->tp, argc, argv);
+}
+
+/*
+ * Note that we don't verify the fetch_insn code, since it does not come
+ * from user space.
+ */
+static int
+process_fetch_insn(struct fetch_insn *code, void *rec, void *edata,
+ void *dest, void *base)
+{
+ void *baddr = rec;
+ unsigned long val;
+ int ret;
+
+retry:
+ /* 1st stage: get value from context */
+ switch (code->op) {
+ case FETCH_OP_BADDR:
+ val = (unsigned long)baddr;
+ break;
+ case FETCH_NOP_SYMBOL: /* Ignore a place holder */
+ code++;
+ goto retry;
+ default:
+ ret = process_common_fetch_insn(code, &val);
+ if (ret < 0)
+ return ret;
+ }
+ code++;
+
+ return process_fetch_insn_bottom(code, val, dest, base);
+}
+NOKPROBE_SYMBOL(process_fetch_insn)
+
+static void wprobe_trace_handler(struct trace_wprobe *tw,
+ unsigned long addr,
+ struct pt_regs *regs,
+ struct trace_event_file *trace_file)
+{
+ struct wprobe_trace_entry_head *entry;
+ struct trace_event_call *call = trace_probe_event_call(&tw->tp);
+ struct trace_event_buffer fbuffer;
+ int dsize;
+
+ if (WARN_ON_ONCE(call != trace_file->event_call))
+ return;
+
+ if (trace_trigger_soft_disabled(trace_file))
+ return;
+
+ if (READ_ONCE(tw->addr) != addr)
+ return;
+
+ dsize = __get_data_size(&tw->tp, (void *)addr, NULL);
+
+ entry = trace_event_buffer_reserve(&fbuffer, trace_file,
+ sizeof(*entry) + tw->tp.size + dsize);
+ if (!entry)
+ return;
+
+ entry->ip = instruction_pointer(regs);
+ store_trace_args(&entry[1], &tw->tp, (void *)addr, NULL, sizeof(*entry), dsize);
+
+ fbuffer.regs = regs;
+ trace_event_buffer_commit(&fbuffer);
+}
+
+static void wprobe_perf_handler(struct perf_event *bp,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct trace_wprobe *tw = bp->overflow_handler_context;
+ struct event_file_link *link;
+ unsigned long addr = bp->attr.bp_addr;
+
+ trace_probe_for_each_link_rcu(link, &tw->tp)
+ wprobe_trace_handler(tw, addr, regs, link->file);
+}
+
+static int __register_trace_wprobe(struct trace_wprobe *tw)
+{
+ struct perf_event_attr attr;
+ int i, ret;
+
+ if (tw->bp_event)
+ return -EINVAL;
+
+ for (i = 0; i < tw->tp.nr_args; i++) {
+ ret = traceprobe_update_arg(&tw->tp.args[i]);
+ if (ret)
+ return ret;
+ }
+
+ hw_breakpoint_init(&attr);
+ attr.bp_addr = tw->addr;
+ attr.bp_len = tw->len;
+ attr.bp_type = tw->type;
+
+ tw->bp_event = register_wide_hw_breakpoint(&attr, wprobe_perf_handler, tw);
+ if (IS_ERR_PCPU(tw->bp_event)) {
+ int ret = PTR_ERR_PCPU(tw->bp_event);
+
+ tw->bp_event = NULL;
+ return ret;
+ }
+
+ return 0;
+}
+
+static void __unregister_trace_wprobe(struct trace_wprobe *tw)
+{
+ if (tw->bp_event) {
+ unregister_wide_hw_breakpoint(tw->bp_event);
+ tw->bp_event = NULL;
+ }
+}
+
+static void free_trace_wprobe(struct trace_wprobe *tw)
+{
+ if (tw) {
+ trace_probe_cleanup(&tw->tp);
+ kfree(tw->symbol);
+ kfree(tw);
+ }
+}
+DEFINE_FREE(free_trace_wprobe, struct trace_wprobe *,
+ if (!IS_ERR_OR_NULL(_T))
+ free_trace_wprobe(_T))
+
+
+static struct trace_wprobe *alloc_trace_wprobe(const char *group,
+ const char *event,
+ const char *symbol,
+ unsigned long addr,
+ int len, int type, int nargs)
+{
+ struct trace_wprobe *tw __free(free_trace_wprobe) = NULL;
+ int ret;
+
+ tw = kzalloc_flex(*tw, tp.args, nargs);
+ if (!tw)
+ return ERR_PTR(-ENOMEM);
+
+ if (symbol) {
+ tw->symbol = kstrdup(symbol, GFP_KERNEL);
+ if (!tw->symbol)
+ return ERR_PTR(-ENOMEM);
+ }
+ tw->addr = addr;
+ tw->len = len;
+ tw->type = type;
+
+ ret = trace_probe_init(&tw->tp, event, group, false, nargs);
+ if (ret < 0)
+ return ERR_PTR(ret);
+
+ dyn_event_init(&tw->devent, &trace_wprobe_ops);
+ return_ptr(tw);
+}
+
+static struct trace_wprobe *find_trace_wprobe(const char *event,
+ const char *group)
+{
+ struct dyn_event *pos;
+ struct trace_wprobe *tw;
+
+ for_each_trace_wprobe(tw, pos)
+ if (strcmp(trace_probe_name(&tw->tp), event) == 0 &&
+ strcmp(trace_probe_group_name(&tw->tp), group) == 0)
+ return tw;
+ return NULL;
+}
+
+static enum print_line_t
+print_wprobe_event(struct trace_iterator *iter, int flags,
+ struct trace_event *event)
+{
+ struct wprobe_trace_entry_head *field;
+ struct trace_seq *s = &iter->seq;
+ struct trace_probe *tp;
+
+ field = (struct wprobe_trace_entry_head *)iter->ent;
+ tp = trace_probe_primary_from_call(
+ container_of(event, struct trace_event_call, event));
+ if (WARN_ON_ONCE(!tp))
+ goto out;
+
+ trace_seq_printf(s, "%s: (", trace_probe_name(tp));
+
+ if (!seq_print_ip_sym_offset(s, field->ip, flags))
+ goto out;
+
+ trace_seq_putc(s, ')');
+
+ if (trace_probe_print_args(s, tp->args, tp->nr_args,
+ (u8 *)&field[1], field) < 0)
+ goto out;
+
+ trace_seq_putc(s, '\n');
+out:
+ return trace_handle_return(s);
+}
+
+static int wprobe_event_define_fields(struct trace_event_call *event_call)
+{
+ int ret;
+ struct wprobe_trace_entry_head field;
+ struct trace_probe *tp;
+
+ tp = trace_probe_primary_from_call(event_call);
+ if (WARN_ON_ONCE(!tp))
+ return -ENOENT;
+
+ DEFINE_FIELD(unsigned long, ip, FIELD_STRING_IP, 0);
+
+ return traceprobe_define_arg_fields(event_call, sizeof(field), tp);
+}
+
+static struct trace_event_functions wprobe_funcs = {
+ .trace = print_wprobe_event
+};
+
+static struct trace_event_fields wprobe_fields_array[] = {
+ { .type = TRACE_FUNCTION_TYPE,
+ .define_fields = wprobe_event_define_fields },
+ {}
+};
+
+static int wprobe_register(struct trace_event_call *event,
+ enum trace_reg type, void *data);
+
+static inline void init_trace_event_call(struct trace_wprobe *tw)
+{
+ struct trace_event_call *call = trace_probe_event_call(&tw->tp);
+
+ call->event.funcs = &wprobe_funcs;
+ call->class->fields_array = wprobe_fields_array;
+ call->flags = TRACE_EVENT_FL_WPROBE;
+ call->class->reg = wprobe_register;
+}
+
+static int register_wprobe_event(struct trace_wprobe *tw)
+{
+ init_trace_event_call(tw);
+ return trace_probe_register_event_call(&tw->tp);
+}
+
+static int register_trace_wprobe_event(struct trace_wprobe *tw)
+{
+ struct trace_wprobe *old_tw;
+ int ret;
+
+ guard(mutex)(&event_mutex);
+
+ old_tw = find_trace_wprobe(trace_probe_name(&tw->tp),
+ trace_probe_group_name(&tw->tp));
+ if (old_tw) {
+ /*
+ * Wprobe does not support sibling probes because the event
+ * trigger (set_wprobe/clear_wprobe) identifies the target
+ * wprobe by its event name. Having multiple wprobes sharing
+ * the same event name would make the target ambiguous.
+ */
+ trace_probe_log_set_index(0);
+ trace_probe_log_err(0, WPROBE_NO_SIBLING);
+ return -EBUSY;
+ }
+
+ ret = register_wprobe_event(tw);
+ if (ret) {
+ trace_probe_log_set_index(0);
+ if (ret == -EEXIST)
+ trace_probe_log_err(0, EVENT_EXIST);
+ else if (ret != -ENOMEM)
+ trace_probe_log_err(0, FAIL_REG_PROBE);
+ return ret;
+ }
+
+ dyn_event_add(&tw->devent, trace_probe_event_call(&tw->tp));
+ return 0;
+}
+static int unregister_wprobe_event(struct trace_wprobe *tw)
+{
+ return trace_probe_unregister_event_call(&tw->tp);
+}
+
+static int unregister_trace_wprobe(struct trace_wprobe *tw)
+{
+ if (trace_probe_has_sibling(&tw->tp))
+ goto unreg;
+
+ if (trace_probe_is_enabled(&tw->tp))
+ return -EBUSY;
+
+ if (trace_event_dyn_busy(trace_probe_event_call(&tw->tp)))
+ return -EBUSY;
+
+ if (unregister_wprobe_event(tw))
+ return -EBUSY;
+
+unreg:
+ __unregister_trace_wprobe(tw);
+ dyn_event_remove(&tw->devent);
+ trace_probe_unlink(&tw->tp);
+
+ return 0;
+}
+
+static int enable_trace_wprobe(struct trace_event_call *call,
+ struct trace_event_file *file)
+{
+ struct trace_probe *tp;
+ struct trace_wprobe *tw;
+ bool enabled;
+ int ret = 0;
+
+ tp = trace_probe_primary_from_call(call);
+ if (WARN_ON_ONCE(!tp))
+ return -ENODEV;
+ enabled = trace_probe_is_enabled(tp);
+
+ if (file) {
+ ret = trace_probe_add_file(tp, file);
+ if (ret)
+ return ret;
+ } else {
+ trace_probe_set_flag(tp, TP_FLAG_PROFILE);
+ }
+
+ if (!enabled) {
+ list_for_each_entry(tw, trace_probe_probe_list(tp), tp.list) {
+ ret = __register_trace_wprobe(tw);
+ if (ret < 0) {
+ struct trace_wprobe *tmp;
+
+ list_for_each_entry(tmp, trace_probe_probe_list(tp), tp.list) {
+ if (tmp == tw)
+ break;
+ __unregister_trace_wprobe(tmp);
+ }
+ if (file)
+ trace_probe_remove_file(tp, file);
+ else
+ trace_probe_clear_flag(tp, TP_FLAG_PROFILE);
+ return ret;
+ }
+ }
+ }
+
+ return 0;
+}
+
+static int disable_trace_wprobe(struct trace_event_call *call,
+ struct trace_event_file *file)
+{
+ struct trace_wprobe *tw;
+ struct trace_probe *tp;
+
+ tp = trace_probe_primary_from_call(call);
+ if (WARN_ON_ONCE(!tp))
+ return -ENODEV;
+
+ if (file) {
+ if (!trace_probe_get_file_link(tp, file))
+ return -ENOENT;
+ if (!trace_probe_has_single_file(tp))
+ goto out;
+ trace_probe_clear_flag(tp, TP_FLAG_TRACE);
+ } else {
+ trace_probe_clear_flag(tp, TP_FLAG_PROFILE);
+ }
+
+ if (!trace_probe_is_enabled(tp)) {
+ list_for_each_entry(tw, trace_probe_probe_list(tp), tp.list) {
+ __unregister_trace_wprobe(tw);
+ }
+ }
+
+out:
+ if (file)
+ trace_probe_remove_file(tp, file);
+
+ return 0;
+}
+
+static int wprobe_register(struct trace_event_call *event,
+ enum trace_reg type, void *data)
+{
+ struct trace_event_file *file = data;
+
+ switch (type) {
+ case TRACE_REG_REGISTER:
+ return enable_trace_wprobe(event, file);
+ case TRACE_REG_UNREGISTER:
+ return disable_trace_wprobe(event, file);
+
+#ifdef CONFIG_PERF_EVENTS
+ case TRACE_REG_PERF_REGISTER:
+ case TRACE_REG_PERF_UNREGISTER:
+ case TRACE_REG_PERF_OPEN:
+ case TRACE_REG_PERF_CLOSE:
+ case TRACE_REG_PERF_ADD:
+ case TRACE_REG_PERF_DEL:
+ return -EOPNOTSUPP;
+#endif
+ }
+ return 0;
+}
+
+static int parse_address_spec(const char *spec, unsigned long *addr, int *type,
+ int *len, char **symbol)
+{
+ char *_spec __free(kfree) = NULL;
+ int _len = HW_BREAKPOINT_LEN_4;
+ int _type = HW_BREAKPOINT_RW;
+ unsigned long _addr = 0;
+ char *at, *col;
+
+ _spec = kstrdup(spec, GFP_KERNEL);
+ if (!_spec)
+ return -ENOMEM;
+
+ at = strchr(_spec, '@');
+ col = strchr(_spec, ':');
+
+ if (!at) {
+ trace_probe_log_err(0, BAD_ACCESS_FMT);
+ return -EINVAL;
+ }
+
+ if (at != _spec) {
+ *at = '\0';
+
+ if (strcmp(_spec, "r") == 0)
+ _type = HW_BREAKPOINT_R;
+ else if (strcmp(_spec, "w") == 0)
+ _type = HW_BREAKPOINT_W;
+ else if (strcmp(_spec, "rw") == 0)
+ _type = HW_BREAKPOINT_RW;
+ else {
+ trace_probe_log_err(0, BAD_ACCESS_TYPE);
+ return -EINVAL;
+ }
+ }
+
+ if (col) {
+ *col = '\0';
+ if (kstrtoint(col + 1, 0, &_len)) {
+ trace_probe_log_err(col + 1 - _spec, BAD_ACCESS_LEN);
+ return -EINVAL;
+ }
+
+ switch (_len) {
+ case 1:
+ _len = HW_BREAKPOINT_LEN_1;
+ break;
+ case 2:
+ _len = HW_BREAKPOINT_LEN_2;
+ break;
+ case 4:
+ _len = HW_BREAKPOINT_LEN_4;
+ break;
+ case 8:
+ _len = HW_BREAKPOINT_LEN_8;
+ break;
+ default:
+ trace_probe_log_err(col + 1 - _spec, BAD_ACCESS_LEN);
+ return -EINVAL;
+ }
+ }
+
+ if (kstrtoul(at + 1, 0, &_addr) != 0) {
+ char *off_str = strpbrk(at + 1, "+-");
+ int offset = 0;
+
+ if (off_str) {
+ if (kstrtoint(off_str, 0, &offset) != 0) {
+ trace_probe_log_err(off_str - _spec, BAD_PROBE_ADDR);
+ return -EINVAL;
+ }
+ *off_str = '\0';
+ }
+ _addr = kallsyms_lookup_name(at + 1);
+ if (!_addr) {
+ trace_probe_log_err(at + 1 - _spec, BAD_ACCESS_ADDR);
+ return -ENOENT;
+ }
+ _addr += offset;
+ *symbol = kstrdup(at + 1, GFP_KERNEL);
+ if (!*symbol)
+ return -ENOMEM;
+ }
+
+ if (_addr != 0 && _addr < TASK_SIZE) {
+ trace_probe_log_err(at + 1 - _spec, BAD_ACCESS_ADDR);
+ return -EINVAL;
+ }
+
+ *addr = _addr;
+ *type = _type;
+ *len = _len;
+ return 0;
+}
+
+static int __trace_wprobe_create(int argc, const char *argv[])
+{
+ /*
+ * Argument syntax:
+ * w[:[GRP/][EVENT]] SPEC
+ *
+ * SPEC:
+ * [r|w|rw]@[ADDR|SYMBOL[+OFFS]][:LEN]
+ */
+ struct traceprobe_parse_context *ctx __free(traceprobe_parse_context) = NULL;
+ struct trace_wprobe *tw __free(free_trace_wprobe) = NULL;
+ const char *event = NULL, *group = WPROBE_EVENT_SYSTEM;
+ const char *tplog __free(trace_probe_log_clear) = NULL;
+ char *symbol __free(kfree) = NULL;
+ char *gbuf __free(kfree) = NULL;
+ char *ebuf __free(kfree) = NULL;
+ unsigned long addr;
+ int len, type, i;
+ int ret;
+
+ if (argv[0][0] != 'w')
+ return -ECANCELED;
+
+ tplog = trace_probe_log_init("wprobe", argc, argv);
+
+ if (argc < 2) {
+ trace_probe_log_set_index(0);
+ trace_probe_log_err(0, NO_ARG_BODY);
+ return -EINVAL;
+ }
+
+ if (argv[0][1] != '\0') {
+ if (argv[0][1] != ':') {
+ trace_probe_log_set_index(0);
+ trace_probe_log_err(1, WPROBE_NO_MAXACT);
+ return -EINVAL;
+ }
+ event = &argv[0][2];
+ }
+
+ trace_probe_log_set_index(1);
+ ret = parse_address_spec(argv[1], &addr, &type, &len, &symbol);
+ if (ret < 0)
+ return ret;
+
+ trace_probe_log_set_index(0);
+ if (event) {
+ gbuf = kmalloc(MAX_EVENT_NAME_LEN, GFP_KERNEL);
+ if (!gbuf)
+ return -ENOMEM;
+ ret = traceprobe_parse_event_name(&event, &group, gbuf,
+ event - argv[0]);
+ if (ret)
+ return ret;
+ }
+
+ if (!event) {
+ /* Make a new event name */
+ ebuf = kmalloc(MAX_EVENT_NAME_LEN, GFP_KERNEL);
+ if (!ebuf)
+ return -ENOMEM;
+ if (symbol)
+ snprintf(ebuf, MAX_EVENT_NAME_LEN, "%s", symbol);
+ else
+ snprintf(ebuf, MAX_EVENT_NAME_LEN, "w_0x%lx", addr);
+ sanitize_event_name(ebuf);
+ event = ebuf;
+ }
+
+ argc -= 2; argv += 2;
+ if (argc > MAX_TRACE_ARGS) {
+ trace_probe_log_set_index(2);
+ trace_probe_log_err(0, TOO_MANY_ARGS);
+ return -E2BIG;
+ }
+ tw = alloc_trace_wprobe(group, event, symbol, addr, len, type, argc);
+ if (IS_ERR(tw))
+ return PTR_ERR(tw);
+
+ ctx = kzalloc_obj(*ctx);
+ if (!ctx)
+ return -ENOMEM;
+
+ ctx->flags = TPARG_FL_KERNEL | TPARG_FL_WPROBE;
+
+ /* parse arguments */
+ for (i = 0; i < argc; i++) {
+ trace_probe_log_set_index(i + 2);
+ ctx->offset = 0;
+ ret = traceprobe_parse_probe_arg(&tw->tp, i, argv[i], ctx);
+ if (ret)
+ return ret; /* This can be -ENOMEM */
+ }
+
+ ret = traceprobe_set_print_fmt(&tw->tp, PROBE_PRINT_NORMAL);
+ if (ret < 0)
+ return ret;
+
+ ret = register_trace_wprobe_event(tw);
+ if (!ret)
+ tw = NULL; /* To avoid free */
+
+ return ret;
+}
+
+static int trace_wprobe_create(const char *raw_command)
+{
+ return trace_probe_create(raw_command, __trace_wprobe_create);
+}
+
+static int trace_wprobe_release(struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+ int ret = unregister_trace_wprobe(tw);
+
+ if (!ret)
+ free_trace_wprobe(tw);
+ return ret;
+}
+
+static int trace_wprobe_show(struct seq_file *m, struct dyn_event *ev)
+{
+ struct trace_wprobe *tw = to_trace_wprobe(ev);
+ int i;
+
+ seq_printf(m, "w:%s/%s", trace_probe_group_name(&tw->tp),
+ trace_probe_name(&tw->tp));
+
+ const char *type_str;
+
+ if (tw->type == HW_BREAKPOINT_R)
+ type_str = "r";
+ else if (tw->type == HW_BREAKPOINT_W)
+ type_str = "w";
+ else
+ type_str = "rw";
+
+ int len;
+
+ if (tw->len == HW_BREAKPOINT_LEN_1)
+ len = 1;
+ else if (tw->len == HW_BREAKPOINT_LEN_2)
+ len = 2;
+ else if (tw->len == HW_BREAKPOINT_LEN_4)
+ len = 4;
+ else
+ len = 8;
+
+ if (tw->symbol) {
+ unsigned long sym_addr = kallsyms_lookup_name(tw->symbol);
+ long offset = sym_addr ? (long)(tw->addr - sym_addr) : 0;
+
+ if (offset)
+ seq_printf(m, " %s@%s%+ld:%d", type_str, tw->symbol, offset, len);
+ else
+ seq_printf(m, " %s@%s:%d", type_str, tw->symbol, len);
+ } else {
+ seq_printf(m, " %s@0x%lx:%d", type_str, tw->addr, len);
+ }
+
+ for (i = 0; i < tw->tp.nr_args; i++)
+ seq_printf(m, " %s=%s", tw->tp.args[i].name, tw->tp.args[i].comm);
+ seq_putc(m, '\n');
+
+ return 0;
+}
+
+static __init int init_wprobe_trace(void)
+{
+ return dyn_event_register(&trace_wprobe_ops);
+}
+fs_initcall(init_wprobe_trace);
+
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 04/11] tracing/wprobe: Add wprobe (watchpoint probe) trace event support
2026-08-07 15:34 ` [PATCH v12 04/11] tracing/wprobe: Add wprobe (watchpoint probe) trace event support Masami Hiramatsu (Google)
@ 2026-08-07 15:59 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 15:59 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Medium] Hardware breakpoints are not synchronized with module unloads, leading to dangling breakpoints on reallocated memory.
--
--- Patch [4]: [PATCH v12 04/11] tracing/wprobe: Add wprobe (watchpoint probe) trace event support ---
commit 6f99af1ed7b9527ff7af700fbd99ed2a3d0632d9
Author: Masami Hiramatsu (Google) <mhiramat@kernel.org>
tracing/wprobe: Add wprobe (watchpoint probe) trace event support
This commit adds wprobe, a hardware-breakpoint-based dynamic trace event.
It allows users to monitor memory accesses by setting hardware breakpoints
on specified addresses or symbols and logging the access context.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
> new file mode 100644
> index 0000000000000..df592d9280a47
> --- /dev/null
> +++ b/kernel/trace/trace_wprobe.c
[ ... ]
> +static int __register_trace_wprobe(struct trace_wprobe *tw)
> +{
> + struct perf_event_attr attr;
> + int i, ret;
[ ... ]
> + hw_breakpoint_init(&attr);
> + attr.bp_addr = tw->addr;
> + attr.bp_len = tw->len;
> + attr.bp_type = tw->type;
> +
> + tw->bp_event = register_wide_hw_breakpoint(&attr, wprobe_perf_handler, tw);
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
[Severity: Medium]
When a watchpoint probe is created for a module symbol, does this leave a
dangling hardware breakpoint if the module is later unloaded?
Since the wprobe subsystem doesn't seem to register a module notifier
(unlike kprobes) to clean up when modules are unloaded, the breakpoint stays
active on the raw virtual address. If that memory is freed and subsequently
reallocated for a different module or vmalloc allocation, could this lead
to spurious trace events and unintended performance overhead on the reused
memory?
[ ... ]
> +static int parse_address_spec(const char *spec, unsigned long *addr, int *type,
> + int *len, char **symbol)
> +{
[ ... ]
> + if (kstrtoul(at + 1, 0, &_addr) != 0) {
> + char *off_str = strpbrk(at + 1, "+-");
> + int offset = 0;
> +
> + if (off_str) {
> + if (kstrtoint(off_str, 0, &offset) != 0) {
> + trace_probe_log_err(off_str - _spec, BAD_PROBE_ADDR);
> + return -EINVAL;
> + }
> + *off_str = '\0';
> + }
> + _addr = kallsyms_lookup_name(at + 1);
[Severity: Medium]
Is it possible to track the module lifecycle here when resolving the symbol?
Currently, kallsyms_lookup_name() simply translates the module symbol to an
address, but without pinning the module or reacting to unload events, the
hardware breakpoint loses its relationship to the original symbol once the
module disappears.
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=4
^ permalink raw reply [flat|nested] 19+ messages in thread
* [PATCH v12 05/11] x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (3 preceding siblings ...)
2026-08-07 15:34 ` [PATCH v12 04/11] tracing/wprobe: Add wprobe (watchpoint probe) trace event support Masami Hiramatsu (Google)
@ 2026-08-07 15:34 ` Masami Hiramatsu (Google)
2026-08-07 15:34 ` [PATCH v12 06/11] selftests: tracing: Add a basic testcase for wprobe Masami Hiramatsu (Google)
` (5 subsequent siblings)
10 siblings, 0 replies; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:34 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add CONFIG_HAVE_POST_BREAKPOINT_HOOK which indicates the hw_breakpoint
on that architecture fires after the target memory has been modified.
This is currently x86 only behavior.
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v11:
- Move select to alphabetically sorted place.
---
arch/Kconfig | 10 ++++++++++
arch/x86/Kconfig | 1 +
kernel/trace/Kconfig | 1 +
3 files changed, 12 insertions(+)
diff --git a/arch/Kconfig b/arch/Kconfig
index bea383408e32..a096952987a8 100644
--- a/arch/Kconfig
+++ b/arch/Kconfig
@@ -457,6 +457,16 @@ config HAVE_MIXED_BREAKPOINTS_REGS
Select this option if your arch implements breakpoints under the
latter fashion.
+config HAVE_POST_BREAKPOINT_HOOK
+ bool
+ depends on HAVE_HW_BREAKPOINT
+ help
+ Depending on the arch implementation of hardware breakpoints,
+ some of them provide breakpoint hook after the target memory
+ is modified.
+ Select this option if your arch implements breakpoints overflow
+ handler hooks after the target memory is modified.
+
config HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR
bool
depends on HAVE_HW_BREAKPOINT
diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig
index 33908103d9a0..533010ab29a9 100644
--- a/arch/x86/Kconfig
+++ b/arch/x86/Kconfig
@@ -287,6 +287,7 @@ config X86
select MMU_GATHER_RCU_TABLE_FREE
select MMU_GATHER_MERGE_VMAS
select HAVE_POSIX_CPU_TIMERS_TASK_WORK
+ select HAVE_POST_BREAKPOINT_HOOK
select HAVE_REGS_AND_STACK_ACCESS_API
select HAVE_RELIABLE_STACKTRACE if UNWINDER_ORC || STACK_VALIDATION
select HAVE_FUNCTION_ARG_ACCESS_API
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index b58c2565024f..d9b6fa5c35d9 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -866,6 +866,7 @@ config WPROBE_EVENTS
bool "Enable wprobe-based dynamic events"
depends on TRACING
depends on HAVE_HW_BREAKPOINT
+ depends on HAVE_POST_BREAKPOINT_HOOK
select PROBE_EVENTS
select DYNAMIC_EVENTS
help
^ permalink raw reply related [flat|nested] 19+ messages in thread* [PATCH v12 06/11] selftests: tracing: Add a basic testcase for wprobe
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (4 preceding siblings ...)
2026-08-07 15:34 ` [PATCH v12 05/11] x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires Masami Hiramatsu (Google)
@ 2026-08-07 15:34 ` Masami Hiramatsu (Google)
2026-08-07 15:34 ` [PATCH v12 07/11] selftests: tracing: Add syntax " Masami Hiramatsu (Google)
` (4 subsequent siblings)
10 siblings, 0 replies; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:34 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add 'add_remove_wprobe.tc' testcase for testing wprobe event that
tests adding and removing operations of the wprobe event.
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v9:
- Fix command check logic to prevent early exit under 'set -e'
(errexit) when grep or test fails.
- Simplify enable/disable status checks by removing cat pipes.
Changes in v8:
- Fixed silently test failure path.
---
tools/testing/selftests/ftrace/config | 1
.../ftrace/test.d/dynevent/add_remove_wprobe.tc | 63 ++++++++++++++++++++
2 files changed, 64 insertions(+)
create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
diff --git a/tools/testing/selftests/ftrace/config b/tools/testing/selftests/ftrace/config
index 544de0db5f58..d2f503722020 100644
--- a/tools/testing/selftests/ftrace/config
+++ b/tools/testing/selftests/ftrace/config
@@ -27,3 +27,4 @@ CONFIG_STACK_TRACER=y
CONFIG_TRACER_SNAPSHOT=y
CONFIG_UPROBES=y
CONFIG_UPROBE_EVENTS=y
+CONFIG_WPROBE_EVENTS=y
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
new file mode 100644
index 000000000000..647c37d5e4c8
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/add_remove_wprobe.tc
@@ -0,0 +1,63 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: Generic dynamic event - add/remove wprobe events
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README
+
+echo 0 > events/enable
+echo > dynamic_events
+
+# Use jiffies as a variable that is frequently written to.
+TARGET=jiffies
+
+echo "w:my_wprobe w@$TARGET" >> dynamic_events
+
+if ! grep -q my_wprobe dynamic_events; then
+ echo "Failed to create wprobe event"
+ exit_fail
+fi
+
+if [ ! -d events/wprobes/my_wprobe ]; then
+ echo "Failed to create wprobe event directory"
+ exit_fail
+fi
+
+echo 1 > events/wprobes/my_wprobe/enable
+
+# Check if the event is enabled
+if ! grep -q 1 events/wprobes/my_wprobe/enable; then
+ echo "Failed to enable wprobe event"
+ exit_fail
+fi
+
+# Let some time pass to trigger the breakpoint
+sleep 1
+
+# Check if we got any trace output
+if ! grep -q my_wprobe trace; then
+ echo "wprobe event was not triggered"
+ exit_fail
+fi
+
+echo 0 > events/wprobes/my_wprobe/enable
+
+# Check if the event is disabled
+if ! grep -q 0 events/wprobes/my_wprobe/enable; then
+ echo "Failed to disable wprobe event"
+ exit_fail
+fi
+
+echo "-:my_wprobe" >> dynamic_events
+
+if grep -q my_wprobe dynamic_events; then
+ echo "Failed to remove wprobe event"
+ exit_fail
+fi
+
+if [ -d events/wprobes/my_wprobe ]; then
+ echo "Failed to remove wprobe event directory"
+ exit_fail
+fi
+
+clear_trace
+
+exit 0
^ permalink raw reply related [flat|nested] 19+ messages in thread* [PATCH v12 07/11] selftests: tracing: Add syntax testcase for wprobe
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (5 preceding siblings ...)
2026-08-07 15:34 ` [PATCH v12 06/11] selftests: tracing: Add a basic testcase for wprobe Masami Hiramatsu (Google)
@ 2026-08-07 15:34 ` Masami Hiramatsu (Google)
2026-08-07 15:34 ` [PATCH v12 08/11] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers Masami Hiramatsu (Google)
` (3 subsequent siblings)
10 siblings, 0 replies; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:34 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add "wprobe_syntax_errors.tc" testcase for testing syntax errors
of the watch probe events.
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v12:
- Add NO_ARG_BODY error case.
Changes in v11:
- Add WPROBE_NO_MAXACT syntax error case.
---
.../test.d/dynevent/wprobes_syntax_errors.tc | 22 ++++++++++++++++++++
1 file changed, 22 insertions(+)
create mode 100644 tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
diff --git a/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc b/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
new file mode 100644
index 000000000000..c279bea8d8b9
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/dynevent/wprobes_syntax_errors.tc
@@ -0,0 +1,22 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: Watch probe event parser error log check
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README
+
+check_error() { # command-with-error-pos-by-^
+ ftrace_errlog_check 'wprobe' "$1" 'dynamic_events'
+}
+
+check_error '^w' # NO_ARG_BODY
+check_error 'w^10 w@jiffies' # WPROBE_NO_MAXACT
+check_error 'w ^symbol' # BAD_ACCESS_FMT
+check_error 'w ^a@symbol' # BAD_ACCESS_TYPE
+check_error 'w w@^symbol' # BAD_ACCESS_ADDR
+check_error 'w w@jiffies^+offset' # BAD_ACCESS_ADDR
+check_error 'w w@jiffies:^100' # BAD_ACCESS_LEN
+check_error 'w w@jiffies ^$arg1' # BAD_VAR
+check_error 'w w@jiffies ^$retval' # BAD_VAR
+check_error 'w w@jiffies ^$stack' # BAD_VAR
+check_error 'w w@jiffies ^%ax' # BAD_VAR
+
+exit 0
^ permalink raw reply related [flat|nested] 19+ messages in thread* [PATCH v12 08/11] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (6 preceding siblings ...)
2026-08-07 15:34 ` [PATCH v12 07/11] selftests: tracing: Add syntax " Masami Hiramatsu (Google)
@ 2026-08-07 15:34 ` Masami Hiramatsu (Google)
2026-08-07 15:54 ` sashiko-bot
2026-08-07 15:34 ` [PATCH v12 09/11] selftests: ftrace: Add wprobe trigger testcase Masami Hiramatsu (Google)
` (2 subsequent siblings)
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:34 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add set_wprobe and clear_wprobe event triggers to dynamically attach
and detach hardware breakpoint address monitoring based on event field
contents.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v12:
- Decrement trigger data->count only when watchpoint state is actually changed.
- Remove dyn_event_ops_mutex to fix lockdep circular dependency deadlock.
- Add event_trigger_init() call to prevent premature freeing of trigger_data.
- Use WRITE_ONCE() when modifying tw->addr to pair with READ_ONCE().
- Add missing braces to else-block in wprobe_trigger_print().
- Initialize tw->addr only after trace_event_enable_disable() succeeds.
- Counter counts only if the trigger is actually working.
Changes in v11:
- Soft-enable (register) wprobe event on WPROBE_DEFAULT_CLEAR_ADDRESS.
- Add work_pending check before updating tw->addr.
---
Documentation/trace/wprobetrace.rst | 98 +++++++
include/linux/trace_events.h | 1
kernel/trace/Kconfig | 10 +
kernel/trace/trace.h | 1
kernel/trace/trace_dynevent.h | 1
kernel/trace/trace_events_trigger.c | 2
kernel/trace/trace_probe.c | 2
kernel/trace/trace_probe.h | 9 +
kernel/trace/trace_wprobe.c | 527 +++++++++++++++++++++++++++++++++++
9 files changed, 648 insertions(+), 3 deletions(-)
diff --git a/Documentation/trace/wprobetrace.rst b/Documentation/trace/wprobetrace.rst
index ad5f089b5ef5..a579347735d2 100644
--- a/Documentation/trace/wprobetrace.rst
+++ b/Documentation/trace/wprobetrace.rst
@@ -68,3 +68,101 @@ Here is an example to add a wprobe event on a variable `jiffies`.
<idle>-0 [000] d.Z1. 717.026373: my_jiffies: (tick_do_update_jiffies64+0xbe/0x130)
You can see the code which writes to `jiffies` is `tick_do_update_jiffies64()`.
+
+Combination with trigger action
+-------------------------------
+The event trigger action can extend the utilization of this wprobe.
+
+- set_wprobe:WPEVENT:FIELD[+|-ADJUST]
+- clear_wprobe:WPEVENT[:FIELD[+|-]ADJUST]
+
+Set these triggers to the target event, then the WPROBE event will be
+setup to trace the memory access at FIELD[+|-ADJUST] address.
+When clear_wprobe is hit, if FIELD is NOT specified, the WPEVENT is
+forcibly cleared. If FIELD[[+|-]ADJUST] is set, it clears WPEVENT only
+if its watching address is the same as the FIELD[[+|-]ADJUST] value.
+
+Notes:
+The set_wprobe trigger does not change the type and length, these
+must be set when creating a new wprobe.
+
+The WPROBE event must be disabled when setting the new trigger
+and it will be busy afterwards. Recommended usage is to add a new
+wprobe at NULL address and keep disabled.
+
+Wprobe triggers only support target addresses in kernel memory. If a
+set_wprobe trigger evaluates to a user-space memory address or NULL
+pointer, the trigger action ignores the update and skips setting the
+watchpoint.
+
+Wprobe triggers are not supported on kprobe_events, because kprobes
+themselves can use software breakpoints which conflicts with wprobe
+operation.
+
+
+For example, trace the first 8 bytes of the dentry data structure passed
+to do_truncate() until it is deleted by dentry_kill().
+(Note: all tracefs setup uses '>>' so that it does not kick do_truncate())
+::
+
+ # echo 'w:watch rw@0:8 address=$addr value=+0($addr)' >> dynamic_events
+ # echo 'f:truncate do_truncate dentry=$arg2' >> dynamic_events
+ # echo 'set_wprobe:watch:dentry' >> events/fprobes/truncate/trigger
+ # echo 'f:dentry_kill dentry_kill dentry=$arg1' >> dynamic_events
+ # echo 'clear_wprobe:watch:dentry' >> events/fprobes/dentry_kill/trigger
+ # echo 1 >> events/fprobes/truncate/enable
+ # echo 1 >> events/fprobes/dentry_kill/enable
+
+ # echo aaa > /tmp/hoge
+ # echo bbb > /tmp/hoge
+ # echo ccc > /tmp/hoge
+ # rm /tmp/hoge
+
+Then, the trace data will show::
+
+ # tracer: nop
+ #
+ # entries-in-buffer/entries-written: 32/32 #P:8
+ #
+ # _-----=> irqs-off/BH-disabled
+ # / _----=> need-resched
+ # | / _---=> hardirq/softirq
+ # || / _--=> preempt-depth
+ # ||| / _-=> migrate-disable
+ # |||| / delay
+ # TASK-PID CPU# ||||| TIMESTAMP FUNCTION
+ # | | | ||||| | |
+ sh-107 [004] ...1. 9.990418: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004ad6618
+ sh-107 [004] ...1. 9.990914: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b3de78
+ sh-107 [004] ...1. 9.993175: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049ddd40
+ sh-107 [004] ..... 9.995198: truncate: (do_truncate+0x4/0x120) dentry=0xffff8880048083a8
+ sh-107 [004] ...1. 9.995389: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db998
+ sh-107 [004] ..Zff 9.997503: watch: (lookup_fast+0xaa/0x150) address=0xffff8880048083a8 value=0x8200080
+ sh-107 [004] ..Zff 9.997509: watch: (path_openat+0x211/0xda0) address=0xffff8880048083a8 value=0x8200080
+ sh-107 [004] ..Zff 9.997514: watch: (path_openat+0xa56/0xda0) address=0xffff8880048083a8 value=0x8200080
+ sh-107 [004] ..Zff 9.997518: watch: (path_openat+0xae2/0xda0) address=0xffff8880048083a8 value=0x8200080
+ sh-107 [004] ..... 9.997521: truncate: (do_truncate+0x4/0x120) dentry=0xffff8880048083a8
+ sh-107 [004] ...1. 9.997582: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004808270
+ sh-107 [004] ...1. 9.999365: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db728
+ sh-107 [004] ...1. 9.999388: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b1c000
+ rm-113 [005] ..Zff 10.000965: watch: (lookup_fast+0xaa/0x150) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.000971: watch: (path_lookupat+0x97/0x1e0) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.000984: watch: (lookup_fast+0xaa/0x150) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.000988: watch: (path_lookupat+0x97/0x1e0) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.001010: watch: (lookup_one_qstr_excl+0x28/0x140) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.001014: watch: (lookup_one_qstr_excl+0xd1/0x140) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.001018: watch: (may_delete_dentry+0x1c/0x200) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.001021: watch: (may_delete_dentry+0x195/0x200) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] ..Zff 10.001031: watch: (vfs_unlink+0x5e/0x260) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] d.Z.. 10.001067: watch: (d_make_discardable+0x1b/0x40) address=0xffff8880048083a8 value=0x8200080
+ rm-113 [005] d.Z.. 10.001071: watch: (d_make_discardable+0x29/0x40) address=0xffff8880048083a8 value=0x200080
+ rm-113 [005] ...1. 10.001072: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880048083a8
+ rm-113 [005] ...1. 10.001218: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880048083a8
+ sh-107 [004] ...1. 10.001416: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db110
+ sh-107 [004] ...1. 10.001444: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff8880049db248
+ sh-107 [004] ...1. 10.001500: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004ad6618
+ sh-107 [004] ...1. 10.002067: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b41e78
+ sh-107 [004] ...1. 10.904920: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004b41e78
+ sh-107 [004] ...1. 10.905129: dentry_kill: (dentry_kill+0x0/0x2c0) dentry=0xffff888004ad6618
+
+You can see the watch event is correctly configured on the dentry.
diff --git a/include/linux/trace_events.h b/include/linux/trace_events.h
index 43ffd9a76d88..f81a5308c116 100644
--- a/include/linux/trace_events.h
+++ b/include/linux/trace_events.h
@@ -738,6 +738,7 @@ enum event_trigger_type {
ETT_EVENT_HIST = (1 << 4),
ETT_HIST_ENABLE = (1 << 5),
ETT_EVENT_EPROBE = (1 << 6),
+ ETT_EVENT_WPROBE = (1 << 7),
};
extern int filter_match_preds(struct event_filter *filter, void *rec);
diff --git a/kernel/trace/Kconfig b/kernel/trace/Kconfig
index d9b6fa5c35d9..5fd8ed63c516 100644
--- a/kernel/trace/Kconfig
+++ b/kernel/trace/Kconfig
@@ -876,6 +876,16 @@ config WPROBE_EVENTS
Those events can be inserted wherever hardware breakpoints can be
set, and record accessed memory address and values.
+config WPROBE_TRIGGERS
+ depends on WPROBE_EVENTS
+ depends on HAVE_MODIFY_LOCAL_HW_BREAKPOINT_ADDR
+ bool
+ default y
+ help
+ This adds an event trigger which will set the wprobe on a specific
+ field of an event. This allows user to trace the memory access of
+ an address pointed by the event field.
+
config BPF_EVENTS
depends on BPF_SYSCALL
depends on (KPROBE_EVENTS || UPROBE_EVENTS) && PERF_EVENTS
diff --git a/kernel/trace/trace.h b/kernel/trace/trace.h
index 64851a8d021f..a789a722bc8b 100644
--- a/kernel/trace/trace.h
+++ b/kernel/trace/trace.h
@@ -1983,6 +1983,7 @@ trigger_data_alloc(struct event_command *cmd_ops, char *cmd, char *param,
void *private_data);
extern void trigger_data_free(struct event_trigger_data *data);
extern int event_trigger_init(struct event_trigger_data *data);
+extern void event_trigger_free(struct event_trigger_data *data);
extern int trace_event_trigger_enable_disable(struct trace_event_file *file,
int trigger_enable);
extern void update_cond_flag(struct trace_event_file *file);
diff --git a/kernel/trace/trace_dynevent.h b/kernel/trace/trace_dynevent.h
index beee3f8d7544..77c2c84dcdb5 100644
--- a/kernel/trace/trace_dynevent.h
+++ b/kernel/trace/trace_dynevent.h
@@ -64,6 +64,7 @@ struct dyn_event {
};
extern struct list_head dyn_event_list;
+extern struct mutex dyn_event_ops_mutex;
static inline
int dyn_event_init(struct dyn_event *ev, struct dyn_event_operations *ops)
diff --git a/kernel/trace/trace_events_trigger.c b/kernel/trace/trace_events_trigger.c
index ad83419cb420..fa409ebd73c2 100644
--- a/kernel/trace/trace_events_trigger.c
+++ b/kernel/trace/trace_events_trigger.c
@@ -589,7 +589,7 @@ int event_trigger_init(struct event_trigger_data *data)
* Usually used directly as the @free method in event trigger
* implementations.
*/
-static void
+void
event_trigger_free(struct event_trigger_data *data)
{
if (WARN_ON_ONCE(data->ref <= 0))
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 9f4cad18977a..36e07275a04d 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -20,7 +20,7 @@
#undef C
#define C(a, b) b
-static const char *trace_probe_err_text[] = { ERRORS };
+const char *trace_probe_err_text[] = { ERRORS };
static const char *reserved_field_names[] = {
"common_type",
diff --git a/kernel/trace/trace_probe.h b/kernel/trace/trace_probe.h
index 6543d4c2cda5..e08f17c99138 100644
--- a/kernel/trace/trace_probe.h
+++ b/kernel/trace/trace_probe.h
@@ -634,7 +634,12 @@ extern int traceprobe_define_arg_fields(struct trace_event_call *event_call,
C(TYPECAST_SYM_OFFSET, "@SYM+/-OFFSET with typecast needs parentheses"), \
C(USED_ARG_NAME, "This argument name is already used"), \
C(WPROBE_NO_MAXACT, "Watchpoint probe does not support maxactive"), \
- C(WPROBE_NO_SIBLING, "Watchpoint probe does not support sibling probes"),
+ C(WPROBE_NO_SIBLING, "Watchpoint probe does not support sibling probes"), \
+ C(WPROBE_ON_KPROBE, "Wprobe trigger is not supported on kprobe event"), \
+ C(WPROBE_NOT_FOUND, "Target wprobe event is not found"), \
+ C(WPROBE_BUSY, "Target wprobe event is already enabled"), \
+ C(WPROBE_NEED_FIELD, "Wprobe trigger requires a target field"), \
+ C(WPROBE_BAD_FIELD, "Target field must be pointer size"),
#undef C
#define C(a, b) TP_ERR_##a
@@ -658,6 +663,8 @@ void __trace_probe_log_err(int offset, int err);
DEFINE_FREE(trace_probe_log_clear, const char *, if (_T) trace_probe_log_clear())
+extern const char *trace_probe_err_text[];
+
#define trace_probe_log_err(offs, err) \
__trace_probe_log_err(offs, TP_ERR_##err)
diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
index df592d9280a4..c64bbdc90a40 100644
--- a/kernel/trace/trace_wprobe.c
+++ b/kernel/trace/trace_wprobe.c
@@ -6,7 +6,9 @@
*/
#define pr_fmt(fmt) "trace_wprobe: " fmt
+#include <linux/atomic.h>
#include <linux/compiler.h>
+#include <linux/errno.h>
#include <linux/hw_breakpoint.h>
#include <linux/kallsyms.h>
#include <linux/list.h>
@@ -15,11 +17,16 @@
#include <linux/perf_event.h>
#include <linux/rculist.h>
#include <linux/security.h>
+#include <linux/spinlock.h>
#include <linux/tracepoint.h>
#include <linux/uaccess.h>
+#include <linux/workqueue.h>
+#include <linux/irq_work.h>
+#include <linux/preempt.h>
#include <asm/ptrace.h>
+#include "trace.h"
#include "trace_dynevent.h"
#include "trace_probe.h"
#include "trace_probe_kernel.h"
@@ -50,6 +57,17 @@ struct trace_wprobe {
int len;
int type;
const char *symbol;
+ raw_spinlock_t lock;
+ struct irq_work irq_work;
+ struct work_struct work;
+ atomic_t missed;
+ /*
+ * work_pending is set to 1 before irq_work_queue() and cleared to 0
+ * after wprobe_work_func() finishes on_each_cpu(). This prevents a
+ * new trigger from overwriting tw->addr while the work is propagating
+ * the old address to per-CPU debug registers via IPI.
+ */
+ atomic_t work_pending;
struct trace_probe tp;
};
@@ -198,14 +216,62 @@ static int __register_trace_wprobe(struct trace_wprobe *tw)
static void __unregister_trace_wprobe(struct trace_wprobe *tw)
{
if (tw->bp_event) {
+ irq_work_sync(&tw->irq_work);
+ cancel_work_sync(&tw->work);
unregister_wide_hw_breakpoint(tw->bp_event);
tw->bp_event = NULL;
}
}
+static int trace_wprobe_update_local(struct trace_wprobe *tw, unsigned long addr)
+{
+ struct perf_event * __percpu *pevent;
+ struct perf_event *bp;
+
+ pevent = tw->bp_event;
+ if (!pevent)
+ return -EINVAL;
+
+ bp = *this_cpu_ptr(pevent);
+ if (!bp)
+ return -EINVAL;
+
+ return modify_local_hw_breakpoint_addr(bp, addr);
+}
+
+static void wprobe_smp_update_func(void *info)
+{
+ struct trace_wprobe *tw = info;
+ unsigned long addr = READ_ONCE(tw->addr);
+
+ trace_wprobe_update_local(tw, addr);
+}
+
+static void wprobe_work_func(struct work_struct *work)
+{
+ struct trace_wprobe *tw = container_of(work, struct trace_wprobe, work);
+
+ on_each_cpu(wprobe_smp_update_func, tw, true);
+ /*
+ * Clear work_pending after all CPUs have updated their local debug
+ * registers. A new trigger may now update tw->addr and queue a new
+ * irq_work.
+ */
+ atomic_set(&tw->work_pending, 0);
+}
+
+static void wprobe_irq_work_func(struct irq_work *irq_work)
+{
+ struct trace_wprobe *tw = container_of(irq_work, struct trace_wprobe, irq_work);
+
+ schedule_work(&tw->work);
+}
+
static void free_trace_wprobe(struct trace_wprobe *tw)
{
if (tw) {
+ irq_work_sync(&tw->irq_work);
+ cancel_work_sync(&tw->work);
trace_probe_cleanup(&tw->tp);
kfree(tw->symbol);
kfree(tw);
@@ -237,6 +303,11 @@ static struct trace_wprobe *alloc_trace_wprobe(const char *group,
tw->addr = addr;
tw->len = len;
tw->type = type;
+ raw_spin_lock_init(&tw->lock);
+ init_irq_work(&tw->irq_work, wprobe_irq_work_func);
+ INIT_WORK(&tw->work, wprobe_work_func);
+ atomic_set(&tw->missed, 0);
+ atomic_set(&tw->work_pending, 0);
ret = trace_probe_init(&tw->tp, event, group, false, nargs);
if (ret < 0)
@@ -763,3 +834,459 @@ static __init int init_wprobe_trace(void)
}
fs_initcall(init_wprobe_trace);
+#ifdef CONFIG_WPROBE_TRIGGERS
+
+static int wprobe_trigger_global_enabled;
+
+#define SET_WPROBE_STR "set_wprobe"
+#define CLEAR_WPROBE_STR "clear_wprobe"
+#define WPROBE_DEFAULT_CLEAR_ADDRESS ((unsigned long)&wprobe_trigger_global_enabled)
+#define wprobe_trigger_log_err(file, glob, offs, err) \
+ tracing_log_err((file)->tr, "wprobe_trigger", glob, trace_probe_err_text, TP_ERR_##err, offs)
+
+struct wprobe_trigger_data {
+ struct rcu_head rcu;
+ struct trace_event_file *file;
+ struct trace_wprobe *tw;
+ int offset;
+ long adjust;
+ const char *field;
+ bool clear;
+};
+
+static void wprobe_trigger(struct event_trigger_data *data,
+ struct trace_buffer *buffer, void *rec,
+ struct ring_buffer_event *event)
+{
+ struct wprobe_trigger_data *wprobe_data = data->private_data;
+ struct trace_wprobe *tw = wprobe_data->tw;
+ unsigned long addr, flags;
+ bool changed = false;
+
+ if (in_nmi()) {
+ atomic_inc(&tw->missed);
+ return;
+ }
+
+ addr = *(unsigned long *)((char *)rec + wprobe_data->offset);
+ addr += wprobe_data->adjust;
+
+ raw_spin_lock_irqsave(&tw->lock, flags);
+
+ /* count < 0 means endless, 0 means trigger count exhausted */
+ if (!data->count)
+ goto out;
+
+ if (!wprobe_data->clear) {
+ if (addr < TASK_SIZE) {
+ atomic_inc(&tw->missed);
+ goto out;
+ }
+ if (tw->addr == WPROBE_DEFAULT_CLEAR_ADDRESS) {
+ /* Skip if a previous work is still propagating the address */
+ if (atomic_read(&tw->work_pending)) {
+ atomic_inc(&tw->missed);
+ goto out;
+ }
+ WRITE_ONCE(tw->addr, addr);
+ changed = true;
+ clear_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &wprobe_data->file->flags);
+ }
+ } else {
+ if (tw->addr != WPROBE_DEFAULT_CLEAR_ADDRESS) {
+ /* Skip if a previous work is still propagating the address */
+ if (atomic_read(&tw->work_pending)) {
+ atomic_inc(&tw->missed);
+ goto out;
+ }
+ if (!wprobe_data->field || tw->addr == addr) {
+ WRITE_ONCE(tw->addr, WPROBE_DEFAULT_CLEAR_ADDRESS);
+ changed = true;
+ set_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &wprobe_data->file->flags);
+ }
+ }
+ }
+
+ if (changed) {
+ if (data->count > 0)
+ data->count--;
+
+ /*
+ * Mark the work as pending before queuing irq_work so that
+ * subsequent triggers skip updating tw->addr until the work
+ * has finished propagating the address to all CPUs.
+ */
+ atomic_set(&tw->work_pending, 1);
+ irq_work_queue(&tw->irq_work);
+ }
+
+out:
+ raw_spin_unlock_irqrestore(&tw->lock, flags);
+}
+
+static void free_wprobe_trigger_data(struct wprobe_trigger_data *wprobe_data)
+{
+ if (wprobe_data) {
+ kfree(wprobe_data->field);
+ kfree(wprobe_data);
+ }
+}
+DEFINE_FREE(free_wprobe_trigger_data, struct wprobe_trigger_data *, free_wprobe_trigger_data(_T));
+
+static void free_private_wprobe_trigger_data(struct event_trigger_data *data)
+{
+ free_wprobe_trigger_data(data->private_data);
+}
+
+static int wprobe_trigger_print(struct seq_file *m,
+ struct event_trigger_data *data)
+{
+ struct wprobe_trigger_data *wprobe_data = data->private_data;
+
+ if (wprobe_data->clear) {
+ seq_printf(m, "%s:%s", CLEAR_WPROBE_STR,
+ trace_event_name(wprobe_data->file->event_call));
+ if (wprobe_data->field) {
+ seq_printf(m, ":%s%+ld",
+ wprobe_data->field, wprobe_data->adjust);
+ }
+ } else {
+ seq_printf(m, "%s:%s:%s%+ld", SET_WPROBE_STR,
+ trace_event_name(wprobe_data->file->event_call),
+ wprobe_data->field, wprobe_data->adjust);
+ }
+
+ if (data->count == -1)
+ seq_puts(m, ":unlimited");
+ else
+ seq_printf(m, ":count=%ld", data->count);
+
+ if (data->filter_str)
+ seq_printf(m, " if %s\n", data->filter_str);
+ else
+ seq_putc(m, '\n');
+
+ return 0;
+}
+
+static struct wprobe_trigger_data *
+wprobe_trigger_alloc(struct trace_wprobe *tw, struct trace_event_file *file,
+ bool clear)
+{
+ struct wprobe_trigger_data *wprobe_data;
+
+ wprobe_data = kzalloc_obj(*wprobe_data);
+ if (!wprobe_data)
+ return NULL;
+
+ wprobe_data->tw = tw;
+ wprobe_data->clear = clear;
+ wprobe_data->file = file;
+
+ return wprobe_data;
+}
+
+static void wprobe_trigger_free(struct event_trigger_data *data)
+{
+ struct wprobe_trigger_data *wprobe_data = data->private_data;
+
+ if (WARN_ON_ONCE(data->ref <= 0))
+ return;
+
+ data->ref--;
+ if (!data->ref) {
+ /* Remove the SOFT_MODE flag */
+ trace_event_enable_disable(wprobe_data->file, 0, 1);
+ trace_event_put_ref(wprobe_data->file->event_call);
+ trigger_data_free(data);
+ }
+}
+
+static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
+ struct trace_event_file *file,
+ char *glob, char *cmd,
+ char *param_and_filter)
+{
+ /*
+ * set_wprobe:EVENT:FIELD[+OFFS]
+ * clear_wprobe:EVENT[:FIELD[+OFFS]]
+ */
+ struct wprobe_trigger_data *wprobe_data __free(free_wprobe_trigger_data) = NULL;
+ struct event_trigger_data *trigger_data __free(kfree) = NULL;
+ char *event_str, *field_str, *count_str, *comment;
+ struct trace_event_file *wprobe_file;
+ struct trace_array *tr = file->tr;
+ struct trace_event_call *event;
+ bool remove, clear = false;
+ struct trace_wprobe *tw;
+ char *param, *filter;
+ int ret;
+
+ remove = event_trigger_check_remove(glob);
+
+ if (!strcmp(cmd, CLEAR_WPROBE_STR))
+ clear = true;
+
+ if (param_and_filter) {
+ if (*(param_and_filter - 1) == '\0')
+ *(param_and_filter - 1) = ':';
+ comment = strchr(param_and_filter, '#');
+ if (comment)
+ *comment = '\0';
+ }
+
+ if (event_trigger_empty_param(param_and_filter)) {
+ wprobe_trigger_log_err(file, glob, strlen(cmd) + 1, WPROBE_NOT_FOUND);
+ return -EINVAL;
+ }
+
+ ret = event_trigger_separate_filter(param_and_filter, ¶m, &filter, true);
+ if (ret)
+ return ret;
+
+ if (file->event_call->flags & TRACE_EVENT_FL_KPROBE) {
+ wprobe_trigger_log_err(file, glob, 0, WPROBE_ON_KPROBE);
+ return -EOPNOTSUPP;
+ }
+
+ event_str = strsep(¶m, ":");
+
+ /* Find target wprobe */
+ tw = find_trace_wprobe(event_str, WPROBE_EVENT_SYSTEM);
+ if (!tw) {
+ wprobe_trigger_log_err(file, glob, event_str - glob, WPROBE_NOT_FOUND);
+ return -ENOENT;
+ }
+ /* The target wprobe must not be used (unless clear) */
+ if (!remove && !clear && trace_probe_is_enabled(&tw->tp)) {
+ wprobe_trigger_log_err(file, glob, event_str - glob, WPROBE_BUSY);
+ return -EBUSY;
+ }
+
+ wprobe_file = find_event_file(tr, WPROBE_EVENT_SYSTEM, event_str);
+ if (!wprobe_file) {
+ wprobe_trigger_log_err(file, glob, event_str - glob, WPROBE_NOT_FOUND);
+ return -EINVAL;
+ }
+
+ wprobe_data = wprobe_trigger_alloc(tw, wprobe_file, clear);
+ if (!wprobe_data)
+ return -ENOMEM;
+
+ /* clear_wprobe does not need field. */
+ if (!clear) {
+ char *offs;
+
+ /* Find target field, which must be equivarent to "void *" */
+ field_str = strsep(¶m, ":");
+ if (!field_str) {
+ wprobe_trigger_log_err(file, glob, strlen(glob), WPROBE_NEED_FIELD);
+ return -EINVAL;
+ }
+
+ offs = strpbrk(field_str, "+-");
+ if (offs) {
+ long val;
+
+ if (kstrtol(offs, 0, &val) < 0) {
+ wprobe_trigger_log_err(file, glob, offs - glob, BAD_DEREF_OFFS);
+ return -EINVAL;
+ }
+ wprobe_data->adjust = val;
+ *offs = '\0';
+ }
+
+ event = file->event_call;
+ field = trace_find_event_field(event, field_str);
+ if (!field) {
+ wprobe_trigger_log_err(file, glob, field_str - glob, NO_EVENT_FIELD);
+ return -ENOENT;
+ }
+
+ if (field->size != sizeof(void *)) {
+ wprobe_trigger_log_err(file, glob, field_str - glob, WPROBE_BAD_FIELD);
+ return -ENOEXEC;
+ }
+ wprobe_data->offset = field->offset;
+ wprobe_data->field = kstrdup(field_str, GFP_KERNEL);
+ if (!wprobe_data->field)
+ return -ENOMEM;
+ }
+
+ trigger_data = trigger_data_alloc(cmd_ops, cmd, param, wprobe_data);
+ if (!trigger_data)
+ return -ENOMEM;
+
+ trigger_data->private_data_free = free_private_wprobe_trigger_data;
+
+ if (remove) {
+ event_trigger_unregister(cmd_ops, file, glob+1, trigger_data);
+ return 0;
+ }
+
+ ret = event_trigger_parse_num(param, trigger_data);
+ if (ret) {
+ wprobe_trigger_log_err(file, glob, param - glob, BAD_IMM);
+ return ret;
+ }
+
+ ret = event_trigger_set_filter(cmd_ops, file, filter, trigger_data);
+ if (ret < 0)
+ return ret;
+
+ /* Soft-enable (register) wprobe event on WPROBE_DEFAULT_CLEAR_ADDRESS */
+ if (!trace_event_try_get_ref(wprobe_file->event_call)) {
+ event_trigger_reset_filter(cmd_ops, trigger_data);
+ return -ENODEV;
+ }
+
+ ret = trace_event_enable_disable(wprobe_file, 1, 1);
+ if (ret < 0) {
+ trace_event_put_ref(wprobe_file->event_call);
+ event_trigger_reset_filter(cmd_ops, trigger_data);
+ return ret;
+ }
+
+ if (!clear)
+ WRITE_ONCE(tw->addr, WPROBE_DEFAULT_CLEAR_ADDRESS);
+
+ event_trigger_init(trigger_data);
+
+ ret = event_trigger_register(cmd_ops, file, glob, trigger_data);
+ if (ret) {
+ event_trigger_reset_filter(cmd_ops, trigger_data);
+ trace_event_enable_disable(wprobe_file, 0, 1);
+ trace_event_put_ref(wprobe_file->event_call);
+ tracepoint_synchronize_unregister();
+ event_trigger_free(trigger_data);
+ return ret;
+ }
+ /* Make it NULL to avoid freeing trigger_data and wprobe_data by __free() */
+ wprobe_data = NULL;
+ event_trigger_free(trigger_data);
+ trigger_data = NULL;
+
+ return 0;
+}
+
+/* Return event_trigger_data if there is a trigger which points the same wprobe */
+static struct event_trigger_data *
+wprobe_trigger_find_same(struct event_trigger_data *test,
+ struct trace_event_file *file)
+{
+ struct wprobe_trigger_data *test_wprobe_data = test->private_data;
+ struct wprobe_trigger_data *wprobe_data;
+ struct event_trigger_data *iter;
+
+ list_for_each_entry(iter, &file->triggers, list) {
+ wprobe_data = iter->private_data;
+ if (!wprobe_data ||
+ iter->cmd_ops->trigger_type !=
+ test->cmd_ops->trigger_type)
+ continue;
+ if (wprobe_data->tw == test_wprobe_data->tw &&
+ wprobe_data->clear == test_wprobe_data->clear)
+ return iter;
+ }
+ return NULL;
+}
+
+static int wprobe_register_trigger(char *glob,
+ struct event_trigger_data *data,
+ struct trace_event_file *file)
+{
+ int ret = 0;
+
+ lockdep_assert_held(&event_mutex);
+
+ /* The same wprobe is not accept on the same file (event) */
+ if (wprobe_trigger_find_same(data, file))
+ return -EEXIST;
+
+ if (data->cmd_ops->init) {
+ ret = data->cmd_ops->init(data);
+ if (ret < 0)
+ return ret;
+ }
+
+ list_add_rcu(&data->list, &file->triggers);
+
+ update_cond_flag(file);
+ ret = trace_event_trigger_enable_disable(file, 1);
+ if (ret < 0) {
+ list_del_rcu(&data->list);
+ update_cond_flag(file);
+ }
+ return ret;
+}
+
+static void wprobe_unregister_trigger(char *glob,
+ struct event_trigger_data *test,
+ struct trace_event_file *file)
+{
+ struct event_trigger_data *data;
+
+ lockdep_assert_held(&event_mutex);
+
+ data = wprobe_trigger_find_same(test, file);
+ if (!data)
+ return;
+
+ list_del_rcu(&data->list);
+ trace_event_trigger_enable_disable(file, 0);
+ update_cond_flag(file);
+ tracepoint_synchronize_unregister();
+ if (data->cmd_ops->free)
+ data->cmd_ops->free(data);
+}
+
+static struct event_command trigger_wprobe_set_cmd = {
+ .name = SET_WPROBE_STR,
+ .trigger_type = ETT_EVENT_WPROBE,
+ /* This triggers after when the event is recorded. */
+ .flags = EVENT_CMD_FL_NEEDS_REC,
+ .parse = wprobe_trigger_cmd_parse,
+ .reg = wprobe_register_trigger,
+ .unreg = wprobe_unregister_trigger,
+ .set_filter = set_trigger_filter,
+ .trigger = wprobe_trigger,
+ .count_func = event_trigger_count,
+ .print = wprobe_trigger_print,
+ .init = event_trigger_init,
+ .free = wprobe_trigger_free,
+};
+
+static struct event_command trigger_wprobe_clear_cmd = {
+ .name = CLEAR_WPROBE_STR,
+ .trigger_type = ETT_EVENT_WPROBE,
+ /* This triggers after when the event is recorded. */
+ .flags = EVENT_CMD_FL_NEEDS_REC,
+ .parse = wprobe_trigger_cmd_parse,
+ .reg = wprobe_register_trigger,
+ .unreg = wprobe_unregister_trigger,
+ .set_filter = set_trigger_filter,
+ .trigger = wprobe_trigger,
+ .count_func = event_trigger_count,
+ .print = wprobe_trigger_print,
+ .init = event_trigger_init,
+ .free = wprobe_trigger_free,
+};
+
+static __init int init_trigger_wprobe_cmds(void)
+{
+ int ret;
+
+ ret = register_event_command(&trigger_wprobe_set_cmd);
+ if (WARN_ON(ret < 0))
+ return ret;
+ ret = register_event_command(&trigger_wprobe_clear_cmd);
+ if (WARN_ON(ret < 0))
+ unregister_event_command(&trigger_wprobe_set_cmd);
+
+ if (!ret)
+ wprobe_trigger_global_enabled = 1;
+
+ return ret;
+}
+fs_initcall(init_trigger_wprobe_cmds);
+#endif /* CONFIG_WPROBE_TRIGGERS */
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 08/11] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers
2026-08-07 15:34 ` [PATCH v12 08/11] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers Masami Hiramatsu (Google)
@ 2026-08-07 15:54 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 15:54 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 5 potential issue(s) to consider:
- [High] Mixing scope-based cleanup (`__free()`) with explicit resource management causes double-frees and memory leaks.
- [High] Hardware breakpoint fallback address spans adjacent kernel variables, causing NMI storms.
- [High] Double initialization of the event trigger reference count causes unconditional memory leaks.
- [Medium] Trigger `count` is double-decremented, breaking the `:count=1` feature.
- [Medium] The `clear_wprobe` trigger fails to parse its documented optional FIELD parameter.
--
--- Patch [8]: [PATCH v12 08/11] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers ---
commit cb59d7e61a8c78c308ca9ae32e1de403900f7a05
Author: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Subject: tracing/wprobe: Add set_wprobe and clear_wprobe event triggers
This commit adds set_wprobe and clear_wprobe event triggers to dynamically
attach and detach hardware breakpoint address monitoring based on event field
contents.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
> --- a/kernel/trace/trace_wprobe.c
> +++ b/kernel/trace/trace_wprobe.c
[ ... ]
> +static int wprobe_trigger_global_enabled;
> +
> +#define SET_WPROBE_STR "set_wprobe"
> +#define CLEAR_WPROBE_STR "clear_wprobe"
> +#define WPROBE_DEFAULT_CLEAR_ADDRESS ((unsigned long)&wprobe_trigger_global_enabled)
[Severity: High]
Will this 4-byte fallback variable safely accommodate 8-byte hardware
breakpoints?
When a wprobe is temporarily disabled, its hardware breakpoint is pointed here
at WPROBE_DEFAULT_CLEAR_ADDRESS, but it retains its configured length (e.g., 8
bytes for a 64-bit pointer).
Could this cause the breakpoint to monitor the 4-byte variable plus 4 adjacent
bytes in the .bss section, potentially triggering NMI storms if a frequently
accessed kernel variable happens to reside in those adjacent bytes?
[ ... ]
> +static void wprobe_trigger(struct event_trigger_data *data,
> + struct trace_buffer *buffer, void *rec,
> + struct ring_buffer_event *event)
> +{
[ ... ]
> + /* count < 0 means endless, 0 means trigger count exhausted */
> + if (!data->count)
> + goto out;
[ ... ]
> + if (changed) {
> + if (data->count > 0)
> + data->count--;
[ ... ]
> +static struct event_command trigger_wprobe_set_cmd = {
> + .name = SET_WPROBE_STR,
[ ... ]
> + .trigger = wprobe_trigger,
> + .count_func = event_trigger_count,
[Severity: Medium]
Does this manual decrement in wprobe_trigger() conflict with the core
event_trigger_count decrement function?
The trigger commands are assigned .count_func = event_trigger_count, which
unconditionally executes before the main .trigger callback and pre-decrements
the count.
If a user specifies :count=1, the core framework reduces it to 0.
wprobe_trigger() will then see a count of 0 and abort immediately. Does this
render a single-count trigger completely non-functional?
[ ... ]
> +static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
> + struct trace_event_file *file,
> + char *glob, char *cmd,
> + char *param_and_filter)
> +{
> + /*
> + * set_wprobe:EVENT:FIELD[+OFFS]
> + * clear_wprobe:EVENT[:FIELD[+OFFS]]
> + */
> + struct wprobe_trigger_data *wprobe_data __free(free_wprobe_trigger_data) = NULL;
> + struct event_trigger_data *trigger_data __free(kfree) = NULL;
[ ... ]
> + /* clear_wprobe does not need field. */
> + if (!clear) {
> + char *offs;
[Severity: Medium]
Does skipping the field parsing here break the documented optional FIELD
parameter for clear_wprobe?
The documentation claims clear_wprobe:WPEVENT[:FIELD] conditionally clears a
watchpoint, but if clear is true, the parsing code skips extracting the field
string.
It looks like the unparsed string is then passed to event_trigger_parse_num(),
which expects a numeric string and fails with -EINVAL.
[ ... ]
> + ret = event_trigger_parse_num(param, trigger_data);
> + if (ret) {
> + wprobe_trigger_log_err(file, glob, param - glob, BAD_IMM);
> + return ret;
> + }
[Severity: High]
Can this early error path leak memory?
The trigger_data uses __free(kfree), which performs a shallow free. If
event_trigger_parse_num() fails and returns early, does this permanently leak
internal allocations like data->name?
[ ... ]
> + event_trigger_init(trigger_data);
> +
> + ret = event_trigger_register(cmd_ops, file, glob, trigger_data);
[Severity: High]
Will this result in a double initialization of the reference count?
We call event_trigger_init(trigger_data) explicitly here, which increments the
reference count. Then event_trigger_register() calls the .reg callback
wprobe_register_trigger(), which executes data->cmd_ops->init(data) and
increments the reference count a second time.
When the trigger is later unregistered, wprobe_trigger_free() decrements the
count by only 1, leaving a permanent reference and memory leak.
> + if (ret) {
> + event_trigger_reset_filter(cmd_ops, trigger_data);
> + trace_event_enable_disable(wprobe_file, 0, 1);
> + trace_event_put_ref(wprobe_file->event_call);
> + tracepoint_synchronize_unregister();
> + event_trigger_free(trigger_data);
> + return ret;
> + }
[Severity: High]
Can this late error path cause a double-free?
Both trigger_data and wprobe_data are declared with scope-based cleanup
(__free). If event_trigger_register() fails, the code manually calls
event_trigger_free(trigger_data), freeing both objects.
When the function returns, the __free macros will execute and attempt to free
them a second time.
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=8
^ permalink raw reply [flat|nested] 19+ messages in thread
* [PATCH v12 09/11] selftests: ftrace: Add wprobe trigger testcase
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (7 preceding siblings ...)
2026-08-07 15:34 ` [PATCH v12 08/11] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers Masami Hiramatsu (Google)
@ 2026-08-07 15:34 ` Masami Hiramatsu (Google)
2026-08-07 15:53 ` sashiko-bot
2026-08-07 15:35 ` [PATCH v12 10/11] tracing/wprobe: Support BTF typecast in fetchargs Masami Hiramatsu (Google)
2026-08-07 15:35 ` [PATCH v12 11/11] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger Masami Hiramatsu (Google)
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:34 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Add a testcase for checking wprobe trigger. This sets set_wprobe and
clear_wprobe triggers on fprobe event and static trace event to
monitor memory accesses within the trace-events-sample kernel module.
Also add a testcase for verifying wprobe trigger syntax error logging.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v12:
- Add trigger-wprobe-syntax-errors.tc for verifying wprobe trigger syntax
error logging.
- Fix requires line in trigger-wprobe-syntax-errors.tc so test is not
evaluated as unsupported prior to execution.
- Define dfd=$arg1 fetcharg on testevent for trigger syntax checks.
Changes in v11:
- Update testcase to use trace-events-sample kernel module instead of
VFS file operations.
---
tools/testing/selftests/ftrace/config | 1
.../test.d/trigger/trigger-wprobe-syntax-errors.tc | 31 +++++++
.../ftrace/test.d/trigger/trigger-wprobe.tc | 85 ++++++++++++++++++++
3 files changed, 117 insertions(+)
create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc
create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc
diff --git a/tools/testing/selftests/ftrace/config b/tools/testing/selftests/ftrace/config
index d2f503722020..ecdee77f360f 100644
--- a/tools/testing/selftests/ftrace/config
+++ b/tools/testing/selftests/ftrace/config
@@ -28,3 +28,4 @@ CONFIG_TRACER_SNAPSHOT=y
CONFIG_UPROBES=y
CONFIG_UPROBE_EVENTS=y
CONFIG_WPROBE_EVENTS=y
+CONFIG_WPROBE_TRIGGERS=y
diff --git a/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc
new file mode 100644
index 000000000000..7e02313cf1c7
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc
@@ -0,0 +1,31 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: event trigger - test wprobe trigger syntax errors
+# requires: dynamic_events error_log "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README "f[:[<group>/][<event>]] <func-name>[%return] [<args>]":README
+
+check_error() { # command-with-error-pos-by-^
+ ftrace_errlog_check "wprobe_trigger" "$1" "events/fprobes/testevent/trigger"
+}
+
+# Add a dummy fprobe event to attach triggers to
+echo 'f:fprobes/testevent do_sys_open dfd=$arg1' > dynamic_events
+
+# Add a target wprobe event
+echo 'w:watch rw@0:8' >> dynamic_events
+
+# Test errors on trigger syntax
+check_error 'set_wprobe:^non_exist_wprobe:dfd' # WPROBE_NOT_FOUND
+check_error 'set_wprobe:^' # WPROBE_NOT_FOUND
+check_error 'set_wprobe:watch^' # WPROBE_NEED_FIELD
+check_error 'set_wprobe:watch:^non_exist_field' # NO_EVENT_FIELD
+
+# Enable target wprobe event and test WPROBE_BUSY error
+echo 1 > events/wprobes/watch/enable
+check_error 'set_wprobe:^watch:dfd' # WPROBE_BUSY
+echo 0 > events/wprobes/watch/enable
+
+# Cleanup
+echo '-:watch' >> dynamic_events
+echo '-:testevent' >> dynamic_events
+
+exit 0
diff --git a/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc
new file mode 100644
index 000000000000..4071fe0c878b
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc
@@ -0,0 +1,85 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: event trigger - test wprobe trigger
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README events/sched/sched_process_fork/trigger
+
+rmmod trace-events-sample ||:
+if ! modprobe trace-events-sample ; then
+ echo "No trace-events sample module - please make CONFIG_SAMPLE_TRACE_EVENTS=m"
+ exit_unresolved
+fi
+
+cleanup_wprobe_triggers() {
+ if [ -f events/fprobes/testevent/trigger ]; then
+ reset_trigger_file events/fprobes/testevent/trigger || true
+ fi
+ if [ -f events/sample-trace/foo_bar_with_fn/trigger ]; then
+ reset_trigger_file events/sample-trace/foo_bar_with_fn/trigger || true
+ fi
+ echo 0 > events/enable 2>/dev/null || true
+ echo > dynamic_events 2>/dev/null || true
+ sleep 1
+ rmmod trace-events-sample 2>/dev/null || true
+ return 0
+}
+
+trap cleanup_wprobe_triggers EXIT
+
+echo 0 > tracing_on
+
+# we will skip this test if fprobe is not supported.
+if ! grep -Fq "f[:[<group>/][<event>]] <func-name>[%return] [<args>]" README; then
+ echo "UNRESOLVED: fprobe is not supported"
+ exit_unresolved
+fi
+
+# we will skip this test if the target function does not exist.
+if ! grep -wq "sample_timer_cb" /proc/kallsyms; then
+ echo "UNRESOLVED: sample_timer_cb not found"
+ exit_unresolved
+fi
+
+:;: "Add a wprobe event used by trigger" ;:
+echo 'w:watch rw@0:8 address=$addr value=$value' > dynamic_events
+
+:;: "Add events for triggering wprobe" ;:
+echo 'f:fprobes/testevent sample_timer_cb timer=t' >> dynamic_events
+
+:;: "Enable all events before setting triggers" ;:
+echo 1 > tracing_on
+echo 1 >> events/fprobes/testevent/enable
+echo 1 >> events/sample-trace/foo_bar_with_fn/enable
+
+:;: "Set set_wprobe trigger on testevent" ;:
+echo 'set_wprobe:watch:timer' >> events/fprobes/testevent/trigger
+cat events/fprobes/testevent/trigger | grep ^set_wprobe
+
+# Wait for sample_timer_cb to fire and set_wprobe trigger to activate
+sleep 2
+
+:;: "Check set_wprobe trigger activated the watchpoint" ;:
+cat trace | grep watch
+
+:;: "Set clear_wprobe trigger on foo_bar_with_fn" ;:
+echo 'clear_wprobe:watch' >> events/sample-trace/foo_bar_with_fn/trigger
+cat events/sample-trace/foo_bar_with_fn/trigger | grep ^clear_wprobe
+
+# Clear trace and wait to ensure no new watchpoint events are generated
+clear_trace
+sleep 1
+
+:;: "Ensure clear_wprobe trigger deactivated the watchpoint" ;:
+! grep -q watch trace
+
+:;: "Remove wprobe triggers" ;:
+echo '!set_wprobe:watch:timer' >> events/fprobes/testevent/trigger
+echo '!clear_wprobe:watch' >> events/sample-trace/foo_bar_with_fn/trigger
+! grep ^set_wprobe events/fprobes/testevent/trigger
+! grep ^clear_wprobe events/sample-trace/foo_bar_with_fn/trigger
+
+:;: "Disable events and remove dynamic events" ;:
+echo 0 > events/enable
+echo > dynamic_events
+clear_trace
+
+exit 0
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 09/11] selftests: ftrace: Add wprobe trigger testcase
2026-08-07 15:34 ` [PATCH v12 09/11] selftests: ftrace: Add wprobe trigger testcase Masami Hiramatsu (Google)
@ 2026-08-07 15:53 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 15:53 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
- [Medium] Race condition in `trigger-wprobe.tc` causes spurious test failures because `set_wprobe` trigger is not removed before verifying `clear_wprobe`.
- [Medium] `trigger-wprobe-syntax-errors.tc` relies on `do_sys_open` without verifying its existence, causing the test to fail on kernels where the function is unavailable.
--
--- Patch [9]: [PATCH v12 09/11] selftests: ftrace: Add wprobe trigger testcase ---
commit d811031bfa18879b95fa2051a73496d3fcc8f562
Author: Masami Hiramatsu (Google) <mhiramat@kernel.org>
selftests: ftrace: Add wprobe trigger testcase
This patch adds a testcase for checking the wprobe trigger. It sets
set_wprobe and clear_wprobe triggers on fprobe events and static trace events
to monitor memory accesses. It also includes a testcase to verify wprobe
trigger syntax error logging.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc
> new file mode 100644
> index 0000000000000..7e02313cf1c73
> --- /dev/null
> +++ b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-syntax-errors.tc
> @@ -0,0 +1,31 @@
> +#!/bin/sh
> +# SPDX-License-Identifier: GPL-2.0
> +# description: event trigger - test wprobe trigger syntax errors
> +# requires: dynamic_events error_log "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README "f[:[<group>/][<event>]] <func-name>[%return] [<args>]":README
> +
> +check_error() { # command-with-error-pos-by-^
> + ftrace_errlog_check "wprobe_trigger" "$1" "events/fprobes/testevent/trigger"
> +}
> +
> +# Add a dummy fprobe event to attach triggers to
> +echo 'f:fprobes/testevent do_sys_open dfd=$arg1' > dynamic_events
[Severity: Medium]
Will this script fail unconditionally on kernels where do_sys_open is
unavailable as a valid fprobe target?
On many modern kernels, do_sys_open has been removed, renamed (e.g., to
do_sys_openat2), or heavily inlined, which causes this echo command to
dynamic_events to fail with an error.
Could we check against /proc/kallsyms before using it, similar to how
sample_timer_cb is checked in trigger-wprobe.tc?
[ ... ]
> diff --git a/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc
> new file mode 100644
> index 0000000000000..4071fe0c878b9
> --- /dev/null
> +++ b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe.tc
[ ... ]
> +:;: "Set clear_wprobe trigger on foo_bar_with_fn" ;:
> +echo 'clear_wprobe:watch' >> events/sample-trace/foo_bar_with_fn/trigger
> +cat events/sample-trace/foo_bar_with_fn/trigger | grep ^clear_wprobe
> +
> +# Clear trace and wait to ensure no new watchpoint events are generated
> +clear_trace
> +sleep 1
> +
> +:;: "Ensure clear_wprobe trigger deactivated the watchpoint" ;:
> +! grep -q watch trace
[Severity: Medium]
Is there a race condition here that could cause spurious test failures?
Since the set_wprobe trigger on testevent (sample_timer_cb) is still active
and fires concurrently with foo_bar_with_fn, if sample_timer_cb fires after
foo_bar_with_fn during this 1-second window, the wprobe could be re-armed.
Memory accesses would then generate new watch events, causing this assertion
to fail.
Should the set_wprobe trigger be removed before verifying the clear_wprobe
behavior?
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=9
^ permalink raw reply [flat|nested] 19+ messages in thread
* [PATCH v12 10/11] tracing/wprobe: Support BTF typecast in fetchargs
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (8 preceding siblings ...)
2026-08-07 15:34 ` [PATCH v12 09/11] selftests: ftrace: Add wprobe trigger testcase Masami Hiramatsu (Google)
@ 2026-08-07 15:35 ` Masami Hiramatsu (Google)
2026-08-07 15:35 ` [PATCH v12 11/11] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger Masami Hiramatsu (Google)
10 siblings, 0 replies; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:35 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Allow BTF typecast syntax (STRUCT)FETCHARG->MEMBER in wprobe event
fetchargs. Previously, handle_typecast() rejected any probe context
that was not a function entry/return or tracepoint event probe.
Wprobe events use $addr (the accessed address) and $value (the value
at that address). By enabling BTF typecast, users can now cast these
to a concrete struct type and access its fields directly. For example:
echo 'w:watch rw@0:8 dflag=(dentry)$addr->d_flags' >> dynamic_events
With a set_wprobe trigger pointing the watchpoint at a dentry address,
the resulting trace shows d_flags being accessed at that location.
Note that $addr and $value are restricted to kernel-space memory,
which is consistent with the existing TPARG_FL_KERNEL flag used when
parsing wprobe fetchargs.
Assisted-by: Antigravity:gemini-3.5-flash
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v11:
- Update trigger-wprobe-btf-typecast.tc to use trace-events-sample
kernel module.
- Fix commit comment.
Changes in v9:
- Newly added.
---
kernel/trace/trace_probe.c | 13 +++-
kernel/trace/trace_probe.h | 5 +
tools/testing/selftests/ftrace/config | 1
.../test.d/trigger/trigger-wprobe-btf-typecast.tc | 72 ++++++++++++++++++++
4 files changed, 90 insertions(+), 1 deletion(-)
create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-typecast.tc
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 36e07275a04d..3209dfb2509a 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -889,6 +889,16 @@ static int query_btf_struct(const char *sname, struct traceprobe_parse_context *
ctx->struct_btf = NULL;
}
+ if (ctx->btf) {
+ id = btf_find_by_name_kind(ctx->btf, sname, BTF_KIND_STRUCT);
+ if (id > 0) {
+ btf_get(ctx->btf);
+ ctx->struct_btf = ctx->btf;
+ ctx->last_struct = btf_type_by_id(ctx->struct_btf, id);
+ return 0;
+ }
+ }
+
id = bpf_find_btf_id(sname, BTF_KIND_STRUCT, &btf);
if (id < 0)
return id;
@@ -964,7 +974,8 @@ static int handle_typecast(char *arg, struct traceprobe_parse_context *ctx)
if (!(tparg_is_event_probe(ctx->flags) ||
tparg_is_function_entry(ctx->flags) ||
- tparg_is_function_return(ctx->flags))) {
+ tparg_is_function_return(ctx->flags) ||
+ tparg_is_wprobe(ctx->flags))) {
trace_probe_log_err(ctx->offset, NOSUP_BTFARG);
return -EOPNOTSUPP;
}
diff --git a/kernel/trace/trace_probe.h b/kernel/trace/trace_probe.h
index e08f17c99138..8c495097acf2 100644
--- a/kernel/trace/trace_probe.h
+++ b/kernel/trace/trace_probe.h
@@ -437,6 +437,11 @@ static inline bool tparg_is_event_probe(unsigned int flags)
return !!(flags & TPARG_FL_TEVENT);
}
+static inline bool tparg_is_wprobe(unsigned int flags)
+{
+ return !!(flags & TPARG_FL_WPROBE);
+}
+
/* Each typecast consumes nested level. So the max number of typecast is 8. */
#define TRACEPROBE_MAX_NESTED_LEVEL 8
diff --git a/tools/testing/selftests/ftrace/config b/tools/testing/selftests/ftrace/config
index ecdee77f360f..f067874902ed 100644
--- a/tools/testing/selftests/ftrace/config
+++ b/tools/testing/selftests/ftrace/config
@@ -1,5 +1,6 @@
CONFIG_BPF_SYSCALL=y
CONFIG_DEBUG_INFO_BTF=y
+CONFIG_DEBUG_INFO_BTF_MODULES=y
CONFIG_DEBUG_INFO_DWARF4=y
CONFIG_EPROBE_EVENTS=y
CONFIG_FPROBE=y
diff --git a/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-typecast.tc b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-typecast.tc
new file mode 100644
index 000000000000..3f2bebb28837
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-typecast.tc
@@ -0,0 +1,72 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: event trigger - test wprobe trigger with BTF typecast fetchargs
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README events/sched/sched_process_fork/trigger "[(structname[,field])]<argname>[->field[->field|.field...]]":README
+
+rmmod trace-events-sample ||:
+if ! modprobe trace-events-sample ; then
+ echo "No trace-events sample module - please make CONFIG_SAMPLE_TRACE_EVENTS=m"
+ exit_unresolved
+fi
+
+cleanup_wprobe_triggers() {
+ if [ -f events/fprobes/testevent/trigger ]; then
+ reset_trigger_file events/fprobes/testevent/trigger || true
+ fi
+ echo 0 > events/enable 2>/dev/null || true
+ echo > dynamic_events 2>/dev/null || true
+ sleep 1
+ rmmod trace-events-sample 2>/dev/null || true
+ return 0
+}
+
+trap cleanup_wprobe_triggers EXIT
+
+echo 0 > tracing_on
+
+# we will skip this test if fprobe is not supported.
+if ! grep -Fq "f[:[<group>/][<event>]] <func-name>[%return] [<args>]" README; then
+ echo "UNRESOLVED: fprobe is not supported"
+ exit_unresolved
+fi
+
+# we will skip this test if the target function does not exist.
+if ! grep -wq "sample_timer_cb" /proc/kallsyms; then
+ echo "UNRESOLVED: sample_timer_cb not found"
+ exit_unresolved
+fi
+
+:;: "Add a wprobe event with BTF typecast fetchargs" ;:
+# (foo_timer_data,timer)$addr->counter reads counter from struct foo_timer_data via BTF typecast
+echo 'w:watch rw@0:8 address=$addr counter=(foo_timer_data,timer)$addr->counter' >> dynamic_events
+
+:;: "Check the wprobe event is registered with counter field" ;:
+grep -q "counter" dynamic_events
+
+:;: "Add fprobe event for sample_timer_cb" ;:
+echo 'f:fprobes/testevent sample_timer_cb timer=t' >> dynamic_events
+
+:;: "Enable all events before setting triggers" ;:
+echo 1 > tracing_on
+echo 1 >> events/fprobes/testevent/enable
+
+:;: "Set set_wprobe trigger on testevent" ;:
+echo 'set_wprobe:watch:timer' >> events/fprobes/testevent/trigger
+cat events/fprobes/testevent/trigger | grep ^set_wprobe
+
+# Wait for sample_timer_cb to fire and set_wprobe trigger to activate
+sleep 3
+
+:;: "Check set_wprobe trigger activated the watchpoint" ;:
+cat trace | grep watch
+
+:;: "Remove wprobe triggers" ;:
+echo '!set_wprobe:watch:timer' >> events/fprobes/testevent/trigger
+! grep ^set_wprobe events/fprobes/testevent/trigger
+
+:;: "Disable events and remove dynamic events" ;:
+echo 0 > events/enable
+echo > dynamic_events
+clear_trace
+
+exit 0
^ permalink raw reply related [flat|nested] 19+ messages in thread* [PATCH v12 11/11] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger
2026-08-07 15:33 [PATCH v12 00/11] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
` (9 preceding siblings ...)
2026-08-07 15:35 ` [PATCH v12 10/11] tracing/wprobe: Support BTF typecast in fetchargs Masami Hiramatsu (Google)
@ 2026-08-07 15:35 ` Masami Hiramatsu (Google)
2026-08-07 15:59 ` sashiko-bot
10 siblings, 1 reply; 19+ messages in thread
From: Masami Hiramatsu (Google) @ 2026-08-07 15:35 UTC (permalink / raw)
To: Steven Rostedt, Peter Zijlstra, Ingo Molnar, x86
Cc: Jinchao Wang, Mathieu Desnoyers, Masami Hiramatsu,
Thomas Gleixner, Borislav Petkov, Dave Hansen, H . Peter Anvin,
Alexander Shishkin, Ian Rogers, linux-kernel, linux-trace-kernel,
linux-doc, linux-perf-users
From: Masami Hiramatsu (Google) <mhiramat@kernel.org>
Allow set_wprobe trigger to use BTF struct offset resolution to specify
the target address field.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v12:
- Refactor field parsing logic into wprobe_trigger_field_parse().
- Remove unused variable count_str in wprobe_trigger_cmd_parse().
---
kernel/trace/trace_wprobe.c | 228 +++++++++++++++++---
.../test.d/trigger/trigger-wprobe-btf-offset.tc | 74 ++++++
2 files changed, 267 insertions(+), 35 deletions(-)
create mode 100644 tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-offset.tc
diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
index c64bbdc90a40..565c106a9380 100644
--- a/kernel/trace/trace_wprobe.c
+++ b/kernel/trace/trace_wprobe.c
@@ -27,6 +27,7 @@
#include <asm/ptrace.h>
#include "trace.h"
+#include "trace_btf.h"
#include "trace_dynevent.h"
#include "trace_probe.h"
#include "trace_probe_kernel.h"
@@ -962,9 +963,10 @@ static int wprobe_trigger_print(struct seq_file *m,
seq_printf(m, ":count=%ld", data->count);
if (data->filter_str)
- seq_printf(m, " if %s\n", data->filter_str);
- else
- seq_putc(m, '\n');
+ seq_printf(m, " if %s", data->filter_str);
+
+ seq_printf(m, " # offset:%d adjust:%ld\n",
+ wprobe_data->offset, wprobe_data->adjust);
return 0;
}
@@ -1002,6 +1004,181 @@ static void wprobe_trigger_free(struct event_trigger_data *data)
}
}
+#ifdef CONFIG_PROBE_EVENTS_BTF_ARGS
+
+static int get_offset_of_field(struct btf *btf, const struct btf_type *type, char *field_name)
+{
+ const struct btf_member *field;
+ int bitoffs = 0;
+ u32 anon_offs;
+ char *next;
+
+ do {
+ next = strchr(field_name, '.');
+ if (next)
+ *next++ = '\0';
+
+ field = btf_find_struct_member(btf, type, field_name, &anon_offs);
+ if (IS_ERR_OR_NULL(field))
+ return -ENOENT;
+
+ if (btf_type_kflag(type)) {
+ /* Reject bitfield member access */
+ if (BTF_MEMBER_BITFIELD_SIZE(field->offset))
+ return -EINVAL;
+ bitoffs += anon_offs + BTF_MEMBER_BIT_OFFSET(field->offset);
+ } else {
+ bitoffs += anon_offs + field->offset;
+ }
+
+ field_name = next;
+ if (next) {
+ type = btf_type_skip_modifiers(btf, field->type, NULL);
+ if (!type)
+ return -ENOENT;
+ }
+ } while (next);
+ return bitoffs / BITS_PER_BYTE;
+}
+
+/* btf_put(NULL) is acceptable. */
+DEFINE_FREE(btf_put, struct btf *, btf_put(_T))
+
+/* parse typecast: (TYPE[,ASGN])EVENT_FIELD->FIELD[.SUBFIELD...] and set adjust. */
+static int wprobe_trigger_typecast_parse(char **field_str_ptr,
+ struct trace_event_file *file,
+ struct wprobe_trigger_data *wprobe_data,
+ const char *glob)
+{
+ struct btf *btf __free(btf_put) = NULL;
+ const struct btf_type *type;
+ char *assign_field;
+ char *event_field;
+ char *type_field;
+ char *type_name;
+ char *offs;
+ long val = 0;
+ int id;
+ int adjust;
+
+ type_name = *field_str_ptr + 1;
+ event_field = strchr(type_name, ')');
+ if (!event_field) {
+ wprobe_trigger_log_err(file, glob, type_name - glob, DEREF_OPEN_BRACE);
+ return -EINVAL;
+ }
+ *event_field++ = '\0';
+
+ /* Check the optional assign field. */
+ assign_field = strchr(type_name, ',');
+ if (assign_field)
+ *assign_field++ = '\0';
+
+ /* Get the type field name. */
+ type_field = strstr(event_field, "->");
+ if (!type_field) {
+ wprobe_trigger_log_err(file, glob, event_field - glob, TYPECAST_REQ_FIELD);
+ return -EINVAL;
+ }
+ *type_field = '\0';
+ type_field += 2;
+
+ offs = strpbrk(type_field, "+-");
+ if (offs) {
+ if (kstrtol(offs, 0, &val) < 0) {
+ wprobe_trigger_log_err(file, glob, offs - glob, BAD_DEREF_OFFS);
+ return -EINVAL;
+ }
+ *offs = '\0';
+ }
+
+ /* find type from BTF */
+ id = bpf_find_btf_id(type_name, BTF_KIND_STRUCT, &btf);
+ if (id < 0) {
+ wprobe_trigger_log_err(file, glob, type_name - glob, BAD_BTF_TID);
+ return id;
+ }
+
+ type = btf_type_by_id(btf, id);
+ if (!type) {
+ wprobe_trigger_log_err(file, glob, type_name - glob, BAD_BTF_TID);
+ return -EINVAL;
+ }
+
+ adjust = get_offset_of_field(btf, type, type_field);
+ if (adjust < 0) {
+ wprobe_trigger_log_err(file, glob, type_field - glob, NO_BTF_FIELD);
+ return adjust;
+ }
+ wprobe_data->adjust = adjust + val;
+
+ if (assign_field) {
+ /* assign_field should be a struct field */
+ adjust = get_offset_of_field(btf, type, assign_field);
+ if (adjust < 0) {
+ wprobe_trigger_log_err(file, glob, assign_field - glob, NO_BTF_FIELD);
+ return adjust;
+ }
+ wprobe_data->adjust -= adjust;
+ }
+
+ *field_str_ptr = event_field;
+ return 0;
+}
+#else
+static int wprobe_trigger_typecast_parse(char **field_str_ptr,
+ struct trace_event_file *file,
+ struct wprobe_trigger_data *wprobe_data,
+ const char *glob)
+{
+ wprobe_trigger_log_err(file, glob, *field_str_ptr - glob, NOSUP_BTFARG);
+ return -EOPNOTSUPP;
+}
+#endif /* CONFIG_PROBE_EVENTS_BTF_ARGS */
+
+static int wprobe_trigger_field_parse(char *field_str, struct trace_event_file *file,
+ struct wprobe_trigger_data *wprobe_data,
+ const char *glob)
+{
+ struct ftrace_event_field *field;
+ char *offs;
+
+ if (field_str[0] == '(') {
+ int ret = wprobe_trigger_typecast_parse(&field_str, file, wprobe_data, glob);
+
+ if (ret < 0)
+ return ret;
+ } else {
+ offs = strpbrk(field_str, "+-");
+ if (offs) {
+ long val;
+
+ if (kstrtol(offs, 0, &val) < 0) {
+ wprobe_trigger_log_err(file, glob, offs - glob, BAD_DEREF_OFFS);
+ return -EINVAL;
+ }
+ wprobe_data->adjust = val;
+ *offs = '\0';
+ }
+ }
+
+ field = trace_find_event_field(file->event_call, field_str);
+ if (!field) {
+ wprobe_trigger_log_err(file, glob, field_str - glob, NO_EVENT_FIELD);
+ return -ENOENT;
+ }
+ if (field->size != sizeof(void *)) {
+ wprobe_trigger_log_err(file, glob, field_str - glob, WPROBE_BAD_FIELD);
+ return -ENOEXEC;
+ }
+ wprobe_data->offset = field->offset;
+ wprobe_data->field = kstrdup(field_str, GFP_KERNEL);
+ if (!wprobe_data->field)
+ return -ENOMEM;
+
+ return 0;
+}
+
static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
struct trace_event_file *file,
char *glob, char *cmd,
@@ -1013,10 +1190,9 @@ static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
*/
struct wprobe_trigger_data *wprobe_data __free(free_wprobe_trigger_data) = NULL;
struct event_trigger_data *trigger_data __free(kfree) = NULL;
- char *event_str, *field_str, *count_str, *comment;
+ char *event_str, *comment;
struct trace_event_file *wprobe_file;
struct trace_array *tr = file->tr;
- struct trace_event_call *event;
bool remove, clear = false;
struct trace_wprobe *tw;
char *param, *filter;
@@ -1075,42 +1251,24 @@ static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
/* clear_wprobe does not need field. */
if (!clear) {
- char *offs;
+ char *field_str = strsep(¶m, ":");
- /* Find target field, which must be equivarent to "void *" */
- field_str = strsep(¶m, ":");
if (!field_str) {
wprobe_trigger_log_err(file, glob, strlen(glob), WPROBE_NEED_FIELD);
return -EINVAL;
}
-
- offs = strpbrk(field_str, "+-");
- if (offs) {
- long val;
-
- if (kstrtol(offs, 0, &val) < 0) {
- wprobe_trigger_log_err(file, glob, offs - glob, BAD_DEREF_OFFS);
- return -EINVAL;
- }
- wprobe_data->adjust = val;
- *offs = '\0';
- }
-
- event = file->event_call;
- field = trace_find_event_field(event, field_str);
- if (!field) {
- wprobe_trigger_log_err(file, glob, field_str - glob, NO_EVENT_FIELD);
- return -ENOENT;
- }
-
- if (field->size != sizeof(void *)) {
- wprobe_trigger_log_err(file, glob, field_str - glob, WPROBE_BAD_FIELD);
- return -ENOEXEC;
+ ret = wprobe_trigger_field_parse(field_str, file, wprobe_data, glob);
+ if (ret < 0)
+ return ret;
+ } else if (param) {
+ char *orig_param = param;
+ char *field_str = strsep(¶m, ":");
+
+ ret = wprobe_trigger_field_parse(field_str, file, wprobe_data, glob);
+ if (ret < 0) {
+ /* field_str was not a field, so it must be count_str */
+ param = orig_param;
}
- wprobe_data->offset = field->offset;
- wprobe_data->field = kstrdup(field_str, GFP_KERNEL);
- if (!wprobe_data->field)
- return -ENOMEM;
}
trigger_data = trigger_data_alloc(cmd_ops, cmd, param, wprobe_data);
diff --git a/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-offset.tc b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-offset.tc
new file mode 100644
index 000000000000..dda179a23282
--- /dev/null
+++ b/tools/testing/selftests/ftrace/test.d/trigger/trigger-wprobe-btf-offset.tc
@@ -0,0 +1,74 @@
+#!/bin/sh
+# SPDX-License-Identifier: GPL-2.0
+# description: event trigger - test set_wprobe trigger with BTF struct offset
+# requires: dynamic_events "w[:[<group>/][<event>]] [r|w|rw]@<addr>[:<len>]":README events/sched/sched_process_fork/trigger "[(structname[,field])]<argname>[->field[->field|.field...]]":README
+
+rmmod trace-events-sample ||:
+if ! modprobe trace-events-sample ; then
+ echo "No trace-events sample module - please make CONFIG_SAMPLE_TRACE_EVENTS=m"
+ exit_unresolved
+fi
+
+cleanup_wprobe_triggers() {
+ if [ -f events/fprobes/testevent/trigger ]; then
+ reset_trigger_file events/fprobes/testevent/trigger || true
+ fi
+ echo 0 > events/enable 2>/dev/null || true
+ echo > dynamic_events 2>/dev/null || true
+ sleep 1
+ rmmod trace-events-sample 2>/dev/null || true
+ return 0
+}
+
+trap cleanup_wprobe_triggers EXIT
+
+echo 0 > tracing_on
+
+# we will skip this test if fprobe is not supported.
+if ! grep -Fq "f[:[<group>/][<event>]] <func-name>[%return] [<args>]" README; then
+ echo "UNRESOLVED: fprobe is not supported"
+ exit_unresolved
+fi
+
+# we will skip this test if the target function does not exist.
+if ! grep -wq "sample_timer_cb" /proc/kallsyms; then
+ echo "UNRESOLVED: sample_timer_cb not found"
+ exit_unresolved
+fi
+
+:;: "Add a wprobe event watching 8 bytes" ;:
+echo 'w:watch rw@0:8 address=$addr value=$value' >> dynamic_events
+
+:;: "Add fprobe event for sample_timer_cb" ;:
+# sample_timer_cb(struct timer_list *t)
+# container_of(t, struct foo_timer_data, timer)
+echo 'f:fprobes/testevent sample_timer_cb timer=t' >> dynamic_events
+
+:;: "Enable all events before setting triggers" ;:
+echo 1 > tracing_on
+echo 1 >> events/fprobes/testevent/enable
+
+:;: "Set set_wprobe trigger using BTF struct offset resolution" ;:
+# Syntax: set_wprobe:WPEVENT:(STRUCT,FIELD)EVENT_FIELD->MEMBER
+# (foo_timer_data,timer) is the BTF struct type and field name
+# timer->expires is the struct member whose offset is resolved automatically via BTF
+echo 'set_wprobe:watch:(foo_timer_data,timer)timer->timer.expires' >> events/fprobes/testevent/trigger
+cat events/fprobes/testevent/trigger | grep ^set_wprobe
+
+# Wait for sample_timer_cb to fire and set_wprobe trigger to activate
+sleep 3
+
+:;: "Check set_wprobe trigger activated the watchpoint" ;:
+cat trace | grep watch
+
+:;: "Remove wprobe triggers" ;:
+# Since we don't know actual offset of timer->expires in foo_timer_data, we use reset_trigger_file
+reset_trigger_file events/fprobes/testevent/trigger
+! grep ^set_wprobe events/fprobes/testevent/trigger
+
+:;: "Disable events and remove dynamic events" ;:
+echo 0 > events/enable
+echo > dynamic_events
+clear_trace
+
+exit 0
^ permalink raw reply related [flat|nested] 19+ messages in thread* Re: [PATCH v12 11/11] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger
2026-08-07 15:35 ` [PATCH v12 11/11] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger Masami Hiramatsu (Google)
@ 2026-08-07 15:59 ` sashiko-bot
0 siblings, 0 replies; 19+ messages in thread
From: sashiko-bot @ 2026-08-07 15:59 UTC (permalink / raw)
To: Masami Hiramatsu (Google); +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
New issues:
- [Medium] Speculative field parsing for `clear_wprobe` commands unconditionally logs spurious errors and masks genuine syntax/memory errors.
Pre-existing issues:
- [Critical] Mixing `__free` scope-based cleanup with manual async cleanup on the error path causes double free and lockless list corruption.
--
--- Patch [11]: [PATCH v12 11/11] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger ---
commit f9bc82719437cd2a74bad3f90ad83888d83a0ed4
Author: Masami Hiramatsu (Google) <mhiramat@kernel.org>
tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger
Allow set_wprobe trigger to use BTF struct offset resolution to specify
the target address field.
Link: https://lore.kernel.org/all/59637b96946653393a7ad3c7de094094796b39c2.1785067572.git.wangjinchao600@gmail.com/
> diff --git a/kernel/trace/trace_wprobe.c b/kernel/trace/trace_wprobe.c
> index c64bbdc90a40c..565c106a93805 100644
> --- a/kernel/trace/trace_wprobe.c
> +++ b/kernel/trace/trace_wprobe.c
[ ... ]
> @@ -1013,10 +1190,9 @@ static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
> */
> struct wprobe_trigger_data *wprobe_data __free(free_wprobe_trigger_data) = NULL;
> struct event_trigger_data *trigger_data __free(kfree) = NULL;
[Severity: Critical]
This is a pre-existing issue, but does using __free() here cause a double-free
on the error paths?
If event_trigger_register() fails later in wprobe_trigger_cmd_parse(), it
calls event_trigger_free(trigger_data). This queues the trigger onto the
lockless trigger_data_free_list for asynchronous freeing by a background
kthread.
However, when the error path returns without clearing trigger_data, the
compiler-inserted __free destructors execute synchronously. This immediately
frees both wprobe_data and trigger_data while they are still linked in the
kthread's cleanup list.
When the kthread eventually processes the list, could it traverse freed
memory, trigger a use-after-free, and attempt to free the objects again?
> - char *event_str, *field_str, *count_str, *comment;
> + char *event_str, *comment;
> struct trace_event_file *wprobe_file;
> struct trace_array *tr = file->tr;
> - struct trace_event_call *event;
> bool remove, clear = false;
> struct trace_wprobe *tw;
> char *param, *filter;
[ ... ]
> @@ -1075,42 +1251,24 @@ static int wprobe_trigger_cmd_parse(struct event_command *cmd_ops,
>
> /* clear_wprobe does not need field. */
> if (!clear) {
> - char *offs;
> + char *field_str = strsep(¶m, ":");
>
> - /* Find target field, which must be equivarent to "void *" */
> - field_str = strsep(¶m, ":");
> if (!field_str) {
> wprobe_trigger_log_err(file, glob, strlen(glob), WPROBE_NEED_FIELD);
> return -EINVAL;
> }
> -
> - offs = strpbrk(field_str, "+-");
> - if (offs) {
> - long val;
> -
> - if (kstrtol(offs, 0, &val) < 0) {
> - wprobe_trigger_log_err(file, glob, offs - glob, BAD_DEREF_OFFS);
> - return -EINVAL;
> - }
> - wprobe_data->adjust = val;
> - *offs = '\0';
> - }
> -
> - event = file->event_call;
> - field = trace_find_event_field(event, field_str);
> - if (!field) {
> - wprobe_trigger_log_err(file, glob, field_str - glob, NO_EVENT_FIELD);
> - return -ENOENT;
> - }
> -
> - if (field->size != sizeof(void *)) {
> - wprobe_trigger_log_err(file, glob, field_str - glob, WPROBE_BAD_FIELD);
> - return -ENOEXEC;
> + ret = wprobe_trigger_field_parse(field_str, file, wprobe_data, glob);
> + if (ret < 0)
> + return ret;
> + } else if (param) {
> + char *orig_param = param;
> + char *field_str = strsep(¶m, ":");
> +
> + ret = wprobe_trigger_field_parse(field_str, file, wprobe_data, glob);
> + if (ret < 0) {
> + /* field_str was not a field, so it must be count_str */
> + param = orig_param;
> }
[Severity: Medium]
Will this speculative field parsing unconditionally log spurious errors?
When a valid clear_wprobe command is given with a count but no field (e.g.,
clear_wprobe:watch:10), this branch speculatively attempts to parse 10 as a
field string. wprobe_trigger_field_parse() fails and unconditionally calls
wprobe_trigger_log_err(), leaving a bogus NO_EVENT_FIELD error in the trace
error_log:
kernel/trace/trace_wprobe.c:wprobe_trigger_field_parse() {
...
field = trace_find_event_field(file->event_call, field_str);
if (!field) {
wprobe_trigger_log_err(file, glob, field_str - glob, NO_EVENT_FIELD);
return -ENOENT;
}
...
}
Because this code falls back to parsing the parameter as a count upon any
negative return code, the command ultimately succeeds, but the user
receives a false-positive error in the logs.
Additionally, if field parsing encounters a genuine error (like -EINVAL
from a malformed BTF offset or -ENOMEM), will it be swallowed by the fallback
and incorrectly masked?
> - wprobe_data->offset = field->offset;
> - wprobe_data->field = kstrdup(field_str, GFP_KERNEL);
> - if (!wprobe_data->field)
> - return -ENOMEM;
> }
>
> trigger_data = trigger_data_alloc(cmd_ops, cmd, param, wprobe_data);
--
Sashiko AI review · https://sashiko.dev/#/patchset/178611679753.237811.10190793347982785419.stgit@devnote2?part=11
^ permalink raw reply [flat|nested] 19+ messages in thread