The Linux Kernel Mailing List
 help / color / mirror / Atom feed
From: Aaron Tomlin <atomlin@atomlin.com>
To: peterz@infradead.org, mingo@redhat.com, acme@kernel.org,
	namhyung@kernel.org
Cc: mark.rutland@arm.com, alexander.shishkin@linux.intel.com,
	jolsa@kernel.org, irogers@google.com, adrian.hunter@intel.com,
	james.clark@linaro.org, howardchu95@gmail.com,
	atomlin@atomlin.com, neelx@suse.com, chjohnst@mail.com,
	sean@ashe.io, steve@abita.co, rishil1999@outlook.com,
	linux-perf-users@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: [PATCH perf-tools-next v6 3/5] perf trace: Auto-assign kernel symbol beautifier to function pointer fields
Date: Mon, 24 Aug 2026 09:31:20 -0400	[thread overview]
Message-ID: <20260824133122.751733-4-atomlin@atomlin.com> (raw)
In-Reply-To: <20260824133122.751733-1-atomlin@atomlin.com>

Tracepoint fields that convey kernel function pointers, callbacks, and
call sites, such as "function", "func", "fn", "callback", "action",
"handler", "caller", "caller_ip", "location", "callsite", and
"call_site" are currently formatted as generic hexadecimal pointers by
default.

Enhance syscall_arg_fmt__init_array() to automatically detect non-array
function pointer fields by type signature (e.g., typedefs ending with
"_func_t" or "_fn", or C function pointer types containing "(*)") and
assign SCA_KSYM as their default beautifier.

Additionally, register common function pointer and callback field names
within the sorted syscall_arg_fmts__by_name lookup table. To prevent
misclassifying dynamic string arrays, enums, or standard non-pointer
integers sharing generic names, guard SCA_KSYM assignment to true pointer
fields and pointer-sized non-array scalars (via field_is_ptr_sized(),
field_is_enum(), and field_is_plain_int()). This ensures enum fields
continue falling through to BTF pretty-printing. To support
cross-architecture analysis (e.g., analyzing 32-bit trace data on a
64-bit host) and 64-bit instruction pointers, determine target pointer
size via tep_get_long_size() while also supporting 64-bit scalars.

This ensures tracepoint arguments such as
workqueue:workqueue_execute_start.function, csd:csd_function.func,
and xfs:xfs_bunmapi.caller_ip are symbolised automatically without
requiring explicit per-event configuration. For example:

    ❯ sudo tools/perf/perf trace --event workqueue:workqueue_execute_end --max-events 2 --show-cpu
         0.000 [000] kworker/u32:15/236682 workqueue:workqueue_execute_end(work: 0xffffffffab2f1420, function: toggle_allocation_gate)
         0.132 [000] kworker/u32:15/236682 workqueue:workqueue_execute_end(work: 0xffff8ac2c1adc010, function: flush_to_ldisc)

Signed-off-by: Aaron Tomlin <atomlin@atomlin.com>
---
 tools/perf/builtin-trace.c | 113 ++++++++++++++++++++++++-------------
 1 file changed, 75 insertions(+), 38 deletions(-)

diff --git a/tools/perf/builtin-trace.c b/tools/perf/builtin-trace.c
index effb00117bff..f6ab7dabfa4c 100644
--- a/tools/perf/builtin-trace.c
+++ b/tools/perf/builtin-trace.c
@@ -2098,6 +2098,18 @@ static int syscall__alloc_arg_fmts(struct syscall *sc, int nr_args)
 }
 
 static const struct syscall_arg_fmt syscall_arg_fmts__by_name[] = {
+	{ .name = "action",	.scnprintf = SCA_KSYM, },
+	{ .name = "call_site",	.scnprintf = SCA_KSYM, },
+	{ .name = "callback",	.scnprintf = SCA_KSYM, },
+	{ .name = "caller",	.scnprintf = SCA_KSYM, },
+	{ .name = "caller_ip",	.scnprintf = SCA_KSYM, },
+	{ .name = "callsite",	.scnprintf = SCA_KSYM, },
+	{ .name = "cb",		.scnprintf = SCA_KSYM, },
+	{ .name = "fn",		.scnprintf = SCA_KSYM, },
+	{ .name = "func",	.scnprintf = SCA_KSYM, },
+	{ .name = "function",	.scnprintf = SCA_KSYM, },
+	{ .name = "handler",	.scnprintf = SCA_KSYM, },
+	{ .name = "location",	.scnprintf = SCA_KSYM, },
 	{ .name = "msr",	.scnprintf = SCA_X86_MSR,	  .strtoul = STUL_X86_MSR,	   },
 	{ .name = "vector",	.scnprintf = SCA_X86_IRQ_VECTORS, .strtoul = STUL_X86_IRQ_VECTORS, },
 };
@@ -2173,6 +2185,29 @@ static bool field_has_hex_fmt(struct tep_format_field *field, int len)
 	return false;
 }
 
+static bool field_is_enum(const struct tep_format_field *field)
+{
+	return field->type && strstr(field->type, "enum") != NULL;
+}
+
+static bool field_is_plain_int(const struct tep_format_field *field)
+{
+	if (!field->type)
+		return false;
+
+	return !strcmp(field->type, "int") ||
+	       !strcmp(field->type, "unsigned int") ||
+	       !strcmp(field->type, "u32") ||
+	       !strcmp(field->type, "s32");
+}
+
+static bool field_is_ptr_sized(const struct tep_format_field *field)
+{
+	int ptr_size = tep_get_long_size(field->event->tep);
+
+	return field->size == ptr_size || field->size == sizeof(u64);
+}
+
 static struct tep_format_field *
 syscall_arg_fmt__init_array(struct syscall_arg_fmt *arg, struct tep_format_field *field,
 			    bool *use_btf)
@@ -2200,38 +2235,49 @@ syscall_arg_fmt__init_array(struct syscall_arg_fmt *arg, struct tep_format_field
 		    ((len >= 4 && strcmp(field->name + len - 4, "name") == 0) ||
 		     strstr(field->name, "path") != NULL)) {
 			arg->scnprintf = SCA_FILENAME;
-		} else if ((field->flags & TEP_FIELD_IS_POINTER) || strstr(field->name, "addr") ||
-			   field_has_hex_fmt(field, len))
-			arg->scnprintf = SCA_PTR;
-		else if (strcmp(field->type, "pid_t") == 0)
-			arg->scnprintf = SCA_PID;
-		else if (strcmp(field->type, "umode_t") == 0)
-			arg->scnprintf = SCA_MODE_T;
-		else if ((field->flags & TEP_FIELD_IS_ARRAY) && strstr(field->type, "char")) {
-			arg->scnprintf = SCA_CHAR_ARRAY;
-			arg->nr_entries = field->arraylen;
-		} else if ((strcmp(field->type, "int") == 0 ||
-			  strcmp(field->type, "unsigned int") == 0 ||
-			  strcmp(field->type, "long") == 0) &&
-			 len >= 2 && strcmp(field->name + len - 2, "fd") == 0) {
-			/*
-			 * /sys/kernel/tracing/events/syscalls/sys_enter*
-			 * grep -E 'field:.*fd;' .../format|sed -r 's/.*field:([a-z ]+) [a-z_]*fd.+/\1/g'|sort|uniq -c
-			 * 65 int
-			 * 23 unsigned int
-			 * 7 unsigned long
-			 */
-			arg->scnprintf = SCA_FD;
-		} else if (strstr(field->type, "enum") && use_btf != NULL) {
-			*use_btf = true;
-			arg->strtoul = STUL_BTF_TYPE;
+		} else if (field->type && !(field->flags & TEP_FIELD_IS_ARRAY) &&
+			   (strstr(field->type, "(*)") != NULL ||
+			    strstr(field->type, "_func_t") != NULL ||
+			    strstr(field->type, "_fn") != NULL)) {
+			arg->scnprintf = SCA_KSYM;
 		} else {
 			const struct syscall_arg_fmt *fmt =
 				syscall_arg_fmt__find_by_name(field->name);
 
 			if (fmt) {
-				arg->scnprintf = fmt->scnprintf;
-				arg->strtoul   = fmt->strtoul;
+				if (fmt->scnprintf == SCA_KSYM) {
+					if ((field->flags & TEP_FIELD_IS_POINTER) ||
+					    (!field_is_enum(field) && !field_is_plain_int(field) &&
+					     field_is_ptr_sized(field) && !(field->flags & TEP_FIELD_IS_ARRAY))) {
+						arg->scnprintf = fmt->scnprintf;
+						arg->strtoul   = fmt->strtoul;
+					}
+				} else {
+					arg->scnprintf = fmt->scnprintf;
+					arg->strtoul   = fmt->strtoul;
+				}
+			}
+
+			if (arg->scnprintf == NULL) {
+				if ((field->flags & TEP_FIELD_IS_POINTER) || strstr(field->name, "addr") ||
+				    field_has_hex_fmt(field, len)) {
+					arg->scnprintf = SCA_PTR;
+				} else if (strcmp(field->type, "pid_t") == 0) {
+					arg->scnprintf = SCA_PID;
+				} else if (strcmp(field->type, "umode_t") == 0) {
+					arg->scnprintf = SCA_MODE_T;
+				} else if ((field->flags & TEP_FIELD_IS_ARRAY) && strstr(field->type, "char")) {
+					arg->scnprintf = SCA_CHAR_ARRAY;
+					arg->nr_entries = field->arraylen;
+				} else if ((strcmp(field->type, "int") == 0 ||
+					    strcmp(field->type, "unsigned int") == 0 ||
+					    strcmp(field->type, "long") == 0) &&
+					   len >= 2 && strcmp(field->name + len - 2, "fd") == 0) {
+					arg->scnprintf = SCA_FD;
+				} else if (field_is_enum(field) && use_btf != NULL) {
+					*use_btf = true;
+					arg->strtoul = STUL_BTF_TYPE;
+				}
 			}
 		}
 	}
@@ -3299,12 +3345,6 @@ static unsigned char bitmap_byte(const unsigned long *mask, int byte_idx)
 	return b_val;
 }
 
-static bool trace__field_is_ip(const char *name)
-{
-	return !strcmp(name, "__probe_ip") ||
-	       !strcmp(name, "caller_ip") ||
-	       !strcmp(name, "call_site");
-}
 
 static size_t trace__fprintf_tp_fields(struct trace *trace, struct perf_sample *sample,
 				       struct thread *thread, void *augmented_args, int augmented_args_size)
@@ -3406,14 +3446,11 @@ static size_t trace__fprintf_tp_fields(struct trace *trace, struct perf_sample *
 		 * Suppress it by default to avoid cluttering the output.
 		 * If verbose mode is enabled, ensure it is formatted as a
 		 * hexadecimal memory address rather than a signed integer.
-		 *
-		 * caller_ip and call_site are also expected to be instruction
-		 * pointers and should always be represented in hexadecimal.
 		 */
 		is_probe_ip = evsel__is_probe(evsel) && !strcmp(field->name, "__probe_ip");
 
-		if (is_probe_ip || trace__field_is_ip(field->name)) {
-			if (is_probe_ip && !verbose)
+		if (is_probe_ip) {
+			if (!verbose)
 				continue;
 
 			printed += scnprintf(bf + printed, size - printed,
-- 
2.55.0


  parent reply	other threads:[~2026-08-24 13:31 UTC|newest]

Thread overview: 7+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-24 13:31 [PATCH perf-tools-next v6 0/5] perf trace: Symbolise kernel virtual addresses and function pointers Aaron Tomlin
2026-08-24 13:31 ` [PATCH perf-tools-next v6 1/5] perf trace: Fix error checking in btf_struct_scnprintf() Aaron Tomlin
2026-08-24 13:31 ` [PATCH perf-tools-next v6 2/5] perf trace: Introduce kernel symbol beautifier for virtual addresses Aaron Tomlin
2026-08-24 13:31 ` Aaron Tomlin [this message]
2026-08-24 13:31 ` [PATCH perf-tools-next v6 4/5] perf trace: Enhance BTF type formatting to symbolise kernel function pointers Aaron Tomlin
2026-08-24 13:31 ` [PATCH perf-tools-next v6 5/5] perf tests: Add shell test for kernel symbol beautifier Aaron Tomlin
2026-08-24 16:30 ` [PATCH perf-tools-next v6 0/5] perf trace: Symbolise kernel virtual addresses and function pointers Ian Rogers

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260824133122.751733-4-atomlin@atomlin.com \
    --to=atomlin@atomlin.com \
    --cc=acme@kernel.org \
    --cc=adrian.hunter@intel.com \
    --cc=alexander.shishkin@linux.intel.com \
    --cc=chjohnst@mail.com \
    --cc=howardchu95@gmail.com \
    --cc=irogers@google.com \
    --cc=james.clark@linaro.org \
    --cc=jolsa@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-perf-users@vger.kernel.org \
    --cc=mark.rutland@arm.com \
    --cc=mingo@redhat.com \
    --cc=namhyung@kernel.org \
    --cc=neelx@suse.com \
    --cc=peterz@infradead.org \
    --cc=rishil1999@outlook.com \
    --cc=sean@ashe.io \
    --cc=steve@abita.co \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox