* [RFC PATCH 1/2] perf inject: Support piped data with --convert-callchain
@ 2026-09-25 21:00 Serhei Makarov
2026-09-25 21:00 ` [RFC PATCH 2/2] perf test: Test piped data in inject-callchain.sh Serhei Makarov
0 siblings, 1 reply; 2+ messages in thread
From: Serhei Makarov @ 2026-09-25 21:00 UTC (permalink / raw)
To: acme, irogers, namhyung, james.clark
Cc: jolsa, adrian.hunter, peterz, mingo, linux-kernel,
linux-perf-users, Serhei Makarov
When profiling applications without framepointers or non-framepointer
distros, it's helpful to pipe stack sample data directly to perf
inject --convert-callchain, e.g.:
perf record -F 999 --call-graph dwarf -o - -- myapp \
| perf inject --convert-callchain -i - -o perf.data
This piped setup doesn't store stack samples on disk, removing a major
performance bottleneck for DWARF stack unwinding as noted in [1].
To support piped output, write a modified copy of the event attributes
in perf_event__repipe_attr. This gets around the inability to modify
event attribute records in-place after they are written to the pipe.
To support piped input, perform a more-exact sample size calculation
to perf_event__convert_sample_callchain.
[1]: https://rwmj.wordpress.com/2023/02/14/frame-pointers-vs-dwarf-my-verdict/
> The first most obvious thing is that even with the smallest stack
> data collection, DWARF’s perf.data is over 10 times larger, and it
> balloons even larger once you start to collect more reasonable stack
> sizes. For a single minute of data collection, collecting 10s of
> gigabytes of data is not very practical even on high end machines, and
> continuous performance analysis would be impossible at these data
> rates.
Signed-off-by: Serhei Makarov <serhei@serhei.io>
---
tools/perf/builtin-inject.c | 88 ++++++++++++++++++++++++++-----------
1 file changed, 63 insertions(+), 25 deletions(-)
diff --git a/tools/perf/builtin-inject.c b/tools/perf/builtin-inject.c
index f174bc69cec4..29b22b52a631 100644
--- a/tools/perf/builtin-inject.c
+++ b/tools/perf/builtin-inject.c
@@ -219,6 +219,7 @@ static int perf_event__repipe_attr(const struct perf_tool *tool,
union perf_event *event,
struct evlist **pevlist)
{
+ union perf_event *event2;
struct perf_inject *inject = container_of(tool, struct perf_inject,
tool);
int ret;
@@ -231,7 +232,28 @@ static int perf_event__repipe_attr(const struct perf_tool *tool,
if (!inject->output.is_pipe)
return 0;
- return perf_event__repipe_synth(tool, event);
+ /* We can repipe the original event in most cases: */
+ event2 = event;
+
+ if (inject->convert_callchain) {
+ /* Copy event to repipe with corrected final attributes,
+ without confusing downstream users of pevlist: */
+ event2 = (void *)inject->event_copy;
+ if (event2 == NULL) {
+ inject->event_copy = malloc(PERF_SAMPLE_MAX_SIZE);
+ if (!inject->event_copy)
+ return -ENOMEM;
+ event2 = (void *)inject->event_copy;
+ }
+ memcpy(event2, event, event->header.size);
+
+ event2->attr.attr.sample_type &= ~(PERF_SAMPLE_REGS_USER | PERF_SAMPLE_STACK_USER);
+ event2->attr.attr.sample_regs_user = 0;
+ event2->attr.attr.sample_stack_user = 0;
+ event2->attr.attr.exclude_callchain_user = 0;
+ }
+
+ return perf_event__repipe_synth(tool, event2);
}
static int perf_event__repipe_event_update(const struct perf_tool *tool,
@@ -384,6 +406,18 @@ static int perf_event__repipe_sample(const struct perf_tool *tool,
return perf_event__repipe_synth(tool, event);
}
+static bool evsel__has_dwarf_callchain(struct evsel *evsel)
+{
+ struct perf_event_attr *attr = &evsel->core.attr;
+ const u64 dwarf_callchain_flags =
+ PERF_SAMPLE_STACK_USER | PERF_SAMPLE_REGS_USER | PERF_SAMPLE_CALLCHAIN;
+
+ if (!attr->exclude_callchain_user)
+ return false;
+
+ return (attr->sample_type & dwarf_callchain_flags) == dwarf_callchain_flags;
+}
+
static int perf_event__convert_sample_callchain(const struct perf_tool *tool,
union perf_event *event,
struct perf_sample *sample,
@@ -391,15 +425,21 @@ static int perf_event__convert_sample_callchain(const struct perf_tool *tool,
struct machine *machine)
{
struct perf_inject *inject = container_of(tool, struct perf_inject, tool);
- struct callchain_cursor *cursor = get_tls_callchain_cursor();
+ struct callchain_cursor *cursor;
union perf_event *event_copy = (void *)inject->event_copy;
struct callchain_cursor_node *node;
struct thread *thread;
u64 sample_type = evsel->core.attr.sample_type;
u32 sample_size = event->header.size;
+ u64 prev_callchain_nr;
u64 i, k;
int ret;
+ if (!evsel__has_dwarf_callchain(evsel))
+ return perf_event__repipe(tool, event, sample, machine);
+ cursor = get_tls_callchain_cursor();
+ prev_callchain_nr = sample->callchain->nr;
+
if (event_copy == NULL) {
inject->event_copy = malloc(PERF_SAMPLE_MAX_SIZE);
if (!inject->event_copy)
@@ -455,8 +495,18 @@ static int perf_event__convert_sample_callchain(const struct perf_tool *tool,
memcpy(event_copy, event, sizeof(event->header));
/* adjust sample size for stack and regs */
- sample_size -= sample->user_stack.size;
- sample_size -= (hweight64(evsel->core.attr.sample_regs_user) + 1) * sizeof(u64);
+ {
+ /* need to get stack size from the raw event */
+ const u64 *raw_stack = (const u64 *)((void *)event + sample->user_stack.offset);
+ u64 raw_alloc = raw_stack[0];
+ sample_size -= sizeof(u64);
+ if (raw_alloc)
+ sample_size -= raw_alloc + sizeof(u64);
+ }
+ sample_size -= sizeof(u64); /* abi */
+ if (sample->user_regs && sample->user_regs->abi)
+ sample_size -= hweight64(evsel->core.attr.sample_regs_user) * sizeof(u64);
+ sample_size -= (prev_callchain_nr + 1) * sizeof(u64);
sample_size += (sample->callchain->nr + 1) * sizeof(u64);
event_copy->header.size = sample_size;
@@ -2482,18 +2532,6 @@ static int __cmd_inject(struct perf_inject *inject)
return ret;
}
-static bool evsel__has_dwarf_callchain(struct evsel *evsel)
-{
- struct perf_event_attr *attr = &evsel->core.attr;
- const u64 dwarf_callchain_flags =
- PERF_SAMPLE_STACK_USER | PERF_SAMPLE_REGS_USER | PERF_SAMPLE_CALLCHAIN;
-
- if (!attr->exclude_callchain_user)
- return false;
-
- return (attr->sample_type & dwarf_callchain_flags) == dwarf_callchain_flags;
-}
-
int cmd_inject(int argc, const char **argv)
{
struct perf_inject inject = {
@@ -2746,15 +2784,15 @@ int cmd_inject(int argc, const char **argv)
if (inject.convert_callchain) {
struct evsel *evsel;
- if (inject.output.is_pipe || inject.session->data->is_pipe) {
- pr_err("--convert-callchain cannot work with pipe\n");
- goto out_delete;
- }
-
- evlist__for_each_entry(inject.session->evlist, evsel) {
- if (!evsel__has_dwarf_callchain(evsel) && !evsel__is_dummy_event(evsel)) {
- pr_err("--convert-callchain requires DWARF call graph.\n");
- goto out_delete;
+ /* For on-disk data, check evlist up-front.
+ For piped data, evlist is not available yet;
+ check in perf_event__convert_sample_callchain. */
+ if (!inject.session->data->is_pipe) {
+ evlist__for_each_entry(inject.session->evlist, evsel) {
+ if (!evsel__has_dwarf_callchain(evsel) && !evsel__is_dummy_event(evsel)) {
+ pr_err("--convert-callchain requires DWARF call graph.\n");
+ goto out_delete;
+ }
}
}
--
2.55.0
^ permalink raw reply related [flat|nested] 2+ messages in thread
* [RFC PATCH 2/2] perf test: Test piped data in inject-callchain.sh
2026-09-25 21:00 [RFC PATCH 1/2] perf inject: Support piped data with --convert-callchain Serhei Makarov
@ 2026-09-25 21:00 ` Serhei Makarov
0 siblings, 0 replies; 2+ messages in thread
From: Serhei Makarov @ 2026-09-25 21:00 UTC (permalink / raw)
To: acme, irogers, namhyung, james.clark
Cc: jolsa, adrian.hunter, peterz, mingo, linux-kernel,
linux-perf-users, Serhei Makarov
Expand the test to verify --convert-callchain on piped data.
Also, check that it's safe to run 'perf inject --convert-callchain' on
events without DWARF callchain data. In the prior version of the code,
the up-front evsel check (not possible to do with piping) was guarding
against a segfault in the prior version of the code.
$ /opt/perf/bin/perf test -vv "DWARF callchain"
89: perf inject to convert DWARF callchains to regular ones:
--- start ---
test child forked, pid 1952580
recording piped data without DWARF callchain
[ perf record: Woken up 1 times to write data ]
[ perf record: Captured and wrote 0.075 MB - ]
recording data with DWARF callchain
[ perf record: Woken up 33 times to write data ]
[ perf record: Captured and wrote 8.266 MB /tmp/perf-test.B1OMir (1025 samples) ]
recording piped data with DWARF callchain
[ perf record: Woken up 33 times to write data ]
[ perf record: Captured and wrote 8.283 MB - ]
check --convert-callchain without DWARF data
convert DWARF callchain using perf inject
convert piped DWARF callchain using unpiped perf inject
convert piped DWARF callchain using piped (input) perf inject
convert piped DWARF callchain using piped (input+output) perf inject
compare the (unpiped) both result excluding inlined functions
compare the (piped) various results excluding inlined functions
---- end(0) ----
89: perf inject to convert DWARF callchains to regular ones : Ok
Signed-off-by: Serhei Makarov <serhei@serhei.io>
---
tools/perf/tests/shell/inject-callchain.sh | 63 +++++++++++++++++++---
1 file changed, 57 insertions(+), 6 deletions(-)
diff --git a/tools/perf/tests/shell/inject-callchain.sh b/tools/perf/tests/shell/inject-callchain.sh
index a1cba8010f95..900459a8dc9e 100755
--- a/tools/perf/tests/shell/inject-callchain.sh
+++ b/tools/perf/tests/shell/inject-callchain.sh
@@ -23,23 +23,74 @@ trap_cleanup()
exit 1
}
+trap_err()
+{
+ local SIG=$?
+ if (( $? == 139 )); then #SIGSEGV
+ err=1
+ fi
+ echo "Unexpected signal ${SIG} in ${FUNCNAME[1]}"
+ cleanup
+ exit ${err}
+}
+
trap trap_cleanup EXIT TERM INT
+trap trap_err ERR
+
+echo "recording piped data without DWARF callchain"
+perf record -F 999 -o - -- perf test -w noploop | cat > ${TESTDATA}.nodwarf
echo "recording data with DWARF callchain"
perf record -F 999 --call-graph dwarf -o "${TESTDATA}" -- perf test -w noploop
+echo "recording piped data with DWARF callchain"
+perf record -F 999 --call-graph dwarf -o - -- perf test -w noploop | cat > ${TESTDATA}.piped
+
+echo "check --convert-callchain without DWARF data"
+# verify that running on non-DWARF data does not produce segfault
+perf inject -i "${TESTDATA}.nodwarf" --convert-callchain -o "${TESTDATA}.nodwarf.conv"
+cat "${TESTDATA}.nodwarf" | perf inject --convert-callchain -i - -o "${TESTDATA}.nodwarf.conv2"
+cat "${TESTDATA}.nodwarf" | perf inject --convert-callchain -i - -o - | cat > ${TESTDATA}.nodwarf.conv3
+
echo "convert DWARF callchain using perf inject"
perf inject -i "${TESTDATA}" --convert-callchain -o "${TESTDATA}.new"
+echo "convert piped DWARF callchain using unpiped perf inject"
+perf inject -i "${TESTDATA}.piped" --convert-callchain -o "${TESTDATA}.piped.conv"
+
+echo "convert piped DWARF callchain using piped (input) perf inject"
+cat "${TESTDATA}.piped" | perf inject --convert-callchain -i - -o "${TESTDATA}.pipein.conv"
+
+echo "convert piped DWARF callchain using piped (input+output) perf inject"
+cat "${TESTDATA}.piped" | perf inject --convert-callchain -i - -o - | cat > ${TESTDATA}.pipeout.conv
+
+# skip reporting/comparing .nodwarf vs .nodwarf.conv
+
perf report -i "${TESTDATA}" --no-children -q --percent-limit=1 > ${TESTDATA}.out
perf report -i "${TESTDATA}.new" --no-children -q --percent-limit=1 > ${TESTDATA}.new.out
-echo "compare the both result excluding inlined functions"
-if diff -u "${TESTDATA}.out" "${TESTDATA}.new.out" | grep "^- " | grep -qv "(inlined)"; then
- echo "Found some differences"
- diff -u "${TESTDATA}.out" "${TESTDATA}.new.out"
- err=1
-fi
+compare () {
+ local OUT=$1
+ local NEWOUT=$2
+ if diff -u "${OUT}" "${NEWOUT}" | grep "^- " | grep -qv "(inlined)"; then
+ echo "Found some differences between ${OUT} and ${NEWOUT}"
+ diff -u "${OUT}" "${NEWOUT}"
+ err=1
+ fi
+}
+
+echo "compare the (unpiped) both result excluding inlined functions"
+compare "${TESTDATA}.out" "${TESTDATA}.new.out"
+
+perf report -i "${TESTDATA}.piped" --no-children -q --percent-limit=1 > ${TESTDATA}.piped.out
+perf report -i "${TESTDATA}.piped.conv" --no-children -q --percent-limit=1 > ${TESTDATA}.piped.conv.out
+perf report -i "${TESTDATA}.pipein.conv" --no-children -q --percent-limit=1 > ${TESTDATA}.pipein.conv.out
+perf report -i "${TESTDATA}.pipeout.conv" --no-children -q --percent-limit=1 > ${TESTDATA}.pipeout.conv.out
+
+echo "compare the (piped) various results excluding inlined functions"
+compare "${TESTDATA}.piped.out" "${TESTDATA}.piped.conv.out"
+compare "${TESTDATA}.piped.conv.out" "${TESTDATA}.pipein.conv.out"
+compare "${TESTDATA}.piped.conv.out" "${TESTDATA}.pipeout.conv.out"
cleanup
exit $err
--
2.55.0
^ permalink raw reply related [flat|nested] 2+ messages in thread
end of thread, other threads:[~2026-09-25 21:00 UTC | newest]
Thread overview: 2+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-25 21:00 [RFC PATCH 1/2] perf inject: Support piped data with --convert-callchain Serhei Makarov
2026-09-25 21:00 ` [RFC PATCH 2/2] perf test: Test piped data in inject-callchain.sh Serhei Makarov
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox