* [PATCH v2 1/8] perf event: Factor build_id out into its own top-level struct
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 2/8] perf/core: Add BUILD_ID_OFFSET to UAPI Ian Rogers
` (6 subsequent siblings)
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Future changes will use build_id in more contexts. For consistency make
the struct its own type and update usage to be of this type.
Signed-off-by: Ian Rogers <irogers@google.com>
---
include/uapi/linux/perf_event.h | 19 +++++++++-----
tools/include/uapi/linux/perf_event.h | 19 +++++++++-----
tools/lib/perf/include/perf/event.h | 18 ++++++++------
tools/perf/builtin-inject.c | 3 ++-
tools/perf/util/event.c | 4 +--
tools/perf/util/machine.c | 3 ++-
tools/perf/util/python.c | 8 +++---
tools/perf/util/synthetic-events.c | 36 ++++++++++++++-------------
8 files changed, 66 insertions(+), 44 deletions(-)
diff --git a/include/uapi/linux/perf_event.h b/include/uapi/linux/perf_event.h
index fd10aa8d697f..035b4b416ce3 100644
--- a/include/uapi/linux/perf_event.h
+++ b/include/uapi/linux/perf_event.h
@@ -1058,6 +1058,18 @@ enum perf_event_type {
*/
PERF_RECORD_SAMPLE = 9,
+ /*
+ * Build IDs may be present in a number of events. They have a
+ * consistent encoding of:
+ *
+ * struct perf_build_id {
+ * u8 size;
+ * u8 __reserved_1;
+ * u16 __reserved_2;
+ * u8 data[20];
+ * };
+ */
+
/*
* The MMAP2 records are an augmented version of MMAP, they add
* maj, min, ino numbers to be used to uniquely identify each mapping
@@ -1076,12 +1088,7 @@ enum perf_event_type {
* u64 ino;
* u64 ino_generation;
* };
- * struct {
- * u8 build_id_size;
- * u8 __reserved_1;
- * u16 __reserved_2;
- * u8 build_id[20];
- * };
+ * struct perf_build_id build_id;
* };
* u32 prot, flags;
* char filename[];
diff --git a/tools/include/uapi/linux/perf_event.h b/tools/include/uapi/linux/perf_event.h
index c49fc76292f7..63f8866479a3 100644
--- a/tools/include/uapi/linux/perf_event.h
+++ b/tools/include/uapi/linux/perf_event.h
@@ -1099,6 +1099,18 @@ enum perf_event_type {
*/
PERF_RECORD_SAMPLE = 9,
+ /*
+ * Build IDs may be present in a number of events. They have a
+ * consistent encoding of:
+ *
+ * struct perf_build_id {
+ * u8 size;
+ * u8 __reserved_1;
+ * u16 __reserved_2;
+ * u8 data[20];
+ * };
+ */
+
/*
* The MMAP2 records are an augmented version of MMAP, they add
* maj, min, ino numbers to be used to uniquely identify each mapping
@@ -1117,12 +1129,7 @@ enum perf_event_type {
* u64 ino;
* u64 ino_generation;
* };
- * struct {
- * u8 build_id_size;
- * u8 __reserved_1;
- * u16 __reserved_2;
- * u8 build_id[20];
- * };
+ * struct perf_build_id build_id;
* };
* u32 prot, flags;
* char filename[];
diff --git a/tools/lib/perf/include/perf/event.h b/tools/lib/perf/include/perf/event.h
index fdced574c889..173eab43c148 100644
--- a/tools/lib/perf/include/perf/event.h
+++ b/tools/lib/perf/include/perf/event.h
@@ -26,6 +26,15 @@ struct perf_record_mmap {
char filename[PATH_MAX];
};
+#define PERF_BUILD_ID_SIZE 20
+
+struct perf_build_id {
+ __u8 size;
+ __u8 __reserved_1;
+ __u16 __reserved_2;
+ __u8 data[PERF_BUILD_ID_SIZE];
+};
+
struct perf_record_mmap2 {
struct perf_event_header header;
__u32 pid, tid;
@@ -39,12 +48,7 @@ struct perf_record_mmap2 {
__u64 ino;
__u64 ino_generation;
};
- struct {
- __u8 build_id_size;
- __u8 __reserved_1;
- __u16 __reserved_2;
- __u8 build_id[20];
- };
+ struct perf_build_id build_id;
};
__u32 prot;
__u32 flags;
@@ -321,7 +325,7 @@ struct perf_record_header_build_id {
union {
__u8 build_id[24];
struct {
- __u8 data[20];
+ __u8 data[PERF_BUILD_ID_SIZE];
__u8 size;
__u8 reserved1__;
__u16 reserved2__;
diff --git a/tools/perf/builtin-inject.c b/tools/perf/builtin-inject.c
index ce7562fc788c..67f019acb6c0 100644
--- a/tools/perf/builtin-inject.c
+++ b/tools/perf/builtin-inject.c
@@ -811,7 +811,8 @@ static int perf_event__repipe_mmap2(const struct perf_tool *tool,
struct dso_id id = dso_id_empty;
if (event->header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID) {
- build_id__init(&id.build_id, event->mmap2.build_id, event->mmap2.build_id_size);
+ build_id__init(&id.build_id, event->mmap2.build_id.data,
+ event->mmap2.build_id.size);
} else {
id.maj = event->mmap2.maj;
id.min = event->mmap2.min;
diff --git a/tools/perf/util/event.c b/tools/perf/util/event.c
index ea75816d126a..c69ae57ce679 100644
--- a/tools/perf/util/event.c
+++ b/tools/perf/util/event.c
@@ -335,8 +335,8 @@ size_t perf_event__fprintf_mmap2(union perf_event *event, FILE *fp)
char sbuild_id[SBUILD_ID_SIZE];
struct build_id bid;
- build_id__init(&bid, event->mmap2.build_id,
- event->mmap2.build_id_size);
+ build_id__init(&bid, event->mmap2.build_id.data,
+ event->mmap2.build_id.size);
build_id__snprintf(&bid, sbuild_id, sizeof(sbuild_id));
return fprintf(fp, " %d/%d: [%#" PRI_lx64 "(%#" PRI_lx64 ") @ %#" PRI_lx64
diff --git a/tools/perf/util/machine.c b/tools/perf/util/machine.c
index 1d9d3bf57720..4bcb16da6481 100644
--- a/tools/perf/util/machine.c
+++ b/tools/perf/util/machine.c
@@ -1799,7 +1799,8 @@ int machine__process_mmap2_event(struct machine *machine,
perf_event__fprintf_mmap2(event, stdout);
if (event->header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID) {
- build_id__init(&dso_id.build_id, event->mmap2.build_id, event->mmap2.build_id_size);
+ build_id__init(&dso_id.build_id, event->mmap2.build_id.data,
+ event->mmap2.build_id.size);
} else {
dso_id.maj = event->mmap2.maj;
dso_id.min = event->mmap2.min;
diff --git a/tools/perf/util/python.c b/tools/perf/util/python.c
index 95140dfaef9c..cc9257e2cceb 100644
--- a/tools/perf/util/python.c
+++ b/tools/perf/util/python.c
@@ -250,12 +250,12 @@ static PyObject *pyrf_mmap2_event__get_build_id(PyObject *self, void *closure __
if (!(pevent->event.header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID))
Py_RETURN_NONE;
- int size = pevent->event.mmap2.build_id_size;
+ size_t size = pevent->event.mmap2.build_id.size;
- if (size > 20)
- size = 20;
+ if (size > sizeof(pevent->event.mmap2.build_id.data))
+ size = sizeof(pevent->event.mmap2.build_id.data);
- return PyBytes_FromStringAndSize((const char *)pevent->event.mmap2.build_id, size);
+ return PyBytes_FromStringAndSize((const char *)pevent->event.mmap2.build_id.data, size);
}
static PyGetSetDef pyrf_mmap2_event__getset[] = {
diff --git a/tools/perf/util/synthetic-events.c b/tools/perf/util/synthetic-events.c
index b9dbd105cdc3..4d552150bc04 100644
--- a/tools/perf/util/synthetic-events.c
+++ b/tools/perf/util/synthetic-events.c
@@ -452,7 +452,7 @@ static void perf_record_mmap2__read_build_id(struct perf_record_mmap2 *event,
}
if (event->header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID) {
- build_id__init(&dso_id.build_id, event->build_id, event->build_id_size);
+ build_id__init(&dso_id.build_id, event->build_id.data, event->build_id.size);
} else {
dso_id.maj = event->maj;
dso_id.min = event->min;
@@ -479,11 +479,11 @@ static void perf_record_mmap2__read_build_id(struct perf_record_mmap2 *event,
out:
if (rc == 0) {
- memcpy(event->build_id, bid.data, sizeof(bid.data));
- event->build_id_size = (u8) bid.size;
+ memcpy(event->build_id.data, bid.data, sizeof(bid.data));
+ event->build_id.size = (u8) bid.size;
event->header.misc |= PERF_RECORD_MISC_MMAP_BUILD_ID;
- event->__reserved_1 = 0;
- event->__reserved_2 = 0;
+ event->build_id.__reserved_1 = 0;
+ event->build_id.__reserved_2 = 0;
if (dso && !dso__has_build_id(dso))
dso__set_build_id(dso, &bid);
@@ -611,8 +611,10 @@ int perf_event__synthesize_mmap_events(const struct perf_tool *tool,
event->mmap2.prot = prot;
event->mmap2.flags = flags;
- if (!symbol_conf.no_buildid_mmap2)
- perf_record_mmap2__read_build_id(&event->mmap2, machine, false);
+ if (!symbol_conf.no_buildid_mmap2) {
+ perf_record_mmap2__read_build_id(&event->mmap2, machine,
+ /*is_kernel=*/false);
+ }
if (perf_tool__process_synth_event(tool, event, machine, process) != 0) {
rc = -1;
@@ -807,12 +809,12 @@ static int perf_event__synthesize_modules_maps_cb(struct map *map, void *data)
/* Clear stale build ID and entire union from previous module iteration */
event->mmap2.header.misc &= ~PERF_RECORD_MISC_MMAP_BUILD_ID;
- memset(event->mmap2.build_id, 0, sizeof(event->mmap2.build_id));
- event->mmap2.build_id_size = 0;
- event->mmap2.__reserved_1 = 0;
- event->mmap2.__reserved_2 = 0;
+ memset(event->mmap2.build_id.data, 0, sizeof(event->mmap2.build_id.data));
+ event->mmap2.build_id.size = 0;
+ event->mmap2.build_id.__reserved_1 = 0;
+ event->mmap2.build_id.__reserved_2 = 0;
- perf_record_mmap2__read_build_id(&event->mmap2, args->machine, false);
+ perf_record_mmap2__read_build_id(&event->mmap2, args->machine, /*is_kernel=*/false);
} else {
const char *long_name = dso__long_name(dso);
@@ -1293,7 +1295,7 @@ static int __perf_event__synthesize_kernel_mmap(const struct perf_tool *tool,
event->mmap2.len = map__end(map) - event->mmap.start;
event->mmap2.pid = machine->pid;
- perf_record_mmap2__read_build_id(&event->mmap2, machine, true);
+ perf_record_mmap2__read_build_id(&event->mmap2, machine, /*is_kernel=*/true);
} else {
size = snprintf(event->mmap.filename, sizeof(event->mmap.filename),
"%s%s", machine->mmap_name, kmap->ref_reloc_sym->name) + 1;
@@ -2524,10 +2526,10 @@ int perf_event__synthesize_mmap2_build_id(const struct perf_tool *tool,
ev.mmap2.len = len;
ev.mmap2.pgoff = pgoff;
- ev.mmap2.build_id_size = bid->size;
- if (ev.mmap2.build_id_size > sizeof(ev.mmap2.build_id))
- ev.mmap2.build_id_size = sizeof(ev.mmap2.build_id);
- memcpy(ev.mmap2.build_id, bid->data, ev.mmap2.build_id_size);
+ ev.mmap2.build_id.size = bid->size;
+ if (ev.mmap2.build_id.size > sizeof(ev.mmap2.build_id.data))
+ ev.mmap2.build_id.size = sizeof(ev.mmap2.build_id.data);
+ memcpy(ev.mmap2.build_id.data, bid->data, ev.mmap2.build_id.size);
ev.mmap2.prot = prot;
ev.mmap2.flags = flags;
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 1/8] perf event: Factor build_id out into its own top-level struct
2026-10-02 17:38 ` [PATCH v2 1/8] perf event: Factor build_id out into its own top-level struct Ian Rogers
@ 2026-10-03 1:33 ` sashiko-bot
0 siblings, 0 replies; 27+ messages in thread
From: sashiko-bot @ 2026-10-03 1:33 UTC (permalink / raw)
To: Ian Rogers; +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] perf/libperf: API breakage in perf_record_mmap2 due to renamed fields
--
commit cfeda23f684f7f8aa6b272dc2febc41f9f8eb93c
Author: Ian Rogers <irogers@google.com>
perf event: Factor build_id out into its own top-level struct
This commit factors the build ID structure out into its own top-level type,
struct perf_build_id, for consistency. It updates the usage of build IDs
across various perf events to use this new type.
> diff --git a/tools/lib/perf/include/perf/event.h b/tools/lib/perf/include/perf/event.h
> index fdced574c889e..173eab43c1484 100644
> --- a/tools/lib/perf/include/perf/event.h
> +++ b/tools/lib/perf/include/perf/event.h
[ ... ]
> @@ -39,12 +48,7 @@ struct perf_record_mmap2 {
> __u64 ino;
> __u64 ino_generation;
> };
> - struct {
> - __u8 build_id_size;
> - __u8 __reserved_1;
> - __u16 __reserved_2;
> - __u8 build_id[20];
> - };
> + struct perf_build_id build_id;
[Severity: High]
Does this break the source-level API for external libperf consumers?
External user-space applications (like bpftool, pahole, or custom perf data
parsers) that rely on event->mmap2.build_id_size or treat
event->mmap2.build_id as a byte array will fail to compile against the
updated libperf headers.
While this preserves the binary ABI since the memory layout is exactly 24
bytes in both cases, the renamed fields and altered types will break the
API for existing libperf users. Could this transition be handled with an
anonymous union to preserve source compatibility during the update?
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261002173848.3228217-1-irogers@google.com?part=1
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 2/8] perf/core: Add BUILD_ID_OFFSET to UAPI
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
2026-10-02 17:38 ` [PATCH v2 1/8] perf event: Factor build_id out into its own top-level struct Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 3/8] perf/core: Implement BUILD_ID_OFFSET sample type Ian Rogers
` (5 subsequent siblings)
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Introduce PERF_SAMPLE_BUILD_ID_OFFSET and
PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET
Signed-off-by: Ian Rogers <irogers@google.com>
---
include/uapi/linux/perf_event.h | 34 +++++++++++++++++++++++----
tools/include/uapi/linux/perf_event.h | 34 +++++++++++++++++++++++----
tools/lib/perf/include/perf/event.h | 33 +++++++++++++++++++++++++-
tools/perf/util/session.c | 34 ++++++++++++++++++++++++++-
4 files changed, 125 insertions(+), 10 deletions(-)
diff --git a/include/uapi/linux/perf_event.h b/include/uapi/linux/perf_event.h
index 035b4b416ce3..4157aa8a243a 100644
--- a/include/uapi/linux/perf_event.h
+++ b/include/uapi/linux/perf_event.h
@@ -165,8 +165,10 @@ enum perf_event_sample_format {
PERF_SAMPLE_DATA_PAGE_SIZE = 1U << 22,
PERF_SAMPLE_CODE_PAGE_SIZE = 1U << 23,
PERF_SAMPLE_WEIGHT_STRUCT = 1U << 24,
+ PERF_SAMPLE_BUILD_ID_OFFSET = 1U << 25,
+ PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET = 1U << 26,
- PERF_SAMPLE_MAX = 1U << 25, /* non-ABI */
+ PERF_SAMPLE_MAX = 1U << 27, /* non-ABI */
};
#define PERF_SAMPLE_WEIGHT_TYPE (PERF_SAMPLE_WEIGHT | PERF_SAMPLE_WEIGHT_STRUCT)
@@ -815,7 +817,8 @@ struct perf_event_mmap_page {
*
* PERF_RECORD_MISC_EXACT_IP - PERF_RECORD_SAMPLE of precise events
* PERF_RECORD_MISC_SWITCH_OUT_PREEMPT - PERF_RECORD_SWITCH* events
- * PERF_RECORD_MISC_MMAP_BUILD_ID - PERF_RECORD_MMAP2 event
+ * PERF_RECORD_MISC_MMAP_BUILD_ID - PERF_RECORD_MMAP2 and
+ * PERF_RECORD_CALLCHAIN_DEFERRED events
*
*
* PERF_RECORD_MISC_EXACT_IP:
@@ -827,7 +830,7 @@ struct perf_event_mmap_page {
* Indicates that thread was preempted in TASK_RUNNING state.
*
* PERF_RECORD_MISC_MMAP_BUILD_ID:
- * Indicates that mmap2 event carries build ID data.
+ * Indicates that mmap2 or deferred callchain event carries build ID data.
*/
#define PERF_RECORD_MISC_EXACT_IP (1 << 14)
#define PERF_RECORD_MISC_SWITCH_OUT_PREEMPT (1 << 14)
@@ -1054,6 +1057,23 @@ enum perf_event_type {
* { u64 code_page_size;} && PERF_SAMPLE_CODE_PAGE_SIZE
* { u64 size;
* char data[size]; } && PERF_SAMPLE_AUX
+ * { union {
+ * struct {
+ * struct perf_build_id bid;
+ * u64 offset;
+ * };
+ * struct {
+ * u64 zero;
+ * u64 context;
+ * u64 cookie;
+ * u64 ip;
+ * }; && attr.defer_callchain
+ * }; } && PERF_SAMPLE_BUILD_ID_OFFSET
+ * { u64 nr;
+ * struct {
+ * struct perf_build_id bid;
+ * u64 offset;
+ * }[nr]; } && PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET
* };
*/
PERF_RECORD_SAMPLE = 9,
@@ -1260,7 +1280,13 @@ enum perf_event_type {
* struct perf_event_header header;
* u64 cookie;
* u64 nr;
- * u64 ips[nr];
+ * union {
+ * u64 ips[nr];
+ * struct {
+ * struct perf_build_id bid;
+ * u64 offset;
+ * }[nr]; && (header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID)
+ * };
* struct sample_id sample_id;
* };
*/
diff --git a/tools/include/uapi/linux/perf_event.h b/tools/include/uapi/linux/perf_event.h
index 63f8866479a3..56b0cd504a3d 100644
--- a/tools/include/uapi/linux/perf_event.h
+++ b/tools/include/uapi/linux/perf_event.h
@@ -165,8 +165,10 @@ enum perf_event_sample_format {
PERF_SAMPLE_DATA_PAGE_SIZE = 1U << 22,
PERF_SAMPLE_CODE_PAGE_SIZE = 1U << 23,
PERF_SAMPLE_WEIGHT_STRUCT = 1U << 24,
+ PERF_SAMPLE_BUILD_ID_OFFSET = 1U << 25,
+ PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET = 1U << 26,
- PERF_SAMPLE_MAX = 1U << 25, /* non-ABI */
+ PERF_SAMPLE_MAX = 1U << 27, /* non-ABI */
};
#define PERF_SAMPLE_WEIGHT_TYPE (PERF_SAMPLE_WEIGHT | PERF_SAMPLE_WEIGHT_STRUCT)
@@ -840,7 +842,8 @@ struct perf_event_mmap_page {
*
* PERF_RECORD_MISC_EXACT_IP - PERF_RECORD_SAMPLE of precise events
* PERF_RECORD_MISC_SWITCH_OUT_PREEMPT - PERF_RECORD_SWITCH* events
- * PERF_RECORD_MISC_MMAP_BUILD_ID - PERF_RECORD_MMAP2 event
+ * PERF_RECORD_MISC_MMAP_BUILD_ID - PERF_RECORD_MMAP2 and
+ * PERF_RECORD_CALLCHAIN_DEFERRED events
*
*
* PERF_RECORD_MISC_EXACT_IP:
@@ -852,7 +855,7 @@ struct perf_event_mmap_page {
* Indicates that thread was preempted in TASK_RUNNING state.
*
* PERF_RECORD_MISC_MMAP_BUILD_ID:
- * Indicates that mmap2 event carries build ID data.
+ * Indicates that mmap2 or deferred callchain event carries build ID data.
*/
#define PERF_RECORD_MISC_EXACT_IP (1 << 14)
#define PERF_RECORD_MISC_SWITCH_OUT_PREEMPT (1 << 14)
@@ -1095,6 +1098,23 @@ enum perf_event_type {
* { u64 code_page_size;} && PERF_SAMPLE_CODE_PAGE_SIZE
* { u64 size;
* char data[size]; } && PERF_SAMPLE_AUX
+ * { union {
+ * struct {
+ * struct perf_build_id bid;
+ * u64 offset;
+ * };
+ * struct {
+ * u64 zero;
+ * u64 context;
+ * u64 cookie;
+ * u64 ip;
+ * }; && attr.defer_callchain
+ * }; } && PERF_SAMPLE_BUILD_ID_OFFSET
+ * { u64 nr;
+ * struct {
+ * struct perf_build_id bid;
+ * u64 offset;
+ * }[nr]; } && PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET
* };
*/
PERF_RECORD_SAMPLE = 9,
@@ -1301,7 +1321,13 @@ enum perf_event_type {
* struct perf_event_header header;
* u64 cookie;
* u64 nr;
- * u64 ips[nr];
+ * union {
+ * u64 ips[nr];
+ * struct {
+ * struct perf_build_id bid;
+ * u64 offset;
+ * }[nr]; && (header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID)
+ * };
* struct sample_id sample_id;
* };
*/
diff --git a/tools/lib/perf/include/perf/event.h b/tools/lib/perf/include/perf/event.h
index 173eab43c148..3d35a82cccb0 100644
--- a/tools/lib/perf/include/perf/event.h
+++ b/tools/lib/perf/include/perf/event.h
@@ -3,11 +3,24 @@
#define __LIBPERF_EVENT_H
#include <linux/perf_event.h>
+#include <linux/stddef.h>
#include <linux/types.h>
#include <linux/limits.h>
#include <linux/bpf.h>
#include <sys/types.h> /* pid_t */
+#ifndef __DECLARE_FLEX_ARRAY
+#ifdef __cplusplus
+#define __DECLARE_FLEX_ARRAY(T, member) T member[0]
+#else
+#define __DECLARE_FLEX_ARRAY(TYPE, NAME) \
+ struct { \
+ struct { } __empty_ ## NAME; \
+ TYPE NAME[]; \
+ }
+#endif
+#endif
+
/*
* Verify the full field fits within the event, not just its start offset.
* Only valid for fixed-size scalar fields — for trailing arrays like
@@ -162,6 +175,21 @@ struct perf_record_switch {
__u32 next_prev_tid;
};
+struct perf_sample_build_id_offset {
+ union {
+ struct perf_build_id bid;
+ struct {
+ __u64 zero;
+ __u64 context;
+ __u64 cookie;
+ };
+ };
+ union {
+ __u64 offset;
+ __u64 ip;
+ };
+};
+
struct perf_record_callchain_deferred {
struct perf_event_header header;
/*
@@ -171,7 +199,10 @@ struct perf_record_callchain_deferred {
*/
__u64 cookie;
__u64 nr;
- __u64 ips[];
+ union {
+ __DECLARE_FLEX_ARRAY(__u64, ips);
+ __DECLARE_FLEX_ARRAY(struct perf_sample_build_id_offset, bids);
+ };
};
struct perf_record_header_attr {
diff --git a/tools/perf/util/session.c b/tools/perf/util/session.c
index 7fea9e72726c..5e113070b3c6 100644
--- a/tools/perf/util/session.c
+++ b/tools/perf/util/session.c
@@ -1186,6 +1186,38 @@ static int perf_event__header_feature_swap(union perf_event *event,
return 0;
}
+static int perf_event__callchain_deferred_swap(union perf_event *event,
+ bool sample_id_all)
+{
+ u64 nr, max_nr;
+
+ if (!(event->header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID))
+ return perf_event__all64_swap(event, sample_id_all);
+
+ if (event->header.size < sizeof(event->callchain_deferred))
+ return -1;
+
+ event->callchain_deferred.cookie = bswap_64(event->callchain_deferred.cookie);
+ event->callchain_deferred.nr = bswap_64(event->callchain_deferred.nr);
+
+ nr = event->callchain_deferred.nr;
+ max_nr = (event->header.size - sizeof(event->callchain_deferred)) /
+ sizeof(event->callchain_deferred.bids[0]);
+ if (nr > max_nr)
+ return -1;
+
+ for (u64 i = 0; i < nr; i++) {
+ struct perf_sample_build_id_offset *bid =
+ &event->callchain_deferred.bids[i];
+
+ bid->offset = bswap_64(bid->offset);
+ }
+
+ if (sample_id_all)
+ swap_sample_id_all(event, &event->callchain_deferred.bids[nr]);
+ return 0;
+}
+
typedef int (*perf_event__swap_op)(union perf_event *event,
bool sample_id_all);
@@ -1211,7 +1243,7 @@ static perf_event__swap_op perf_event__swap_ops[] = {
[PERF_RECORD_BPF_EVENT] = perf_event__bpf_event_swap,
[PERF_RECORD_TEXT_POKE] = perf_event__text_poke_swap,
[PERF_RECORD_AUX_OUTPUT_HW_ID] = perf_event__all64_swap,
- [PERF_RECORD_CALLCHAIN_DEFERRED] = perf_event__all64_swap,
+ [PERF_RECORD_CALLCHAIN_DEFERRED] = perf_event__callchain_deferred_swap,
[PERF_RECORD_HEADER_ATTR] = perf_event__hdr_attr_swap,
[PERF_RECORD_HEADER_EVENT_TYPE] = perf_event__event_type_swap,
[PERF_RECORD_HEADER_TRACING_DATA] = perf_event__tracing_data_swap,
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* [PATCH v2 3/8] perf/core: Implement BUILD_ID_OFFSET sample type
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
2026-10-02 17:38 ` [PATCH v2 1/8] perf event: Factor build_id out into its own top-level struct Ian Rogers
2026-10-02 17:38 ` [PATCH v2 2/8] perf/core: Add BUILD_ID_OFFSET to UAPI Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 4/8] perf: Refactor thread map and symbol APIs to take perf_sample Ian Rogers
` (4 subsequent siblings)
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Implement PERF_SAMPLE_BUILD_ID_OFFSET and
PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET payload writing in the perf event
core by reusing BPF's stack_map_get_build_id_offset().
When attr.defer_callchain is set, user-space IPs and callchains request
a deferred unwind via unwind_deferred_request() and emit a
PERF_CONTEXT_USER_DEFERRED marker and cookie in the sample. In the
faultable task_work callback (perf_unwind_deferred_callback), the
unwound user IPs are resolved to build IDs and file offsets using
stack_map_get_build_id_offset(..., may_fault=true) and emitted in
PERF_RECORD_CALLCHAIN_DEFERRED with PERF_RECORD_MISC_MMAP_BUILD_ID set.
When attr.defer_callchain is not set (or if deferred unwinding is
unavailable), user-space IPs and callchains are resolved in-place during
perf_output_sample() using stack_map_get_build_id_offset(...,
may_fault=false). If the mmap_lock cannot be trylocked or the ELF
header/note pages are not present in the page cache (nor mlocked into
memory), resolution falls back to emitting a zero-sized build ID
(size = 0) with the virtual address as the offset. Using the deferred
approach (attr.defer_callchain and attr.defer_output) is therefore
preferred as it can fault in pages and wait on locks, resolving more
build IDs without requiring ELF header pages to be mlocked in the page
cache.
Kernel-space IPs and callchain entries are emitted with a zero-sized
build ID (size = 0) and the virtual address as the offset.
Signed-off-by: Ian Rogers <irogers@google.com>
---
include/linux/bpf.h | 14 ++
include/linux/perf_event.h | 1 +
kernel/bpf/stackmap.c | 4 +-
kernel/events/core.c | 298 +++++++++++++++++++++++++++++++++++--
4 files changed, 304 insertions(+), 13 deletions(-)
diff --git a/include/linux/bpf.h b/include/linux/bpf.h
index e57af902560c..750133762436 100644
--- a/include/linux/bpf.h
+++ b/include/linux/bpf.h
@@ -4217,4 +4217,18 @@ static inline int bpf_map_check_op_flags(struct bpf_map *map, u64 flags, u64 all
return 0;
}
+#if defined(CONFIG_BPF_SYSCALL) && defined(CONFIG_PERF_EVENTS)
+void stack_map_get_build_id_offset(struct bpf_stack_build_id *id_offs,
+ u32 trace_nr, bool user, bool may_fault);
+#else
+static inline void stack_map_get_build_id_offset(struct bpf_stack_build_id *id_offs,
+ u32 trace_nr, bool user, bool may_fault)
+{
+ for (u32 i = 0; i < trace_nr; i++) {
+ id_offs[i].status = BPF_STACK_BUILD_ID_IP;
+ memset(id_offs[i].build_id, 0, BPF_BUILD_ID_SIZE);
+ }
+}
+#endif
+
#endif /* _LINUX_BPF_H */
diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h
index 5842552294c1..3868923f6a79 100644
--- a/include/linux/perf_event.h
+++ b/include/linux/perf_event.h
@@ -1347,6 +1347,7 @@ struct perf_sample_data {
u64 data_page_size;
u64 code_page_size;
u64 aux_size;
+ u64 callchain_bid_nr;
} ____cacheline_aligned;
/* default value for data source */
diff --git a/kernel/bpf/stackmap.c b/kernel/bpf/stackmap.c
index d09d4c3fe547..4e06c1697786 100644
--- a/kernel/bpf/stackmap.c
+++ b/kernel/bpf/stackmap.c
@@ -411,8 +411,8 @@ static void stack_map_get_build_id_offset_sleepable(struct bpf_stack_build_id *i
* id_offs[i].build_id is zeroed out and id_offs[i].status is set to
* BPF_STACK_BUILD_ID_IP.
*/
-static void stack_map_get_build_id_offset(struct bpf_stack_build_id *id_offs,
- u32 trace_nr, bool user, bool may_fault)
+void stack_map_get_build_id_offset(struct bpf_stack_build_id *id_offs,
+ u32 trace_nr, bool user, bool may_fault)
{
struct mmap_unlock_irq_work *work;
bool has_user_ctx = user && current && current->mm;
diff --git a/kernel/events/core.c b/kernel/events/core.c
index 33210aff3ee6..7f6bf96f7473 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -460,6 +460,8 @@ static atomic_t nr_bpf_events __read_mostly;
static atomic_t nr_cgroup_events __read_mostly;
static atomic_t nr_text_poke_events __read_mostly;
static atomic_t nr_build_id_events __read_mostly;
+static atomic_t nr_build_id_offset_events __read_mostly;
+static atomic_t nr_callchain_build_id_offset_events __read_mostly;
static LIST_HEAD(pmus);
static DEFINE_MUTEX(pmus_lock);
@@ -2048,6 +2050,14 @@ static int __perf_event_read_size(u64 read_format, int nr_siblings)
return size + nr * entry;
}
+struct perf_sample_build_id_offset {
+ u8 size;
+ u8 res1;
+ u16 res2;
+ u8 build_id[BUILD_ID_SIZE_MAX];
+ u64 offset;
+};
+
static void __perf_event_header_size(struct perf_event *event, u64 sample_type)
{
struct perf_sample_data *data;
@@ -2086,6 +2096,9 @@ static void __perf_event_header_size(struct perf_event *event, u64 sample_type)
if (sample_type & PERF_SAMPLE_CODE_PAGE_SIZE)
size += sizeof(data->code_page_size);
+ if (sample_type & PERF_SAMPLE_BUILD_ID_OFFSET)
+ size += sizeof(struct perf_sample_build_id_offset);
+
event->header_size = size;
}
@@ -5341,7 +5354,7 @@ static bool is_sb_event(struct perf_event *event)
attr->comm || attr->comm_exec ||
attr->task || attr->ksymbol ||
attr->context_switch || attr->text_poke ||
- attr->bpf_event)
+ attr->bpf_event || attr->defer_output)
return true;
return false;
@@ -5619,6 +5632,10 @@ static void unaccount_event(struct perf_event *event)
atomic_dec(&nr_mmap_events);
if (event->attr.build_id)
atomic_dec(&nr_build_id_events);
+ if (event->attr.sample_type & PERF_SAMPLE_BUILD_ID_OFFSET)
+ atomic_dec(&nr_build_id_offset_events);
+ if (event->attr.sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET)
+ atomic_dec(&nr_callchain_build_id_offset_events);
if (event->attr.comm)
atomic_dec(&nr_comm_events);
if (event->attr.namespaces)
@@ -8276,6 +8293,155 @@ static void perf_output_read(struct perf_output_handle *handle,
perf_output_read_one(handle, event, enabled, running);
}
+static struct unwind_work perf_unwind_work;
+
+struct perf_bpf_build_id_buf {
+ local_t active;
+ struct bpf_stack_build_id bids[PERF_MAX_STACK_DEPTH + 1];
+};
+
+static DEFINE_PER_CPU(struct perf_bpf_build_id_buf, perf_bpf_build_id_buf);
+
+static void perf_bpf_to_sample_bid_offset(struct bpf_stack_build_id *bpf_bid)
+{
+ struct perf_sample_build_id_offset *bid = (void *)bpf_bid;
+ u8 size = 0;
+
+ static_assert(sizeof(*bid) == sizeof(*bpf_bid));
+ static_assert(offsetof(struct perf_sample_build_id_offset, build_id) ==
+ offsetof(struct bpf_stack_build_id, build_id));
+ static_assert(offsetof(struct perf_sample_build_id_offset, offset) ==
+ offsetof(struct bpf_stack_build_id, offset));
+
+ if (bpf_bid->status == BPF_STACK_BUILD_ID_VALID)
+ size = BUILD_ID_SIZE_MAX;
+ else
+ memset(bid->build_id, 0, sizeof(bid->build_id));
+
+ bid->size = size;
+ bid->res1 = 0;
+ bid->res2 = 0;
+}
+
+static void perf_output_sample_build_id(struct perf_output_handle *handle,
+ struct perf_event_header *header,
+ struct perf_sample_data *data,
+ struct perf_event *event)
+{
+ u64 sample_type = data->type;
+ bool sample_bid = sample_type & PERF_SAMPLE_BUILD_ID_OFFSET;
+ bool callchain_bid = sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET;
+ bool crosstask = event->ctx->task && event->ctx->task != current;
+ bool has_user_ctx = is_user_task(current) && current->mm && !crosstask;
+ bool is_user = (header->misc & PERF_RECORD_MISC_CPUMODE_MASK) ==
+ PERF_RECORD_MISC_USER;
+ u64 defer_cookie = 0;
+ u64 callchain_nr = callchain_bid ? data->callchain_bid_nr : 0;
+ u64 user_start = callchain_nr;
+ u32 resolve_ip_nr = 0, resolve_cc_nr = 0, total_resolve_nr;
+ struct perf_bpf_build_id_buf *buf = NULL;
+
+ if (sample_bid) {
+ bool defer_user = IS_ENABLED(CONFIG_UNWIND_USER) &&
+ is_user && has_user_ctx &&
+ event->attr.defer_callchain &&
+ !(sample_type & PERF_SAMPLE_CALLCHAIN);
+
+ if (defer_user &&
+ unwind_deferred_request(&perf_unwind_work, &defer_cookie) < 0)
+ defer_cookie = 0;
+
+ if (!defer_cookie && is_user && has_user_ctx)
+ resolve_ip_nr = 1;
+ }
+
+ if (callchain_bid && has_user_ctx) {
+ for (u64 i = 0; i < callchain_nr; i++) {
+ if (data->callchain->ip[i] == PERF_CONTEXT_USER) {
+ user_start = i + 1;
+ break;
+ }
+ }
+ if (user_start < callchain_nr) {
+ resolve_cc_nr = min_t(u64, callchain_nr - user_start,
+ PERF_MAX_STACK_DEPTH);
+ }
+ }
+
+ total_resolve_nr = resolve_ip_nr + resolve_cc_nr;
+ if (total_resolve_nr) {
+ buf = get_cpu_ptr(&perf_bpf_build_id_buf);
+ if (local_cmpxchg(&buf->active, 0, 1) == 0) {
+ if (resolve_ip_nr)
+ buf->bids[0].ip = data->ip;
+ for (u32 i = 0; i < resolve_cc_nr; i++) {
+ buf->bids[resolve_ip_nr + i].ip =
+ data->callchain->ip[user_start + i];
+ }
+ stack_map_get_build_id_offset(buf->bids,
+ total_resolve_nr,
+ true, false);
+ for (u32 i = 0; i < total_resolve_nr; i++)
+ perf_bpf_to_sample_bid_offset(&buf->bids[i]);
+ } else {
+ put_cpu_ptr(&perf_bpf_build_id_buf);
+ buf = NULL;
+ }
+ }
+
+ if (sample_bid) {
+ if (defer_cookie) {
+ u64 deferred_bid[4] = {
+ 0,
+ PERF_CONTEXT_USER_DEFERRED,
+ defer_cookie,
+ data->ip,
+ };
+
+ perf_output_put(handle, deferred_bid);
+ } else if (buf && resolve_ip_nr) {
+ perf_output_copy(handle, &buf->bids[0],
+ sizeof(buf->bids[0]));
+ } else {
+ struct perf_sample_build_id_offset bid_offset = {
+ .offset = data->ip,
+ };
+
+ perf_output_put(handle, bid_offset);
+ }
+ }
+
+ if (callchain_bid) {
+ u64 unres_prefix = buf ? user_start : callchain_nr;
+ u64 resolved_end = unres_prefix + (buf ? resolve_cc_nr : 0);
+
+ perf_output_put(handle, callchain_nr);
+ for (u64 i = 0; i < unres_prefix; i++) {
+ struct perf_sample_build_id_offset bid_offset = {
+ .offset = data->callchain->ip[i],
+ };
+
+ perf_output_put(handle, bid_offset);
+ }
+ if (buf && resolve_cc_nr) {
+ perf_output_copy(handle, &buf->bids[resolve_ip_nr],
+ resolve_cc_nr * sizeof(buf->bids[0]));
+ }
+ for (u64 i = resolved_end; i < callchain_nr; i++) {
+ struct perf_sample_build_id_offset bid_offset = {
+ .offset = data->callchain->ip[i],
+ };
+
+ perf_output_put(handle, bid_offset);
+ }
+ }
+
+ if (buf) {
+ local_set(&buf->active, 0);
+ put_cpu_ptr(&perf_bpf_build_id_buf);
+ }
+}
+
void perf_output_sample(struct perf_output_handle *handle,
struct perf_event_header *header,
struct perf_sample_data *data,
@@ -8455,6 +8621,10 @@ void perf_output_sample(struct perf_output_handle *handle,
perf_aux_sample_output(event, handle, data);
}
+ if (sample_type &
+ (PERF_SAMPLE_BUILD_ID_OFFSET | PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET))
+ perf_output_sample_build_id(handle, header, data, event);
+
if (!event->attr.watermark) {
int wakeup_events = event->attr.wakeup_events;
@@ -8598,8 +8768,6 @@ static u64 perf_get_page_size(unsigned long addr)
static struct perf_callchain_entry __empty_callchain = { .nr = 0, };
-static struct unwind_work perf_unwind_work;
-
struct perf_callchain_entry *
perf_callchain(struct perf_event *event, struct pt_regs *regs)
{
@@ -8663,7 +8831,7 @@ void perf_prepare_sample(struct perf_sample_data *data,
__perf_event_header__init_id(data, event, filtered_sample_type);
- if (filtered_sample_type & PERF_SAMPLE_IP) {
+ if (filtered_sample_type & (PERF_SAMPLE_IP | PERF_SAMPLE_BUILD_ID_OFFSET)) {
data->ip = perf_instruction_pointer(event, regs);
data->sample_flags |= PERF_SAMPLE_IP;
}
@@ -8671,6 +8839,14 @@ void perf_prepare_sample(struct perf_sample_data *data,
if (filtered_sample_type & PERF_SAMPLE_CALLCHAIN)
perf_sample_save_callchain(data, event, regs);
+ if (filtered_sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ if (!(data->sample_flags & PERF_SAMPLE_CALLCHAIN))
+ data->callchain = perf_callchain(event, regs);
+
+ data->dyn_size += sizeof(u64);
+ }
+
+
if (filtered_sample_type & PERF_SAMPLE_RAW) {
data->raw = NULL;
data->dyn_size += sizeof(u64);
@@ -8820,6 +8996,20 @@ void perf_prepare_sample(struct perf_sample_data *data,
data->dyn_size += size + sizeof(u64); /* size above */
data->sample_flags |= PERF_SAMPLE_AUX;
}
+
+ if (filtered_sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ u32 header_size = perf_sample_data_size(data, event);
+ u64 max_nr = 0, nr;
+
+ if (header_size < U16_MAX) {
+ max_nr = (U16_MAX - header_size) /
+ sizeof(struct perf_sample_build_id_offset);
+ }
+ nr = min_t(u64, data->callchain->nr, max_nr);
+ data->callchain_bid_nr = nr;
+ data->dyn_size += nr * sizeof(struct perf_sample_build_id_offset);
+ data->sample_flags |= PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET;
+ }
}
void perf_prepare_header(struct perf_event_header *header,
@@ -10414,6 +10604,8 @@ void perf_event_bpf_event(struct bpf_prog *prog,
struct perf_callchain_deferred_event {
struct unwind_stacktrace *trace;
+ struct perf_sample_build_id_offset *bids;
+ u16 nr_bids;
struct {
struct perf_event_header header;
u64 cookie;
@@ -10422,17 +10614,83 @@ struct perf_callchain_deferred_event {
} event;
};
+static struct perf_sample_build_id_offset *
+perf_resolve_build_id_offsets(struct unwind_stacktrace *trace, u16 *nr_bids_out)
+{
+ struct perf_sample_build_id_offset *bids;
+ struct bpf_stack_build_id *bpf_bids;
+ u16 nr = trace->nr;
+
+ *nr_bids_out = 0;
+ if (!nr || !current->mm)
+ return NULL;
+
+ if (!atomic_read(&nr_callchain_build_id_offset_events)) {
+ if (!atomic_read(&nr_build_id_offset_events))
+ return NULL;
+ nr = 1;
+ }
+
+ bids = kcalloc(nr, sizeof(*bids), GFP_KERNEL);
+ if (!bids)
+ return NULL;
+
+ bpf_bids = (struct bpf_stack_build_id *)bids;
+ for (u16 i = 0; i < nr; i++)
+ bpf_bids[i].ip = trace->entries[i];
+
+ stack_map_get_build_id_offset(bpf_bids, nr, true, true);
+
+ for (u16 i = 0; i < nr; i++)
+ perf_bpf_to_sample_bid_offset(&bpf_bids[i]);
+
+ *nr_bids_out = nr;
+ return bids;
+}
+
static void perf_callchain_deferred_output(struct perf_event *event, void *data)
{
struct perf_callchain_deferred_event *deferred_event = data;
struct perf_output_handle handle;
struct perf_sample_data sample;
- int ret, size = deferred_event->event.header.size;
+ u64 nr = deferred_event->trace->nr;
+ size_t elem_size;
+ bool use_bid;
+ u16 header_size, orig_misc, id_size;
+ int ret;
if (!event->attr.defer_output)
return;
+ if (event->attr.sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ use_bid = deferred_event->bids != NULL;
+ if (use_bid)
+ nr = min_t(u64, nr, deferred_event->nr_bids);
+ } else if ((event->attr.sample_type & PERF_SAMPLE_BUILD_ID_OFFSET) &&
+ !(event->attr.sample_type & PERF_SAMPLE_CALLCHAIN)) {
+ nr = min_t(u64, nr, 1);
+ use_bid = deferred_event->bids != NULL;
+ if (use_bid)
+ nr = min_t(u64, nr, deferred_event->nr_bids);
+ } else {
+ use_bid = false;
+ }
+
+ elem_size = use_bid ? sizeof(struct perf_sample_build_id_offset) :
+ sizeof(u64);
+ id_size = event->attr.sample_id_all ? event->id_header_size : 0;
+ nr = min_t(u64, nr,
+ (U16_MAX - sizeof(deferred_event->event) - id_size) / elem_size);
+
+ orig_misc = deferred_event->event.header.misc;
+ if (use_bid)
+ deferred_event->event.header.misc |= PERF_RECORD_MISC_MMAP_BUILD_ID;
+ header_size = sizeof(deferred_event->event) + (nr * elem_size);
+ deferred_event->event.header.size = header_size;
+ deferred_event->event.nr = nr;
+
/* XXX do we really need sample_id_all for this ??? */
+ perf_sample_data_init(&sample, 0, 0);
perf_event_header__init_id(&deferred_event->event.header, &sample, event);
ret = perf_output_begin(&handle, &sample, event,
@@ -10441,22 +10699,33 @@ static void perf_callchain_deferred_output(struct perf_event *event, void *data)
goto out;
perf_output_put(&handle, deferred_event->event);
- for (int i = 0; i < deferred_event->trace->nr; i++) {
- u64 entry = deferred_event->trace->entries[i];
- perf_output_put(&handle, entry);
+ if (use_bid) {
+ for (u64 i = 0; i < nr; i++)
+ perf_output_put(&handle, deferred_event->bids[i]);
+ } else {
+ for (u64 i = 0; i < nr; i++) {
+ u64 entry = deferred_event->trace->entries[i];
+
+ perf_output_put(&handle, entry);
+ }
}
perf_event__output_id_sample(event, &handle, &sample);
perf_output_end(&handle);
out:
- deferred_event->event.header.size = size;
+ deferred_event->event.header.misc = orig_misc;
}
static void perf_unwind_deferred_callback(struct unwind_work *work,
struct unwind_stacktrace *trace, u64 cookie)
{
+ u16 nr_bids = 0;
+ struct perf_sample_build_id_offset *bids =
+ perf_resolve_build_id_offsets(trace, &nr_bids);
struct perf_callchain_deferred_event deferred_event = {
.trace = trace,
+ .bids = bids,
+ .nr_bids = nr_bids,
.event = {
.header = {
.type = PERF_RECORD_CALLCHAIN_DEFERRED,
@@ -10470,6 +10739,7 @@ static void perf_unwind_deferred_callback(struct unwind_work *work,
};
perf_iterate_sb(perf_callchain_deferred_output, &deferred_event, NULL);
+ kfree(bids);
}
struct perf_text_poke_event {
@@ -13314,6 +13584,10 @@ static void account_event(struct perf_event *event)
atomic_inc(&nr_mmap_events);
if (event->attr.build_id)
atomic_inc(&nr_build_id_events);
+ if (event->attr.sample_type & PERF_SAMPLE_BUILD_ID_OFFSET)
+ atomic_inc(&nr_build_id_offset_events);
+ if (event->attr.sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET)
+ atomic_inc(&nr_callchain_build_id_offset_events);
if (event->attr.comm)
atomic_inc(&nr_comm_events);
if (event->attr.namespaces)
@@ -13582,7 +13856,8 @@ perf_event_alloc(struct perf_event_attr *attr, int cpu,
}
if (!event->parent) {
- if (event->attr.sample_type & PERF_SAMPLE_CALLCHAIN) {
+ if (event->attr.sample_type &
+ (PERF_SAMPLE_CALLCHAIN | PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET)) {
err = get_callchain_buffers(attr->sample_max_stack);
if (err)
return ERR_PTR(err);
@@ -13943,7 +14218,8 @@ SYSCALL_DEFINE5(perf_event_open,
return err;
if (!attr.exclude_kernel ||
- ((attr.sample_type & PERF_SAMPLE_CALLCHAIN) &&
+ ((attr.sample_type &
+ (PERF_SAMPLE_CALLCHAIN | PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET)) &&
!attr.exclude_callchain_kernel)) {
err = perf_allow_kernel();
if (err)
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 3/8] perf/core: Implement BUILD_ID_OFFSET sample type
2026-10-02 17:38 ` [PATCH v2 3/8] perf/core: Implement BUILD_ID_OFFSET sample type Ian Rogers
@ 2026-10-03 1:33 ` sashiko-bot
0 siblings, 0 replies; 27+ messages in thread
From: sashiko-bot @ 2026-10-03 1:33 UTC (permalink / raw)
To: Ian Rogers; +Cc: linux-perf-users
> Implement PERF_SAMPLE_BUILD_ID_OFFSET and
> PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET payload writing in the perf event
> core by reusing BPF's stack_map_get_build_id_offset().
>
> When attr.defer_callchain is set, user-space IPs and callchains request
> a deferred unwind via unwind_deferred_request() and emit a
> PERF_CONTEXT_USER_DEFERRED marker and cookie in the sample. In the
> faultable task_work callback (perf_unwind_deferred_callback), the
> unwound user IPs are resolved to build IDs and file offsets using
> stack_map_get_build_id_offset(..., may_fault=true) and emitted in
> PERF_RECORD_CALLCHAIN_DEFERRED with PERF_RECORD_MISC_MMAP_BUILD_ID set.
>
> When attr.defer_callchain is not set (or if deferred unwinding is
> unavailable), user-space IPs and callchains are resolved in-place during
> perf_output_sample() using stack_map_get_build_id_offset(...,
> may_fault=false). If the mmap_lock cannot be trylocked or the ELF
> header/note pages are not present in the page cache (nor mlocked into
> memory), resolution falls back to emitting a zero-sized build ID
> (size = 0) with the virtual address as the offset. Using the deferred
> approach (attr.defer_callchain and attr.defer_output) is therefore
> preferred as it can fault in pages and wait on locks, resolving more
> build IDs without requiring ELF header pages to be mlocked in the page
> cache.
>
> Kernel-space IPs and callchain entries are emitted with a zero-sized
> build ID (size = 0) and the virtual address as the offset.
>
> Signed-off-by: Ian Rogers <irogers@google.com>
Sashiko has reviewed this patch and found no issues. It looks great!
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261002173848.3228217-1-irogers@google.com?part=3
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 4/8] perf: Refactor thread map and symbol APIs to take perf_sample
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
` (2 preceding siblings ...)
2026-10-02 17:38 ` [PATCH v2 3/8] perf/core: Implement BUILD_ID_OFFSET sample type Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 5/8] perf tools: Internal support for BUILD_ID_OFFSET Ian Rogers
` (3 subsequent siblings)
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Refactor thread__find_map and thread__find_symbol family of APIs to
take a struct perf_sample * instead of cpumode and addr. This prepares
for looking up symbols using build ID and offset, which is contained in
the sample payload instead of the address.
Signed-off-by: Ian Rogers <irogers@google.com>
---
.../arch/powerpc/util/skip-callchain-idx.c | 4 +-
tools/perf/builtin-inject.c | 2 +-
tools/perf/builtin-script.c | 40 +++++++++++----
tools/perf/builtin-timechart.c | 5 +-
tools/perf/builtin-trace.c | 10 ++--
tools/perf/tests/code-reading.c | 4 +-
tools/perf/tests/mmap-thread-lookup.c | 9 +++-
tools/perf/util/annotate-data.c | 5 +-
tools/perf/util/aslr.c | 16 ++++--
tools/perf/util/build-id.c | 2 +-
tools/perf/util/capstone.c | 4 +-
tools/perf/util/cs-etm.c | 4 +-
tools/perf/util/data-convert-json.c | 4 +-
tools/perf/util/debug.c | 4 +-
tools/perf/util/dlfilter.c | 8 ++-
tools/perf/util/event.c | 51 ++++++++++---------
tools/perf/util/intel-pt.c | 14 +++--
tools/perf/util/machine.c | 18 +++++--
tools/perf/util/python.c | 18 +++++--
tools/perf/util/thread.c | 11 ++--
tools/perf/util/thread.h | 14 ++---
tools/perf/util/unwind-libdw.c | 7 ++-
tools/perf/util/unwind-libunwind.c | 8 ++-
23 files changed, 179 insertions(+), 83 deletions(-)
diff --git a/tools/perf/arch/powerpc/util/skip-callchain-idx.c b/tools/perf/arch/powerpc/util/skip-callchain-idx.c
index 472714cfad38..cb29d9eafa89 100644
--- a/tools/perf/arch/powerpc/util/skip-callchain-idx.c
+++ b/tools/perf/arch/powerpc/util/skip-callchain-idx.c
@@ -225,7 +225,9 @@ int arch_skip_callchain_idx(struct thread *thread, struct ip_callchain *chain)
addr_location__init(&al);
ip = chain->ips[1];
- thread__find_symbol(thread, PERF_RECORD_MISC_USER, ip, &al);
+ thread__find_symbol(thread,
+ &(struct perf_sample){.cpumode = PERF_RECORD_MISC_USER, .ip = ip},
+ &al);
if (al.map)
dso = map__dso(al.map);
diff --git a/tools/perf/builtin-inject.c b/tools/perf/builtin-inject.c
index 67f019acb6c0..42ed367d729f 100644
--- a/tools/perf/builtin-inject.c
+++ b/tools/perf/builtin-inject.c
@@ -1167,7 +1167,7 @@ static int perf_event__inject_buildid(const struct perf_tool *tool, union perf_e
goto repipe;
}
- if (thread__find_map(thread, sample->cpumode, sample->ip, &al)) {
+ if (thread__find_map(thread, sample, &al)) {
mark_dso_hit(inject, tool, sample, machine, args.mmap_evsel, al.map,
/*sample_in_dso=*/true);
}
diff --git a/tools/perf/builtin-script.c b/tools/perf/builtin-script.c
index b6691ebb1b4e..99e9e0c60f14 100644
--- a/tools/perf/builtin-script.c
+++ b/tools/perf/builtin-script.c
@@ -1063,6 +1063,7 @@ static int perf_sample__fprintf_brstack(struct perf_sample *sample,
{
struct branch_stack *br = sample->branch_stack;
struct branch_entry *entries = perf_sample__branch_entries(sample);
+ struct perf_sample br_sample = {};
u64 i, from, to;
int printed = 0;
@@ -1079,8 +1080,12 @@ static int perf_sample__fprintf_brstack(struct perf_sample *sample,
addr_location__init(&alf);
addr_location__init(&alt);
- thread__find_map_fb(thread, sample->cpumode, from, &alf);
- thread__find_map_fb(thread, sample->cpumode, to, &alt);
+ br_sample.cpumode = sample->cpumode;
+ br_sample.ip = from;
+ thread__find_map_fb(thread, &br_sample, &alf);
+ br_sample.cpumode = sample->cpumode;
+ br_sample.ip = to;
+ thread__find_map_fb(thread, &br_sample, &alt);
printed += map__fprintf_dsoname_dsoff(alf.map, PRINT_FIELD(DSOFF), alf.addr, fp);
printed += fprintf(fp, "/0x%"PRIx64, to);
@@ -1102,6 +1107,7 @@ static int perf_sample__fprintf_brstacksym(struct perf_sample *sample,
{
struct branch_stack *br = sample->branch_stack;
struct branch_entry *entries = perf_sample__branch_entries(sample);
+ struct perf_sample br_sample = {};
u64 i, from, to;
int printed = 0;
@@ -1116,8 +1122,12 @@ static int perf_sample__fprintf_brstacksym(struct perf_sample *sample,
from = entries[i].from;
to = entries[i].to;
- thread__find_symbol_fb(thread, sample->cpumode, from, &alf);
- thread__find_symbol_fb(thread, sample->cpumode, to, &alt);
+ br_sample.cpumode = sample->cpumode;
+ br_sample.ip = from;
+ thread__find_symbol_fb(thread, &br_sample, &alf);
+ br_sample.cpumode = sample->cpumode;
+ br_sample.ip = to;
+ thread__find_symbol_fb(thread, &br_sample, &alt);
printed += symbol__fprintf_symname_offs(alf.sym, &alf, fp);
if (PRINT_FIELD(DSO))
@@ -1140,6 +1150,7 @@ static int perf_sample__fprintf_brstackoff(struct perf_sample *sample,
{
struct branch_stack *br = sample->branch_stack;
struct branch_entry *entries = perf_sample__branch_entries(sample);
+ struct perf_sample br_sample = {};
u64 i, from, to;
int printed = 0;
@@ -1154,11 +1165,15 @@ static int perf_sample__fprintf_brstackoff(struct perf_sample *sample,
from = entries[i].from;
to = entries[i].to;
- if (thread__find_map_fb(thread, sample->cpumode, from, &alf) &&
+ br_sample.cpumode = sample->cpumode;
+ br_sample.ip = from;
+ if (thread__find_map_fb(thread, &br_sample, &alf) &&
!dso__adjust_symbols(map__dso(alf.map)))
from = map__dso_map_ip(alf.map, from);
- if (thread__find_map_fb(thread, sample->cpumode, to, &alt) &&
+ br_sample.cpumode = sample->cpumode;
+ br_sample.ip = to;
+ if (thread__find_map_fb(thread, &br_sample, &alt) &&
!dso__adjust_symbols(map__dso(alt.map)))
to = map__dso_map_ip(alt.map, to);
@@ -1216,7 +1231,10 @@ static int grab_bb(u8 *buffer, u64 start, u64 end,
}
addr_location__init(&al);
- if (!thread__find_map(thread, *cpumode, start, &al) || (dso = map__dso(al.map)) == NULL) {
+ if (!thread__find_map(thread,
+ &(struct perf_sample){.cpumode = *cpumode,
+ .ip = start}, &al) ||
+ (dso = map__dso(al.map)) == NULL) {
pr_debug("\tcannot resolve %" PRIx64 "-%" PRIx64 "\n", start, end);
goto out;
}
@@ -1291,7 +1309,7 @@ static int print_srccode(struct thread *thread, u8 cpumode, uint64_t addr)
int ret = 0;
addr_location__init(&al);
- thread__find_map(thread, cpumode, addr, &al);
+ thread__find_map(thread, &(struct perf_sample){.cpumode = cpumode, .ip = addr}, &al);
if (!al.map)
goto out;
ret = map__fprintf_srccode(al.map, al.addr, stdout,
@@ -1346,7 +1364,9 @@ static int ip__fprintf_jump(uint64_t ip, struct branch_entry *en,
struct addr_location al;
addr_location__init(&al);
- thread__find_map(thread, x->cpumode, ip, &al);
+ thread__find_map(thread,
+ &(struct perf_sample){.cpumode = x->cpumode,
+ .ip = ip}, &al);
printed += map__fprintf_srcline(al.map, al.addr, " srcline: ", fp);
printed += fprintf(fp, "\t");
addr_location__exit(&al);
@@ -1406,7 +1426,7 @@ static int ip__fprintf_sym(uint64_t addr, struct thread *thread,
int off, printed = 0, ret = 0;
addr_location__init(&al);
- thread__find_map(thread, cpumode, addr, &al);
+ thread__find_map(thread, &(struct perf_sample){.cpumode = cpumode, .ip = addr}, &al);
if ((*lastsym) && al.addr >= (*lastsym)->start && al.addr < (*lastsym)->end)
goto out;
diff --git a/tools/perf/builtin-timechart.c b/tools/perf/builtin-timechart.c
index 3d88c90c6573..745fd04e478a 100644
--- a/tools/perf/builtin-timechart.c
+++ b/tools/perf/builtin-timechart.c
@@ -512,6 +512,7 @@ static char *cat_backtrace(struct perf_sample *sample,
size_t p_len;
u8 cpumode = PERF_RECORD_MISC_USER;
struct ip_callchain *chain = sample->callchain;
+ struct perf_sample ip_sample = {};
FILE *f = open_memstream(&p, &p_len);
bool corrupted = false;
@@ -560,7 +561,9 @@ static char *cat_backtrace(struct perf_sample *sample,
addr_location__init(&tal);
tal.filtered = 0;
- if (thread__find_symbol(al.thread, cpumode, ip, &tal))
+ ip_sample.cpumode = cpumode;
+ ip_sample.ip = ip;
+ if (thread__find_symbol(al.thread, &ip_sample, &tal))
fprintf(f, "..... %016" PRIx64 " %s\n", ip, tal.sym->name);
else
fprintf(f, "..... %016" PRIx64 "\n", ip);
diff --git a/tools/perf/builtin-trace.c b/tools/perf/builtin-trace.c
index d327603ae454..15b73580c5c2 100644
--- a/tools/perf/builtin-trace.c
+++ b/tools/perf/builtin-trace.c
@@ -3700,7 +3700,7 @@ static int trace__pgfault(struct trace *trace,
if (trace->summary_only)
goto out;
- thread__find_symbol(thread, sample->cpumode, sample->ip, &al);
+ thread__find_symbol(thread, sample, &al);
trace__fprintf_entry_head(trace, thread, 0, true, sample->time,
sample->cpu, trace->output);
@@ -3713,10 +3713,14 @@ static int trace__pgfault(struct trace *trace,
fprintf(trace->output, "] => ");
- thread__find_symbol(thread, sample->cpumode, sample->addr, &al);
+ thread__find_symbol(thread,
+ &(struct perf_sample){.cpumode = sample->cpumode,
+ .ip = sample->addr}, &al);
if (!al.map) {
- thread__find_symbol(thread, sample->cpumode, sample->addr, &al);
+ thread__find_symbol(thread,
+ &(struct perf_sample){.cpumode = sample->cpumode,
+ .ip = sample->addr}, &al);
if (al.map)
map_type = 'x';
diff --git a/tools/perf/tests/code-reading.c b/tools/perf/tests/code-reading.c
index c50b526f44fb..f339e2157dfc 100644
--- a/tools/perf/tests/code-reading.c
+++ b/tools/perf/tests/code-reading.c
@@ -394,7 +394,9 @@ static int read_object_code(u64 addr, size_t len, u8 cpumode,
pr_debug("Reading object code for memory address: %#"PRIx64"\n", addr);
addr_location__init(&al);
- if (!thread__find_map(thread, cpumode, addr, &al) || !map__dso(al.map)) {
+ if (!thread__find_map(thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = addr}, &al) || !map__dso(al.map)) {
if (cpumode == PERF_RECORD_MISC_HYPERVISOR) {
pr_debug("Hypervisor address can not be resolved - skipping\n");
goto out;
diff --git a/tools/perf/tests/mmap-thread-lookup.c b/tools/perf/tests/mmap-thread-lookup.c
index 0c5619c6e6e9..dca8bbfe780a 100644
--- a/tools/perf/tests/mmap-thread-lookup.c
+++ b/tools/perf/tests/mmap-thread-lookup.c
@@ -195,8 +195,10 @@ static int mmap_events(synth_cb synth)
pr_debug("looking for map %p\n", td->map);
- thread__find_map(thread, PERF_RECORD_MISC_USER,
- (unsigned long) (td->map + 1), &al);
+ thread__find_map(thread, &(struct perf_sample){
+ .cpumode = PERF_RECORD_MISC_USER,
+ .ip = (unsigned long) (td->map + 1),
+ }, &al);
thread__put(thread);
@@ -208,7 +210,10 @@ static int mmap_events(synth_cb synth)
}
pr_debug("map %p, addr %" PRIx64 "\n", al.map, map__start(al.map));
+
addr_location__exit(&al);
+ if (err)
+ break;
}
machine__delete(machine);
diff --git a/tools/perf/util/annotate-data.c b/tools/perf/util/annotate-data.c
index 19a6ecd67f28..b53dcb11bbb0 100644
--- a/tools/perf/util/annotate-data.c
+++ b/tools/perf/util/annotate-data.c
@@ -848,8 +848,9 @@ bool get_global_var_info(struct data_loc_info *dloc, u64 addr,
mem_addr = addr + map__reloc(dloc->ms->map);
addr_location__init(&al);
- sym = thread__find_symbol_fb(dloc->thread, dloc->cpumode,
- mem_addr, &al);
+ sym = thread__find_symbol_fb(dloc->thread,
+ &(struct perf_sample){.cpumode = dloc->cpumode,
+ .ip = mem_addr}, &al);
if (sym) {
*var_name = sym->name;
/* Calculate type offset from the start of variable */
diff --git a/tools/perf/util/aslr.c b/tools/perf/util/aslr.c
index e365f807b021..acda28a6fd2d 100644
--- a/tools/perf/util/aslr.c
+++ b/tools/perf/util/aslr.c
@@ -155,7 +155,9 @@ static u64 aslr_tool__remap_address(struct aslr_tool *aslr,
return 0; /* No thread. */
addr_location__init(&al);
- if (!thread__find_map(aslr_thread, cpumode, addr, &al)) {
+ if (!thread__find_map(aslr_thread,
+ &(struct perf_sample){.cpumode = cpumode, .ip = addr},
+ &al)) {
/*
* If lookup fails with specified cpumode, try fallback to the other space
* to be robust against bad cpumode in samples.
@@ -169,7 +171,9 @@ static u64 aslr_tool__remap_address(struct aslr_tool *aslr,
else if (cpumode == PERF_RECORD_MISC_GUEST_USER)
effective_cpumode = PERF_RECORD_MISC_GUEST_KERNEL;
- if (!thread__find_map(aslr_thread, effective_cpumode, addr, &al)) {
+ if (!thread__find_map(aslr_thread,
+ &(struct perf_sample){.cpumode = effective_cpumode,
+ .ip = addr}, &al)) {
addr_location__exit(&al);
return 0; /* No mmap. */
}
@@ -286,7 +290,9 @@ static u64 aslr_tool__findnew_mapping(struct aslr_tool *aslr,
remap_key.pid = (cpumode == PERF_RECORD_MISC_KERNEL ||
cpumode == PERF_RECORD_MISC_GUEST_KERNEL) ?
kernel_pid : thread__pid(aslr_thread);
- if (thread__find_map(aslr_thread, cpumode, start, &al)) {
+ if (thread__find_map(aslr_thread,
+ &(struct perf_sample){.cpumode = cpumode, .ip = start},
+ &al)) {
struct dso *dso = map__dso(al.map);
const char *dso_name = dso ? dso__long_name(dso) : NULL;
@@ -343,7 +349,9 @@ static u64 aslr_tool__findnew_mapping(struct aslr_tool *aslr,
remap_addr = top->remapped_max;
addr_location__init(&prev_al);
- if (thread__find_map(aslr_thread, cpumode, start - 1, &prev_al)) {
+ if (thread__find_map(aslr_thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = start - 1}, &prev_al)) {
if (map__end(prev_al.map) == start)
is_contiguous = true;
}
diff --git a/tools/perf/util/build-id.c b/tools/perf/util/build-id.c
index 02233ef0eaff..6c0d33b1609d 100644
--- a/tools/perf/util/build-id.c
+++ b/tools/perf/util/build-id.c
@@ -69,7 +69,7 @@ int build_id__mark_dso_hit(const struct perf_tool *tool __maybe_unused,
}
addr_location__init(&al);
- if (thread__find_map(thread, sample->cpumode, sample->ip, &al))
+ if (thread__find_map(thread, sample, &al))
dso__set_hit(map__dso(al.map));
addr_location__exit(&al);
diff --git a/tools/perf/util/capstone.c b/tools/perf/util/capstone.c
index 95881f15324a..fbad96e1e6a9 100644
--- a/tools/perf/util/capstone.c
+++ b/tools/perf/util/capstone.c
@@ -252,7 +252,9 @@ static size_t print_insn_x86(struct thread *thread, u8 cpumode, struct cs_insn *
addr_location__init(&al);
if (op->type == X86_OP_IMM &&
- thread__find_symbol(thread, cpumode, op->imm, &al)) {
+ thread__find_symbol(thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = op->imm}, &al)) {
printed += fprintf(fp, "%s ", insn[0].mnemonic);
printed += symbol__fprintf_symname_offs(al.sym, &al, fp);
if (print_opts & PRINT_INSN_IMM_HEX)
diff --git a/tools/perf/util/cs-etm.c b/tools/perf/util/cs-etm.c
index 2d1ab34f7b6b..31a11b802c1e 100644
--- a/tools/perf/util/cs-etm.c
+++ b/tools/perf/util/cs-etm.c
@@ -1152,7 +1152,9 @@ static u32 __cs_etm__mem_access(struct cs_etm_queue *etmq,
cpumode = cs_etm__cpu_mode(etmq, address, el);
- if (!thread__find_map(thread, cpumode, address, &al))
+ if (!thread__find_map(thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = address}, &al))
goto out;
dso = map__dso(al.map);
diff --git a/tools/perf/util/data-convert-json.c b/tools/perf/util/data-convert-json.c
index 40888b7c4467..8fa7bfdaadff 100644
--- a/tools/perf/util/data-convert-json.c
+++ b/tools/perf/util/data-convert-json.c
@@ -235,7 +235,9 @@ static int process_sample_event(const struct perf_tool *tool,
fputc(',', out);
addr_location__init(&tal);
- ok = thread__find_symbol(al.thread, cpumode, ip, &tal);
+ ok = thread__find_symbol(al.thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = ip}, &tal);
output_sample_callchain_entry(tool, ip, ok ? &tal : NULL);
addr_location__exit(&tal);
}
diff --git a/tools/perf/util/debug.c b/tools/perf/util/debug.c
index b7095519419e..2f31b3663ec1 100644
--- a/tools/perf/util/debug.c
+++ b/tools/perf/util/debug.c
@@ -339,7 +339,9 @@ void __dump_stack(FILE *file, void **stackdump, size_t stackdump_size)
bool printed = false;
addr_location__init(&al);
- if (thread && thread__find_map(thread, PERF_RECORD_MISC_USER, addr, &al)) {
+ if (thread && thread__find_map(thread,
+ &(struct perf_sample){.cpumode = PERF_RECORD_MISC_USER,
+ .ip = addr}, &al)) {
al.sym = map__find_symbol(al.map, al.addr);
if (al.sym) {
fprintf(file, " #%zd %p in %s ", i, stackdump[i], al.sym->name);
diff --git a/tools/perf/util/dlfilter.c b/tools/perf/util/dlfilter.c
index e11e144af62b..8a33bfe50939 100644
--- a/tools/perf/util/dlfilter.c
+++ b/tools/perf/util/dlfilter.c
@@ -177,7 +177,9 @@ static __s32 dlfilter__resolve_address(void *ctx, __u64 address, struct perf_dlf
return -1;
addr_location__init(&al);
- thread__find_symbol_fb(thread, d->sample->cpumode, address, &al);
+ thread__find_symbol_fb(thread,
+ &(struct perf_sample){.cpumode = d->sample->cpumode,
+ .ip = address}, &al);
al_to_d_al(&al, &d_al);
@@ -314,7 +316,9 @@ static __s32 dlfilter__object_code(void *ctx, __u64 ip, void *buf, __u32 len)
addr_location__init(&a);
- thread__find_map_fb(al->thread, d->sample->cpumode, ip, &a);
+ thread__find_map_fb(al->thread,
+ &(struct perf_sample){.cpumode = d->sample->cpumode,
+ .ip = ip}, &a);
ret = a.map ? code_read(ip, a.map, d->machine, buf, len) : -1;
addr_location__exit(&a);
diff --git a/tools/perf/util/event.c b/tools/perf/util/event.c
index c69ae57ce679..cd13a0c3cb52 100644
--- a/tools/perf/util/event.c
+++ b/tools/perf/util/event.c
@@ -691,7 +691,7 @@ int perf_event__process(const struct perf_tool *tool __maybe_unused,
return machine__process_event(machine, event, sample);
}
-struct map *thread__find_map(struct thread *thread, u8 cpumode, u64 addr,
+struct map *thread__find_map(struct thread *thread, struct perf_sample *sample,
struct addr_location *al)
{
struct maps *maps = thread__maps(thread);
@@ -702,34 +702,34 @@ struct map *thread__find_map(struct thread *thread, u8 cpumode, u64 addr,
thread__zput(al->thread);
al->thread = thread__get(thread);
- al->addr = addr;
- al->cpumode = cpumode;
+ al->addr = sample->ip;
+ al->cpumode = sample->cpumode;
al->filtered = 0;
if (machine == NULL)
return NULL;
- if (cpumode == PERF_RECORD_MISC_KERNEL && perf_host) {
+ if (sample->cpumode == PERF_RECORD_MISC_KERNEL && perf_host) {
al->level = 'k';
maps = machine__kernel_maps(machine);
load_map = !symbol_conf.lazy_load_kernel_maps;
- } else if (cpumode == PERF_RECORD_MISC_USER && perf_host) {
+ } else if (sample->cpumode == PERF_RECORD_MISC_USER && perf_host) {
al->level = '.';
- } else if (cpumode == PERF_RECORD_MISC_GUEST_KERNEL && perf_guest) {
+ } else if (sample->cpumode == PERF_RECORD_MISC_GUEST_KERNEL && perf_guest) {
al->level = 'g';
maps = machine__kernel_maps(machine);
load_map = !symbol_conf.lazy_load_kernel_maps;
- } else if (cpumode == PERF_RECORD_MISC_GUEST_USER && perf_guest) {
+ } else if (sample->cpumode == PERF_RECORD_MISC_GUEST_USER && perf_guest) {
al->level = 'u';
} else {
al->level = 'H';
- if ((cpumode == PERF_RECORD_MISC_GUEST_USER ||
- cpumode == PERF_RECORD_MISC_GUEST_KERNEL) &&
+ if ((sample->cpumode == PERF_RECORD_MISC_GUEST_USER ||
+ sample->cpumode == PERF_RECORD_MISC_GUEST_KERNEL) &&
!perf_guest)
al->filtered |= (1 << HIST_FILTER__GUEST);
- if ((cpumode == PERF_RECORD_MISC_USER ||
- cpumode == PERF_RECORD_MISC_KERNEL) &&
+ if ((sample->cpumode == PERF_RECORD_MISC_USER ||
+ sample->cpumode == PERF_RECORD_MISC_KERNEL) &&
!perf_host)
al->filtered |= (1 << HIST_FILTER__HOST);
@@ -754,33 +754,34 @@ struct map *thread__find_map(struct thread *thread, u8 cpumode, u64 addr,
* because it applies only to the sample 'ip' and not necessary to 'addr' or
* branch stack addresses. If possible, use a fallback to deal with those cases.
*/
-struct map *thread__find_map_fb(struct thread *thread, u8 cpumode, u64 addr,
+struct map *thread__find_map_fb(struct thread *thread, struct perf_sample *sample,
struct addr_location *al)
{
- struct map *map = thread__find_map(thread, cpumode, addr, al);
+ struct map *map = thread__find_map(thread, sample, al);
struct machine *machine = maps__machine(thread__maps(thread));
- u8 addr_cpumode = machine__addr_cpumode(machine, cpumode, addr);
+ u8 addr_cpumode = machine__addr_cpumode(machine, sample->cpumode, sample->ip);
- if (map || addr_cpumode == cpumode)
+ if (map || addr_cpumode == sample->cpumode)
return map;
- return thread__find_map(thread, addr_cpumode, addr, al);
+ sample->cpumode = addr_cpumode;
+ return thread__find_map(thread, sample, al);
}
-struct symbol *thread__find_symbol(struct thread *thread, u8 cpumode,
- u64 addr, struct addr_location *al)
+struct symbol *thread__find_symbol(struct thread *thread, struct perf_sample *sample,
+ struct addr_location *al)
{
al->sym = NULL;
- if (thread__find_map(thread, cpumode, addr, al))
+ if (thread__find_map(thread, sample, al))
al->sym = map__find_symbol(al->map, al->addr);
return al->sym;
}
-struct symbol *thread__find_symbol_fb(struct thread *thread, u8 cpumode,
- u64 addr, struct addr_location *al)
+struct symbol *thread__find_symbol_fb(struct thread *thread, struct perf_sample *sample,
+ struct addr_location *al)
{
al->sym = NULL;
- if (thread__find_map_fb(thread, cpumode, addr, al))
+ if (thread__find_map_fb(thread, sample, al))
al->sym = map__find_symbol(al->map, al->addr);
return al->sym;
}
@@ -816,7 +817,7 @@ int machine__resolve(struct machine *machine, struct addr_location *al,
return -1;
dump_printf(" ... thread: %s:%d\n", thread__comm_str(thread), thread__tid(thread));
- thread__find_map(thread, sample->cpumode, sample->ip, al);
+ thread__find_map(thread, sample, al);
dso = al->map ? map__dso(al->map) : NULL;
dump_printf(" ...... dso: %s\n",
dso
@@ -934,7 +935,9 @@ bool sample_addr_correlates_sym(struct perf_event_attr *attr)
void thread__resolve(struct thread *thread, struct addr_location *al,
struct perf_sample *sample)
{
- thread__find_map_fb(thread, sample->cpumode, sample->addr, al);
+ thread__find_map_fb(thread,
+ &(struct perf_sample){.cpumode = sample->cpumode,
+ .ip = sample->addr}, al);
al->cpu = sample->cpu;
al->sym = NULL;
diff --git a/tools/perf/util/intel-pt.c b/tools/perf/util/intel-pt.c
index 8c21c9f52d57..243c252276ad 100644
--- a/tools/perf/util/intel-pt.c
+++ b/tools/perf/util/intel-pt.c
@@ -752,6 +752,7 @@ static int intel_pt_walk_next_insn(struct intel_pt_insn *intel_pt_insn,
{
struct intel_pt_queue *ptq = data;
struct machine *machine = ptq->pt->machine;
+ struct perf_sample sample = {};
struct thread *thread;
struct addr_location al;
unsigned char buf[INTEL_PT_INSN_BUF_SZ];
@@ -809,10 +810,13 @@ static int intel_pt_walk_next_insn(struct intel_pt_insn *intel_pt_insn,
}
}
+ sample.cpumode = cpumode;
+
while (1) {
struct dso *dso;
- if (!thread__find_map(thread, cpumode, *ip, &al) || !map__dso(al.map)) {
+ sample.ip = *ip;
+ if (!thread__find_map(thread, &sample, &al) || !map__dso(al.map)) {
if (al.map)
intel_pt_log("ERROR: thread has no dso for %#" PRIx64 "\n", *ip);
else
@@ -1007,7 +1011,9 @@ static int __intel_pt_pgd_ip(uint64_t ip, void *data)
return -EINVAL;
addr_location__init(&al);
- if (!thread__find_map(thread, cpumode, ip, &al) || !map__dso(al.map))
+ if (!thread__find_map(thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = ip}, &al) || !map__dso(al.map))
return -EINVAL;
offset = map__map_ip(al.map, ip);
@@ -3656,7 +3662,9 @@ static int intel_pt_find_map(struct thread *thread, u8 cpumode, u64 addr,
struct addr_location *al)
{
if (!al->map || addr < map__start(al->map) || addr >= map__end(al->map)) {
- if (!thread__find_map(thread, cpumode, addr, al))
+ if (!thread__find_map(thread,
+ &(struct perf_sample){.cpumode = cpumode,
+ .ip = addr}, al))
return -1;
}
diff --git a/tools/perf/util/machine.c b/tools/perf/util/machine.c
index 4bcb16da6481..1a13253d0c7f 100644
--- a/tools/perf/util/machine.c
+++ b/tools/perf/util/machine.c
@@ -2064,6 +2064,7 @@ static void ip__resolve_ams(struct thread *thread,
struct addr_map_symbol *ams,
u64 ip)
{
+ static __thread struct perf_sample sample;
struct addr_location al;
addr_location__init(&al);
@@ -2074,7 +2075,8 @@ static void ip__resolve_ams(struct thread *thread,
* Thus, we have to try consecutively until we find a match
* or else, the symbol is unknown
*/
- thread__find_cpumode_addr_location(thread, ip, /*symbols=*/true, &al);
+ sample.ip = ip;
+ thread__find_cpumode_addr_location(thread, &sample, /*symbols=*/true, &al);
ams->addr = ip;
ams->al_addr = al.addr;
@@ -2091,11 +2093,14 @@ static void ip__resolve_data(struct thread *thread,
u8 m, struct addr_map_symbol *ams,
u64 addr, u64 phys_addr, u64 daddr_page_size)
{
+ static __thread struct perf_sample sample;
struct addr_location al;
addr_location__init(&al);
- thread__find_symbol(thread, m, addr, &al);
+ sample.cpumode = m;
+ sample.ip = addr;
+ thread__find_symbol(thread, &sample, &al);
ams->addr = addr;
ams->al_addr = al.addr;
@@ -2218,6 +2223,7 @@ static int add_callchain_ip(struct thread *thread,
u64 branch_from,
bool symbols)
{
+ static __thread struct perf_sample sample;
struct map_symbol ms = {};
struct addr_location al;
int nr_loop_iter = 0, err = 0;
@@ -2228,8 +2234,9 @@ static int add_callchain_ip(struct thread *thread,
al.filtered = 0;
al.sym = NULL;
al.srcline = NULL;
+ sample.ip = ip;
if (!cpumode) {
- thread__find_cpumode_addr_location(thread, ip, symbols, &al);
+ thread__find_cpumode_addr_location(thread, &sample, symbols, &al);
} else {
if (ip >= PERF_CONTEXT_MAX) {
switch (ip) {
@@ -2256,10 +2263,11 @@ static int add_callchain_ip(struct thread *thread,
}
goto out;
}
+ sample.cpumode = *cpumode;
if (symbols)
- thread__find_symbol(thread, *cpumode, ip, &al);
+ thread__find_symbol(thread, &sample, &al);
else
- thread__find_map(thread, *cpumode, ip, &al);
+ thread__find_map(thread, &sample, &al);
}
if (al.sym != NULL) {
diff --git a/tools/perf/util/python.c b/tools/perf/util/python.c
index cc9257e2cceb..a1f156fadc44 100644
--- a/tools/perf/util/python.c
+++ b/tools/perf/util/python.c
@@ -851,7 +851,10 @@ static PyObject *pyrf_sample_event__srccode(PyObject *self, PyObject *args)
if (addr != pevent->sample.ip) {
addr_location__init(&al);
- thread__find_symbol_fb(pevent->al.thread, pevent->sample.cpumode, addr, &al);
+ thread__find_symbol_fb(pevent->al.thread,
+ &(struct perf_sample){
+ .cpumode = pevent->sample.cpumode,
+ .ip = addr}, &al);
} else {
addr_location__init(&al);
al.thread = thread__get(pevent->al.thread);
@@ -1242,8 +1245,12 @@ static int pyrf_sample_event__resolve_addr_al(struct pyrf_event *pevent,
if (pyrf_sample_event__resolve_al(pevent) < 0 || !pevent->al.thread)
return -1;
- thread__find_symbol_fb(pevent->al.thread, pevent->sample.cpumode,
- pevent->sample.addr, addr_al);
+ thread__find_symbol_fb(pevent->al.thread,
+ &(struct perf_sample){
+ .cpumode = pevent->sample.cpumode,
+ .ip = pevent->sample.addr,
+ },
+ addr_al);
return 0;
}
@@ -4363,6 +4370,7 @@ static PyObject *pyrf_call_path__to_callchain(const struct call_path *cp,
struct thread *thread)
{
struct pyrf_callchain *pchain;
+ struct perf_sample sample = {};
const struct call_path *pos;
u64 nr_frames = 0;
@@ -4395,7 +4403,9 @@ static PyObject *pyrf_call_path__to_callchain(const struct call_path *cp,
: PERF_RECORD_MISC_USER;
addr_location__init(&al);
- thread__find_map(thread, cpumode, pos->ip, &al);
+ sample.cpumode = cpumode;
+ sample.ip = pos->ip;
+ thread__find_map(thread, &sample, &al);
frame->map = map__get(al.map);
if (frame->map)
frame->sym = pos->sym;
diff --git a/tools/perf/util/thread.c b/tools/perf/util/thread.c
index 91118c301913..ad3f1ae8a92e 100644
--- a/tools/perf/util/thread.c
+++ b/tools/perf/util/thread.c
@@ -413,7 +413,7 @@ int thread__fork(struct thread *thread, struct thread *parent, u64 timestamp, bo
return thread__clone_maps(thread, parent, do_maps_clone);
}
-void thread__find_cpumode_addr_location(struct thread *thread, u64 addr,
+void thread__find_cpumode_addr_location(struct thread *thread, struct perf_sample *sample,
bool symbols, struct addr_location *al)
{
size_t i;
@@ -425,10 +425,11 @@ void thread__find_cpumode_addr_location(struct thread *thread, u64 addr,
};
for (i = 0; i < ARRAY_SIZE(cpumodes); i++) {
+ sample->cpumode = cpumodes[i];
if (symbols)
- thread__find_symbol(thread, cpumodes[i], addr, al);
+ thread__find_symbol(thread, sample, al);
else
- thread__find_map(thread, cpumodes[i], addr, al);
+ thread__find_map(thread, sample, al);
if (al->map)
break;
@@ -579,6 +580,7 @@ int thread__memcpy(struct thread *thread, struct machine *machine,
void *buf, u64 ip, int len, bool *is64bit)
{
u8 cpumode = PERF_RECORD_MISC_USER;
+ struct perf_sample sample = { .ip = ip };
struct addr_location al;
struct dso *dso;
long offset;
@@ -587,7 +589,8 @@ int thread__memcpy(struct thread *thread, struct machine *machine,
cpumode = PERF_RECORD_MISC_KERNEL;
addr_location__init(&al);
- if (!thread__find_map(thread, cpumode, ip, &al)) {
+ sample.cpumode = cpumode;
+ if (!thread__find_map(thread, &sample, &al)) {
addr_location__exit(&al);
return -1;
}
diff --git a/tools/perf/util/thread.h b/tools/perf/util/thread.h
index d82fce8173ae..f0c50eafeee5 100644
--- a/tools/perf/util/thread.h
+++ b/tools/perf/util/thread.h
@@ -124,17 +124,17 @@ size_t thread__fprintf(struct thread *thread, FILE *fp);
struct thread *thread__main_thread(struct machine *machine, struct thread *thread);
-struct map *thread__find_map(struct thread *thread, u8 cpumode, u64 addr,
+struct map *thread__find_map(struct thread *thread, struct perf_sample *sample,
struct addr_location *al);
-struct map *thread__find_map_fb(struct thread *thread, u8 cpumode, u64 addr,
+struct map *thread__find_map_fb(struct thread *thread, struct perf_sample *sample,
struct addr_location *al);
-struct symbol *thread__find_symbol(struct thread *thread, u8 cpumode,
- u64 addr, struct addr_location *al);
-struct symbol *thread__find_symbol_fb(struct thread *thread, u8 cpumode,
- u64 addr, struct addr_location *al);
+struct symbol *thread__find_symbol(struct thread *thread, struct perf_sample *sample,
+ struct addr_location *al);
+struct symbol *thread__find_symbol_fb(struct thread *thread, struct perf_sample *sample,
+ struct addr_location *al);
-void thread__find_cpumode_addr_location(struct thread *thread, u64 addr,
+void thread__find_cpumode_addr_location(struct thread *thread, struct perf_sample *sample,
bool symbols, struct addr_location *al);
int thread__memcpy(struct thread *thread, struct machine *machine,
diff --git a/tools/perf/util/unwind-libdw.c b/tools/perf/util/unwind-libdw.c
index d49901b1571e..855faa54c0cf 100644
--- a/tools/perf/util/unwind-libdw.c
+++ b/tools/perf/util/unwind-libdw.c
@@ -82,7 +82,9 @@ static int __report_module(struct addr_location *al, u64 ip,
* Some callers will use al->sym, so we can't just use the
* cheaper thread__find_map() here.
*/
- thread__find_symbol(ui->thread, PERF_RECORD_MISC_USER, ip, al);
+ thread__find_symbol(ui->thread,
+ &(struct perf_sample){.cpumode = PERF_RECORD_MISC_USER, .ip = ip},
+ al);
if (al->map)
dso = map__dso(al->map);
@@ -208,13 +210,14 @@ static bool get_thread(Dwfl *dwfl __maybe_unused, pid_t tid, void *arg,
static int access_dso_mem(struct unwind_info *ui, Dwarf_Addr addr,
Dwarf_Word *data, size_t len)
{
+ struct perf_sample sample = { .cpumode = PERF_RECORD_MISC_USER, .ip = addr };
struct addr_location al;
union u64_swap u;
ssize_t size;
struct dso *dso;
addr_location__init(&al);
- if (!thread__find_map(ui->thread, PERF_RECORD_MISC_USER, addr, &al)) {
+ if (!thread__find_map(ui->thread, &sample, &al)) {
pr_debug("unwind: no map for %lx\n", (unsigned long)addr);
goto out_fail;
}
diff --git a/tools/perf/util/unwind-libunwind.c b/tools/perf/util/unwind-libunwind.c
index 6052383e862b..83f043f266da 100644
--- a/tools/perf/util/unwind-libunwind.c
+++ b/tools/perf/util/unwind-libunwind.c
@@ -370,7 +370,9 @@ static struct map *find_map(uint64_t ip, struct unwind_info *ui)
struct map *ret;
addr_location__init(&al);
- thread__find_map(ui->thread, PERF_RECORD_MISC_USER, ip, &al);
+ thread__find_map(ui->thread,
+ &(struct perf_sample){.cpumode = PERF_RECORD_MISC_USER, .ip = ip},
+ &al);
ret = map__get(al.map);
addr_location__exit(&al);
return ret;
@@ -626,7 +628,9 @@ static int entry(uint64_t ip, struct thread *thread, unwind_entry_cb_t cb, void
int ret;
addr_location__init(&al);
- e.ms.sym = thread__find_symbol(thread, PERF_RECORD_MISC_USER, ip, &al);
+ e.ms.sym = thread__find_symbol(thread,
+ &(struct perf_sample){.cpumode = PERF_RECORD_MISC_USER,
+ .ip = ip}, &al);
e.ip = ip;
e.ms.map = al.map;
e.ms.thread = thread__get(al.thread);
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* [PATCH v2 5/8] perf tools: Internal support for BUILD_ID_OFFSET
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
` (3 preceding siblings ...)
2026-10-02 17:38 ` [PATCH v2 4/8] perf: Refactor thread map and symbol APIs to take perf_sample Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 6/8] perf inject: Extend perf inject to support bid_offset conversion Ian Rogers
` (2 subsequent siblings)
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Implement user-space parsing for the new sample types in evsel,
update perf_event_attr_fprintf to display the new bits
in 'perf report -D', and update synthetic event parameter names.
Signed-off-by: Ian Rogers <irogers@google.com>
---
tools/perf/builtin-report.c | 6 +-
tools/perf/builtin-script.c | 16 +--
tools/perf/util/callchain.c | 93 ++++++++++++-----
tools/perf/util/dsos.c | 27 +++++
tools/perf/util/dsos.h | 2 +
tools/perf/util/event.c | 35 +++++--
tools/perf/util/evsel.c | 114 ++++++++++++++++++++-
tools/perf/util/evsel.h | 3 +-
tools/perf/util/evsel_fprintf.c | 9 +-
tools/perf/util/machine.c | 119 +++++++++++++++++-----
tools/perf/util/maps.c | 59 +++++++++++
tools/perf/util/maps.h | 2 +
tools/perf/util/perf_event_attr_fprintf.c | 3 +-
tools/perf/util/sample.c | 9 ++
tools/perf/util/sample.h | 28 +++++
tools/perf/util/session.c | 52 +++++++---
tools/perf/util/synthetic-events.c | 28 +++++
17 files changed, 515 insertions(+), 90 deletions(-)
diff --git a/tools/perf/builtin-report.c b/tools/perf/builtin-report.c
index 57225bc87731..0c8596cf82b5 100644
--- a/tools/perf/builtin-report.c
+++ b/tools/perf/builtin-report.c
@@ -382,7 +382,8 @@ static int report__setup_sample_type(struct report *rep)
session->itrace_synth_opts->add_last_branch)
sample_type |= PERF_SAMPLE_BRANCH_STACK;
- if (!is_pipe && !(sample_type & PERF_SAMPLE_CALLCHAIN)) {
+ if (!is_pipe && !(sample_type & (PERF_SAMPLE_CALLCHAIN |
+ PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET))) {
if (perf_hpp_list.parent) {
ui__error("Selected --sort parent, but no "
"callchain data. Did you call "
@@ -408,7 +409,8 @@ static int report__setup_sample_type(struct report *rep)
if (symbol_conf.cumulate_callchain) {
/* Silently ignore if callchain is missing */
- if (!(sample_type & PERF_SAMPLE_CALLCHAIN)) {
+ if (!(sample_type & (PERF_SAMPLE_CALLCHAIN |
+ PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET))) {
symbol_conf.cumulate_callchain = false;
perf_hpp__cancel_cumulate(session->evlist);
}
diff --git a/tools/perf/builtin-script.c b/tools/perf/builtin-script.c
index 99e9e0c60f14..0c93959574e5 100644
--- a/tools/perf/builtin-script.c
+++ b/tools/perf/builtin-script.c
@@ -498,7 +498,8 @@ static int evsel__check_attr(struct evsel *evsel, struct perf_session *session)
return -EINVAL;
if (PRINT_FIELD(IP)) {
- if (evsel__check_stype(evsel, PERF_SAMPLE_IP, "IP", PERF_OUTPUT_IP))
+ if (evsel__check_stype(evsel, PERF_SAMPLE_IP | PERF_SAMPLE_BUILD_ID_OFFSET,
+ "IP", PERF_OUTPUT_IP))
return -EINVAL;
}
@@ -515,7 +516,8 @@ static int evsel__check_attr(struct evsel *evsel, struct perf_session *session)
return -EINVAL;
if (PRINT_FIELD(SYM) &&
- !(evsel->core.attr.sample_type & (PERF_SAMPLE_IP|PERF_SAMPLE_ADDR))) {
+ !(evsel->core.attr.sample_type & (PERF_SAMPLE_IP | PERF_SAMPLE_ADDR |
+ PERF_SAMPLE_BUILD_ID_OFFSET))) {
pr_err("Display of symbols requested but neither sample IP nor "
"sample address\navailable. Hence, no addresses to convert "
"to symbols.\n");
@@ -527,7 +529,8 @@ static int evsel__check_attr(struct evsel *evsel, struct perf_session *session)
return -EINVAL;
}
if (PRINT_FIELD(DSO) &&
- !(evsel->core.attr.sample_type & (PERF_SAMPLE_IP|PERF_SAMPLE_ADDR))) {
+ !(evsel->core.attr.sample_type & (PERF_SAMPLE_IP | PERF_SAMPLE_ADDR |
+ PERF_SAMPLE_BUILD_ID_OFFSET))) {
pr_err("Display of DSO requested but no address to convert.\n");
return -EINVAL;
}
@@ -1788,7 +1791,7 @@ static int perf_sample__fprintf_bts(struct perf_sample *sample,
unsigned int print_opts = output[type].print_ip_opts;
struct callchain_cursor *cursor = NULL;
- if (symbol_conf.use_callchain && sample->callchain) {
+ if (symbol_conf.use_callchain && (sample->callchain || sample->callchain_bids)) {
cursor = get_tls_callchain_cursor();
if (thread__resolve_callchain(al->thread, cursor,
sample, NULL, NULL,
@@ -2753,7 +2756,7 @@ static void process_event(struct perf_script *script,
if (script->stitch_lbr)
thread__set_lbr_stitch_enable(al->thread, true);
- if (symbol_conf.use_callchain && sample->callchain) {
+ if (symbol_conf.use_callchain && (sample->callchain || sample->callchain_bids)) {
cursor = get_tls_callchain_cursor();
if (thread__resolve_callchain(al->thread, cursor,
sample, NULL, NULL,
@@ -3000,7 +3003,7 @@ static int process_deferred_sample_event(const struct perf_tool *tool,
if (PRINT_FIELD(IP)) {
struct callchain_cursor *cursor = NULL;
- if (symbol_conf.use_callchain && sample->callchain) {
+ if (symbol_conf.use_callchain && (sample->callchain || sample->callchain_bids)) {
cursor = get_tls_callchain_cursor();
if (thread__resolve_callchain(al.thread, cursor,
sample, NULL, NULL,
@@ -3082,6 +3085,7 @@ static int process_attr(const struct perf_tool *tool, union perf_event *event,
/* Enable fields for callchain entries */
if (symbol_conf.use_callchain &&
(sample_type & PERF_SAMPLE_CALLCHAIN ||
+ sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET ||
sample_type & PERF_SAMPLE_BRANCH_STACK ||
(sample_type & PERF_SAMPLE_REGS_USER &&
sample_type & PERF_SAMPLE_STACK_USER))) {
diff --git a/tools/perf/util/callchain.c b/tools/perf/util/callchain.c
index 31c675cbab63..24a229ee2ada 100644
--- a/tools/perf/util/callchain.c
+++ b/tools/perf/util/callchain.c
@@ -1173,7 +1173,8 @@ int sample__resolve_callchain(struct perf_sample *sample,
struct addr_location *al,
int max_stack)
{
- if (sample->callchain == NULL && !symbol_conf.show_branchflag_count)
+ if (sample->callchain == NULL && sample->callchain_bids == NULL &&
+ !symbol_conf.show_branchflag_count)
return 0;
if (symbol_conf.use_callchain || symbol_conf.cumulate_callchain ||
@@ -1186,7 +1187,8 @@ int sample__resolve_callchain(struct perf_sample *sample,
int hist_entry__append_callchain(struct hist_entry *he, struct perf_sample *sample)
{
- if ((!symbol_conf.use_callchain || sample->callchain == NULL) &&
+ if ((!symbol_conf.use_callchain ||
+ (sample->callchain == NULL && sample->callchain_bids == NULL)) &&
!symbol_conf.show_branchflag_count)
return 0;
return callchain_append(he->callchain, get_tls_callchain_cursor(), sample->period);
@@ -1912,32 +1914,75 @@ int sample__for_each_callchain_node(struct thread *thread,
int sample__merge_deferred_callchain(struct perf_sample *sample_orig,
struct perf_sample *sample_callchain)
{
- u64 nr_orig = sample_orig->callchain->nr - 1;
- u64 nr_deferred = sample_callchain->callchain->nr;
- struct ip_callchain *callchain;
-
- if (sample_orig->merged_callchain) {
- /* Already merged. */
- return -EINVAL;
+ if (sample_orig->deferred_bid && sample_callchain->callchain_bids_nr > 0) {
+ const struct perf_sample_build_id_offset *entry =
+ &sample_callchain->callchain_bids[0];
+
+ memcpy(sample_orig->bid.deferred_storage, &entry->bid,
+ sizeof(struct perf_build_id));
+ sample_orig->bid.bid =
+ (struct perf_build_id *)sample_orig->bid.deferred_storage;
+ sample_orig->bid.offset = entry->offset;
+ if (!sample_orig->evsel ||
+ !(sample_orig->evsel->core.attr.sample_type & PERF_SAMPLE_IP))
+ sample_orig->ip = entry->offset;
+ sample_orig->deferred_bid = false;
+ }
+
+ if (sample_orig->callchain_bids_nr >= 2 &&
+ sample_callchain->callchain_bids_nr > 0) {
+ u64 nr_orig = sample_orig->callchain_bids_nr - 1;
+ u64 nr_deferred = sample_callchain->callchain_bids_nr;
+
+ if (sample_orig->callchain_bids[nr_orig - 1].offset ==
+ PERF_CONTEXT_USER_DEFERRED) {
+ struct perf_sample_build_id_offset *bids =
+ calloc(nr_orig + nr_deferred, sizeof(*bids));
+
+ if (bids == NULL)
+ return -ENOMEM;
+
+ memcpy(bids, sample_orig->callchain_bids,
+ nr_orig * sizeof(*bids));
+ memcpy(&bids[nr_orig], sample_callchain->callchain_bids,
+ nr_deferred * sizeof(*bids));
+ if (sample_orig->merged_callchain_bids)
+ free(sample_orig->callchain_bids);
+ sample_orig->callchain_bids = bids;
+ sample_orig->callchain_bids_nr = nr_orig + nr_deferred;
+ sample_orig->merged_callchain_bids = true;
+ }
}
- if (sample_orig->callchain->nr < 2) {
- sample_orig->deferred_callchain = false;
- return -EINVAL;
- }
+ if (sample_orig->callchain && sample_callchain->callchain) {
+ u64 nr_orig = sample_orig->callchain->nr - 1;
+ u64 nr_deferred = sample_callchain->callchain->nr;
+ struct ip_callchain *callchain;
- callchain = calloc(1 + nr_orig + nr_deferred, sizeof(u64));
- if (callchain == NULL)
- return -ENOMEM;
+ if (sample_orig->merged_callchain) {
+ /* Already merged. */
+ return -EINVAL;
+ }
- callchain->nr = nr_orig + nr_deferred;
- /* copy original including PERF_CONTEXT_USER_DEFERRED (but the cookie) */
- memcpy(callchain->ips, sample_orig->callchain->ips, nr_orig * sizeof(u64));
- /* copy deferred user callchains */
- memcpy(&callchain->ips[nr_orig], sample_callchain->callchain->ips,
- nr_deferred * sizeof(u64));
+ if (sample_orig->callchain->nr < 2) {
+ sample_orig->deferred_callchain = false;
+ return -EINVAL;
+ }
+
+ callchain = calloc(1 + nr_orig + nr_deferred, sizeof(u64));
+ if (callchain == NULL)
+ return -ENOMEM;
+
+ callchain->nr = nr_orig + nr_deferred;
+ /* copy original including PERF_CONTEXT_USER_DEFERRED (but the cookie) */
+ memcpy(callchain->ips, sample_orig->callchain->ips, nr_orig * sizeof(u64));
+ /* copy deferred user callchains */
+ memcpy(&callchain->ips[nr_orig], sample_callchain->callchain->ips,
+ nr_deferred * sizeof(u64));
+
+ sample_orig->merged_callchain = true;
+ sample_orig->callchain = callchain;
+ }
- sample_orig->merged_callchain = true;
- sample_orig->callchain = callchain;
return 0;
}
diff --git a/tools/perf/util/dsos.c b/tools/perf/util/dsos.c
index e927e707abac..ae47966ff6af 100644
--- a/tools/perf/util/dsos.c
+++ b/tools/perf/util/dsos.c
@@ -294,6 +294,33 @@ struct dso *dsos__find(struct dsos *dsos, const char *name, bool cmp_short)
return res;
}
+struct dsos__find_build_id_cb_args {
+ const struct build_id *bid;
+ struct dso *res;
+};
+
+static int dsos__find_build_id_cb(struct dso *dso, void *data)
+{
+ struct dsos__find_build_id_cb_args *args = data;
+
+ if (dso__build_id_equal(dso, args->bid)) {
+ args->res = dso__get(dso);
+ return 1;
+ }
+ return 0;
+}
+
+struct dso *dsos__find_by_build_id(struct dsos *dsos, const struct build_id *bid)
+{
+ struct dsos__find_build_id_cb_args args = {
+ .bid = bid,
+ .res = NULL,
+ };
+
+ dsos__for_each_dso(dsos, dsos__find_build_id_cb, &args);
+ return args.res;
+}
+
static void dso__set_basename(struct dso *dso)
{
bool allocated = false;
diff --git a/tools/perf/util/dsos.h b/tools/perf/util/dsos.h
index a26774950866..1b9d71c5857e 100644
--- a/tools/perf/util/dsos.h
+++ b/tools/perf/util/dsos.h
@@ -8,6 +8,7 @@
#include <linux/rbtree.h>
#include "rwsem.h"
+struct build_id;
struct dso;
struct dso_id;
struct kmod_path;
@@ -31,6 +32,7 @@ void dsos__exit(struct dsos *dsos);
int __dsos__add(struct dsos *dsos, struct dso *dso);
int dsos__add(struct dsos *dsos, struct dso *dso);
struct dso *dsos__find(struct dsos *dsos, const char *name, bool cmp_short);
+struct dso *dsos__find_by_build_id(struct dsos *dsos, const struct build_id *bid);
struct dso *dsos__findnew_id(struct dsos *dsos, const char *name, const struct dso_id *id);
diff --git a/tools/perf/util/event.c b/tools/perf/util/event.c
index cd13a0c3cb52..0483d0b25d19 100644
--- a/tools/perf/util/event.c
+++ b/tools/perf/util/event.c
@@ -702,7 +702,6 @@ struct map *thread__find_map(struct thread *thread, struct perf_sample *sample,
thread__zput(al->thread);
al->thread = thread__get(thread);
- al->addr = sample->ip;
al->cpumode = sample->cpumode;
al->filtered = 0;
@@ -735,15 +734,31 @@ struct map *thread__find_map(struct thread *thread, struct perf_sample *sample,
return NULL;
}
- al->map = maps__find(maps, al->addr);
- if (al->map != NULL) {
- /*
- * Kernel maps might be changed when loading symbols so loading
- * must be done prior to using kernel maps.
- */
- if (load_map)
- map__load(al->map);
- al->addr = map__map_ip(al->map, al->addr);
+
+ if (sample->bid.bid && sample->bid.bid->size > 0) {
+ struct build_id bid;
+
+ build_id__init(&bid, sample->bid.bid->data, sample->bid.bid->size);
+ al->addr = sample->bid.offset;
+ al->map = maps__find_by_build_id(maps, &bid);
+ if (al->map != NULL) {
+ if (load_map)
+ map__load(al->map);
+ /* al->addr is already a file offset */
+ }
+ } else {
+ /* When bid.bid->size == 0, bid.ip holds the fallback virtual address. */
+ al->addr = sample->bid.bid ? sample->bid.ip : sample->ip;
+ al->map = maps__find(maps, al->addr);
+ if (al->map != NULL) {
+ /*
+ * Kernel maps might be changed when loading symbols so loading
+ * must be done prior to using kernel maps.
+ */
+ if (load_map)
+ map__load(al->map);
+ al->addr = map__map_ip(al->map, al->addr);
+ }
}
return al->map;
diff --git a/tools/perf/util/evsel.c b/tools/perf/util/evsel.c
index 9c5e7510f0c0..08e9e6567bed 100644
--- a/tools/perf/util/evsel.c
+++ b/tools/perf/util/evsel.c
@@ -3423,6 +3423,21 @@ static int __set_offcpu_sample(struct perf_sample *data)
return -EFAULT;
}
+/*
+ * Before __evsel__parse_sample() is called on a cross-endian perf.data file,
+ * perf_event__all64_swap() has already byte-swapped the entire PERF_RECORD_SAMPLE
+ * payload in 64-bit chunks. That leaves 64-bit fields (like offset, ip, cookie)
+ * in host endianness, but scrambles the 24-byte struct perf_build_id (which
+ * contains a 16-bit reserved field, a 1-byte size, and a 20-byte byte array).
+ * Undo the 64-bit chunk swap on struct perf_build_id and byte-swap its 16-bit
+ * field.
+ */
+static void perf_build_id__undo_bswap_64(struct perf_build_id *bid)
+{
+ mem_bswap_64(bid, sizeof(*bid));
+ bid->__reserved_2 = bswap_16(bid->__reserved_2);
+}
+
int __evsel__parse_sample(struct evsel *evsel, union perf_event *event,
struct perf_sample *data, bool needs_swap)
{
@@ -3450,11 +3465,23 @@ int __evsel__parse_sample(struct evsel *evsel, union perf_event *event,
data->vcpu = -1;
if (event->header.type == PERF_RECORD_CALLCHAIN_DEFERRED) {
- const u64 max_callchain_nr = UINT64_MAX / sizeof(u64);
+ if (event->header.misc & PERF_RECORD_MISC_MMAP_BUILD_ID) {
+ const size_t entry_sz = sizeof(struct perf_sample_build_id_offset);
+ const u64 max_nr = UINT64_MAX / entry_sz;
+ u64 nr = event->callchain_deferred.nr;
- data->callchain = (struct ip_callchain *)&event->callchain_deferred.nr;
- if (data->callchain->nr > max_callchain_nr)
- goto out_efault;
+ if (nr > max_nr)
+ goto out_efault;
+ OVERFLOW_CHECK(event->callchain_deferred.bids, nr * entry_sz, max_size);
+ data->callchain_bids = event->callchain_deferred.bids;
+ data->callchain_bids_nr = nr;
+ } else {
+ const u64 max_callchain_nr = UINT64_MAX / sizeof(u64);
+
+ data->callchain = (struct ip_callchain *)&event->callchain_deferred.nr;
+ if (data->callchain->nr > max_callchain_nr)
+ goto out_efault;
+ }
data->deferred_cookie = event->callchain_deferred.cookie;
@@ -3885,6 +3912,85 @@ int __evsel__parse_sample(struct evsel *evsel, union perf_event *event,
array = (void *)array + sz;
}
+ if (type & PERF_SAMPLE_BUILD_ID_OFFSET) {
+ const struct perf_sample_build_id_offset *entry =
+ (const struct perf_sample_build_id_offset *)array;
+
+ OVERFLOW_CHECK(array, sizeof(*entry), max_size);
+ if (evsel->core.attr.defer_callchain && entry->zero == 0 &&
+ entry->context == PERF_CONTEXT_USER_DEFERRED) {
+ data->deferred_bid = true;
+ data->deferred_callchain = true;
+ data->deferred_cookie = entry->cookie;
+ data->bid.bid = (struct perf_build_id *)&entry->bid;
+ data->bid.offset = entry->ip;
+ } else {
+ if (swapped) {
+ struct perf_build_id *bid =
+ (void *)data->bid.deferred_storage;
+
+ /*
+ * perf_event__all64_swap() already swapped the
+ * sample in 64-bit chunks, so entry->offset is
+ * already in host endianness; undo the 64-bit
+ * swap on the build ID bytes.
+ */
+ memcpy(bid, &entry->bid, sizeof(*bid));
+ perf_build_id__undo_bswap_64(bid);
+ data->bid.bid = bid;
+ } else {
+ data->bid.bid = (struct perf_build_id *)&entry->bid;
+ }
+ data->bid.offset = entry->offset;
+ }
+ if (!(type & PERF_SAMPLE_IP))
+ data->ip = data->bid.offset;
+ array += sizeof(*entry) / sizeof(u64);
+ } else {
+ data->bid.bid = NULL;
+ data->bid.offset = 0;
+ }
+
+ if (type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ const size_t entry_sz = sizeof(struct perf_sample_build_id_offset);
+ const u64 max_callchain_nr = UINT64_MAX / entry_sz;
+
+ OVERFLOW_CHECK_u64(array);
+ sz = *array++;
+ if (sz > max_callchain_nr)
+ goto out_efault;
+
+ OVERFLOW_CHECK(array, sz * entry_sz, max_size);
+ if (evsel->core.attr.defer_callchain && sz >= 2 &&
+ array[(sz - 2) * 4 + 3] == PERF_CONTEXT_USER_DEFERRED) {
+ data->deferred_cookie = array[(sz - 1) * 4 + 3];
+ data->deferred_callchain = true;
+ }
+ if (swapped && sz > 0) {
+ struct perf_sample_build_id_offset *bids = calloc(sz, entry_sz);
+
+ if (!bids)
+ goto out_efault;
+ /*
+ * perf_event__all64_swap() already swapped each 64-bit
+ * word (including bids[i].offset); undo the 64-bit swap
+ * on each build ID.
+ */
+ memcpy(bids, array, sz * entry_sz);
+ for (u64 i = 0; i < sz; i++)
+ perf_build_id__undo_bswap_64(&bids[i].bid);
+ data->callchain_bids = bids;
+ data->merged_callchain_bids = true;
+ } else {
+ data->callchain_bids = (struct perf_sample_build_id_offset *)array;
+ }
+ data->callchain_bids_nr = sz;
+ array += sz * entry_sz / sizeof(u64);
+ } else {
+ data->callchain_bids = NULL;
+ data->callchain_bids_nr = 0;
+ }
+
if (evsel__is_offcpu_event(evsel)) {
if (__set_offcpu_sample(data))
goto out_efault;
diff --git a/tools/perf/util/evsel.h b/tools/perf/util/evsel.h
index 174f3414fd3c..d4aa0191a1fd 100644
--- a/tools/perf/util/evsel.h
+++ b/tools/perf/util/evsel.h
@@ -569,7 +569,8 @@ static inline bool evsel__has_callchain(const struct evsel *evsel)
* For reporting purposes, an evsel sample can have a recorded callchain
* or a callchain synthesized from AUX area data.
*/
- return evsel->core.attr.sample_type & PERF_SAMPLE_CALLCHAIN ||
+ return evsel->core.attr.sample_type & (PERF_SAMPLE_CALLCHAIN |
+ PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) ||
evsel->synth_sample_type & PERF_SAMPLE_CALLCHAIN;
}
diff --git a/tools/perf/util/evsel_fprintf.c b/tools/perf/util/evsel_fprintf.c
index 0f7a25500a44..e91411a1ac18 100644
--- a/tools/perf/util/evsel_fprintf.c
+++ b/tools/perf/util/evsel_fprintf.c
@@ -131,7 +131,7 @@ int sample__fprintf_callchain(struct perf_sample *sample, int left_alignment,
if (cursor == NULL)
return fprintf(fp, "<not enough memory for the callchain cursor>%s", print_oneline ? "" : "\n");
- if (sample->callchain) {
+ if (sample->callchain || sample->callchain_bids) {
callchain_cursor_commit(cursor);
while (1) {
@@ -232,8 +232,11 @@ int sample__fprintf_sym(struct perf_sample *sample, struct addr_location *al,
} else {
printed += fprintf(fp, "%-*.*s", left_alignment, left_alignment, " ");
- if (print_ip)
- printed += fprintf(fp, "%16" PRIx64, sample->ip);
+ if (print_ip) {
+ u64 ip = sample->bid.bid ? sample->bid.offset : sample->ip;
+
+ printed += fprintf(fp, "%16" PRIx64, ip);
+ }
if (print_sym) {
printed += fprintf(fp, " ");
diff --git a/tools/perf/util/machine.c b/tools/perf/util/machine.c
index 1a13253d0c7f..99dc47e95938 100644
--- a/tools/perf/util/machine.c
+++ b/tools/perf/util/machine.c
@@ -2221,7 +2221,9 @@ static int add_callchain_ip(struct thread *thread,
struct branch_flags *flags,
struct iterations *iter,
u64 branch_from,
- bool symbols)
+ bool symbols,
+ struct perf_build_id *bid,
+ u64 offset)
{
static __thread struct perf_sample sample;
struct map_symbol ms = {};
@@ -2235,6 +2237,8 @@ static int add_callchain_ip(struct thread *thread,
al.sym = NULL;
al.srcline = NULL;
sample.ip = ip;
+ sample.bid.bid = bid;
+ sample.bid.offset = offset;
if (!cpumode) {
thread__find_cpumode_addr_location(thread, &sample, symbols, &al);
} else {
@@ -2295,6 +2299,9 @@ static int add_callchain_ip(struct thread *thread,
ms.map = map__get(al.map);
ms.sym = al.sym;
+ if (bid && bid->size > 0 && al.map)
+ ip = map__unmap_ip(al.map, al.addr);
+
if (append_inlines(cursor, &ms, ip, branch, flags, nr_loop_iter,
iter_cycles, branch_from) == 0)
goto out;
@@ -2410,10 +2417,17 @@ static int lbr_callchain_add_kernel_ip(struct thread *thread,
if (callee) {
for (i = 0; i < end + 1; i++) {
+ struct perf_build_id *bid = NULL;
+ u64 offset = 0;
+
+ if (sample->callchain_bids && i < (int)sample->callchain_bids_nr) {
+ bid = &sample->callchain_bids[i].bid;
+ offset = sample->callchain_bids[i].offset;
+ }
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, chain->ips[i],
false, NULL, NULL, branch_from,
- symbols);
+ symbols, bid, offset);
if (err)
return err;
}
@@ -2421,10 +2435,17 @@ static int lbr_callchain_add_kernel_ip(struct thread *thread,
}
for (i = end; i >= 0; i--) {
+ struct perf_build_id *bid = NULL;
+ u64 offset = 0;
+
+ if (sample->callchain_bids && i < (int)sample->callchain_bids_nr) {
+ bid = &sample->callchain_bids[i].bid;
+ offset = sample->callchain_bids[i].offset;
+ }
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, chain->ips[i],
false, NULL, NULL, branch_from,
- symbols);
+ symbols, bid, offset);
if (err)
return err;
}
@@ -2507,7 +2528,7 @@ static int lbr_callchain_add_lbr_ip(struct thread *thread,
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, ip,
true, flags, NULL,
- *branch_from, symbols);
+ *branch_from, symbols, /*bid=*/NULL, /*offset=*/0);
if (err)
return err;
@@ -2532,7 +2553,7 @@ static int lbr_callchain_add_lbr_ip(struct thread *thread,
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, ip,
true, flags, NULL,
- *branch_from, symbols);
+ *branch_from, symbols, /*bid=*/NULL, /*offset=*/0);
if (err)
return err;
save_lbr_cursor_node(thread, cursor, i);
@@ -2547,7 +2568,7 @@ static int lbr_callchain_add_lbr_ip(struct thread *thread,
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, ip,
true, flags, NULL,
- *branch_from, symbols);
+ *branch_from, symbols, /*bid=*/NULL, /*offset=*/0);
if (err)
return err;
save_lbr_cursor_node(thread, cursor, i);
@@ -2567,7 +2588,7 @@ static int lbr_callchain_add_lbr_ip(struct thread *thread,
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, ip,
true, flags, NULL,
- *branch_from, symbols);
+ *branch_from, symbols, /*bid=*/NULL, /*offset=*/0);
if (err)
return err;
}
@@ -2740,12 +2761,16 @@ static int resolve_lbr_callchain_sample(struct thread *thread,
{
bool callee = (callchain_param.order == ORDER_CALLEE);
struct ip_callchain *chain = sample->callchain;
- int chain_nr = min(max_stack, (int)chain->nr), i;
+ int chain_nr, i;
struct lbr_stitch *lbr_stitch;
bool stitched_lbr = false;
u64 branch_from = 0;
int err;
+ if (!chain)
+ return 0;
+
+ chain_nr = min(max_stack, (int)chain->nr);
for (i = 0; i < chain_nr; i++) {
if (chain->ips[i] == PERF_CONTEXT_USER)
break;
@@ -2831,7 +2856,8 @@ static int find_prev_cpumode(struct ip_callchain *chain, struct thread *thread,
if (ip >= PERF_CONTEXT_MAX) {
err = add_callchain_ip(thread, cursor, parent,
root_al, cpumode, ip,
- false, NULL, NULL, 0, symbols);
+ false, NULL, NULL, 0, symbols,
+ /*bid=*/NULL, /*offset=*/0);
break;
}
}
@@ -2859,6 +2885,7 @@ static int thread__resolve_callchain_sample(struct thread *thread,
struct branch_stack *branch = sample->branch_stack;
struct branch_entry *entries = perf_sample__branch_entries(sample);
struct ip_callchain *chain = sample->callchain;
+ struct ip_callchain *alloc_chain = NULL;
int chain_nr = 0;
u8 cpumode = PERF_RECORD_MISC_USER;
int i, j, err, nr_entries, usr_idx;
@@ -2866,6 +2893,17 @@ static int thread__resolve_callchain_sample(struct thread *thread,
int first_call = 0;
u64 leaf_frame_caller;
+ if (!chain && sample->callchain_bids_nr > 0) {
+ alloc_chain = malloc(sizeof(*chain) + sample->callchain_bids_nr * sizeof(u64));
+ if (!alloc_chain)
+ return -ENOMEM;
+ alloc_chain->nr = sample->callchain_bids_nr;
+ for (i = 0; i < (int)sample->callchain_bids_nr; i++)
+ alloc_chain->ips[i] = sample->callchain_bids[i].offset;
+ chain = alloc_chain;
+ sample->callchain = alloc_chain;
+ }
+
if (chain)
chain_nr = chain->nr;
@@ -2876,8 +2914,10 @@ static int thread__resolve_callchain_sample(struct thread *thread,
root_al, max_stack,
!env ? 0 : env->max_branches,
symbols);
- if (err)
- return (err < 0) ? err : 0;
+ if (err) {
+ err = (err < 0) ? err : 0;
+ goto out;
+ }
}
/*
@@ -2940,22 +2980,26 @@ static int thread__resolve_callchain_sample(struct thread *thread,
root_al,
NULL, be[i].to,
true, &be[i].flags,
- NULL, be[i].from, symbols);
+ NULL, be[i].from, symbols,
+ /*bid=*/NULL, /*offset=*/0);
if (!err) {
err = add_callchain_ip(thread, cursor, parent, root_al,
NULL, be[i].from,
true, &be[i].flags,
- &iter[i], 0, symbols);
+ &iter[i], 0, symbols,
+ /*bid=*/NULL, /*offset=*/0);
}
if (err == -EINVAL)
break;
if (err)
- return err;
+ goto out;
}
- if (chain_nr == 0)
- return 0;
+ if (chain_nr == 0) {
+ err = 0;
+ goto out;
+ }
chain_nr -= nr;
}
@@ -2964,12 +3008,16 @@ static int thread__resolve_callchain_sample(struct thread *thread,
if (chain && callchain_param.order != ORDER_CALLEE) {
err = find_prev_cpumode(chain, thread, cursor, parent, root_al,
&cpumode, chain->nr - first_call, symbols);
- if (err)
- return (err < 0) ? err : 0;
+ if (err) {
+ err = (err < 0) ? err : 0;
+ goto out;
+ }
}
for (i = first_call, nr_entries = 0;
i < chain_nr && nr_entries < max_stack; i++) {
u64 ip;
+ struct perf_build_id *bid = NULL;
+ u64 offset = 0;
if (callchain_param.order == ORDER_CALLEE)
j = i;
@@ -2981,13 +3029,19 @@ static int thread__resolve_callchain_sample(struct thread *thread,
continue;
#endif
ip = chain->ips[j];
+ if (sample->callchain_bids && j < (int)sample->callchain_bids_nr) {
+ bid = &sample->callchain_bids[j].bid;
+ offset = sample->callchain_bids[j].offset;
+ }
if (ip < PERF_CONTEXT_MAX)
++nr_entries;
else if (callchain_param.order != ORDER_CALLEE) {
err = find_prev_cpumode(chain, thread, cursor, parent,
root_al, &cpumode, j, symbols);
- if (err)
- return (err < 0) ? err : 0;
+ if (err) {
+ err = (err < 0) ? err : 0;
+ goto out;
+ }
continue;
}
@@ -3013,21 +3067,32 @@ static int thread__resolve_callchain_sample(struct thread *thread,
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, leaf_frame_caller,
- false, NULL, NULL, 0, symbols);
- if (err)
- return (err < 0) ? err : 0;
+ false, NULL, NULL, 0, symbols,
+ /*bid=*/NULL, /*offset=*/0);
+ if (err) {
+ err = (err < 0) ? err : 0;
+ goto out;
+ }
}
}
err = add_callchain_ip(thread, cursor, parent,
root_al, &cpumode, ip,
- false, NULL, NULL, 0, symbols);
+ false, NULL, NULL, 0, symbols, bid, offset);
- if (err)
- return (err < 0) ? err : 0;
+ if (err) {
+ err = (err < 0) ? err : 0;
+ goto out;
+ }
}
- return 0;
+ err = 0;
+out:
+ if (alloc_chain) {
+ sample->callchain = NULL;
+ free(alloc_chain);
+ }
+ return err;
}
static int unwind_entry(struct unwind_entry *entry, void *arg)
diff --git a/tools/perf/util/maps.c b/tools/perf/util/maps.c
index f808df2fe77b..1e83332ea00a 100644
--- a/tools/perf/util/maps.c
+++ b/tools/perf/util/maps.c
@@ -2,8 +2,11 @@
#include <errno.h>
#include <stdlib.h>
#include <linux/zalloc.h>
+#include "build-id.h"
#include "debug.h"
#include "dso.h"
+#include "dsos.h"
+#include "machine.h"
#include "map.h"
#include "maps.h"
#include "rwsem.h"
@@ -1251,6 +1254,62 @@ static int map__strcmp_name(const void *name, const void *b)
return strcmp(name, dso__short_name(dso));
}
+struct map *maps__find_by_build_id(struct maps *maps, struct build_id *bid)
+{
+ struct map *result = NULL;
+
+ down_read(maps__lock(maps));
+ for (unsigned int i = 0, n = maps__nr_maps(maps); i < n; i++) {
+ struct map *map = maps__maps_by_address(maps)[i];
+
+ if (dso__build_id_equal(map__dso(map), bid)) {
+ result = map__get(map);
+ break;
+ }
+ }
+ up_read(maps__lock(maps));
+
+ if (!result && maps__machine(maps)) {
+ struct machine *machine = maps__machine(maps);
+ struct dso *dso = dsos__find_by_build_id(&machine->dsos, bid);
+
+ if (!dso) {
+ char sbuild_id[SBUILD_ID_SIZE];
+ char *origname;
+
+ build_id__snprintf(bid, sbuild_id, sizeof(sbuild_id));
+ origname = build_id_cache__origname(sbuild_id);
+ if (origname) {
+ struct dso_id id = { .build_id = *bid };
+
+ dso = dsos__findnew_id(&machine->dsos, origname, &id);
+ free(origname);
+ }
+ }
+ if (dso) {
+ down_write(maps__lock(maps));
+ for (unsigned int i = 0, n = maps__nr_maps(maps); i < n; i++) {
+ struct map *map = maps__maps_by_address(maps)[i];
+
+ if (dso__build_id_equal(map__dso(map), bid)) {
+ result = map__get(map);
+ break;
+ }
+ }
+ if (!result) {
+ result = map__new2(0, dso);
+ if (result && __maps__insert(maps, result)) {
+ map__put(result);
+ result = NULL;
+ }
+ }
+ up_write(maps__lock(maps));
+ dso__put(dso);
+ }
+ }
+ return result;
+}
+
struct map *maps__find_by_name(struct maps *maps, const char *name)
{
struct map *result = NULL;
diff --git a/tools/perf/util/maps.h b/tools/perf/util/maps.h
index 4ec9b7453a3b..6f1791457068 100644
--- a/tools/perf/util/maps.h
+++ b/tools/perf/util/maps.h
@@ -11,6 +11,7 @@ struct ref_reloc_sym;
struct machine;
struct map;
struct maps;
+struct build_id;
#define KMAP_NAME_LEN 256
@@ -76,6 +77,7 @@ int maps__find_ams(struct maps *maps, struct addr_map_symbol *ams);
int maps__fixup_overlap_and_insert(struct maps *maps, struct map *new);
struct map *maps__find_by_name(struct maps *maps, const char *name);
+struct map *maps__find_by_build_id(struct maps *maps, struct build_id *bid);
struct map *maps__find_next_entry(struct maps *maps, struct map *map);
diff --git a/tools/perf/util/perf_event_attr_fprintf.c b/tools/perf/util/perf_event_attr_fprintf.c
index 2e8be9a357dc..dd653da923c4 100644
--- a/tools/perf/util/perf_event_attr_fprintf.c
+++ b/tools/perf/util/perf_event_attr_fprintf.c
@@ -40,7 +40,8 @@ static void __p_sample_type(char *buf, size_t size, u64 value)
bit_name(IDENTIFIER), bit_name(REGS_INTR), bit_name(DATA_SRC),
bit_name(WEIGHT), bit_name(PHYS_ADDR), bit_name(AUX),
bit_name(CGROUP), bit_name(DATA_PAGE_SIZE), bit_name(CODE_PAGE_SIZE),
- bit_name(WEIGHT_STRUCT),
+ bit_name(WEIGHT_STRUCT), bit_name(BUILD_ID_OFFSET),
+ bit_name(CALLCHAIN_BUILD_ID_OFFSET),
{ .name = NULL, }
};
#undef bit_name
diff --git a/tools/perf/util/sample.c b/tools/perf/util/sample.c
index bccc19e2aaf2..1513eda8765b 100644
--- a/tools/perf/util/sample.c
+++ b/tools/perf/util/sample.c
@@ -24,11 +24,15 @@ void perf_sample__init(struct perf_sample *sample, bool all)
if (all) {
memset(sample, 0, sizeof(*sample));
} else {
+ sample->ip = 0;
sample->evsel = NULL;
sample->user_regs = NULL;
sample->intr_regs = NULL;
sample->merged_callchain = false;
sample->callchain = NULL;
+ sample->merged_callchain_bids = false;
+ sample->callchain_bids = NULL;
+ sample->callchain_bids_nr = 0;
}
}
@@ -42,6 +46,11 @@ void perf_sample__exit(struct perf_sample *sample)
zfree(&sample->callchain);
sample->merged_callchain = false;
}
+ if (sample->merged_callchain_bids) {
+ zfree(&sample->callchain_bids);
+ sample->callchain_bids_nr = 0;
+ sample->merged_callchain_bids = false;
+ }
}
struct regs_dump *perf_sample__user_regs(struct perf_sample *sample)
diff --git a/tools/perf/util/sample.h b/tools/perf/util/sample.h
index cb4b16654876..adf2ff1bd13b 100644
--- a/tools/perf/util/sample.h
+++ b/tools/perf/util/sample.h
@@ -7,7 +7,9 @@
struct evsel;
struct machine;
+struct perf_sample_build_id_offset;
struct thread;
+struct perf_build_id;
/* number of register is bound by the number of bits in regs_dump::mask (64) */
#define PERF_SAMPLE_REGS_CACHE_SIZE (8 * sizeof(u64))
@@ -163,6 +165,22 @@ struct perf_sample {
u64 code_page_size;
/** @cgroup: The sample event PERF_SAMPLE_CGROUP value. */
u64 cgroup;
+ /** @bid: The sample event PERF_SAMPLE_BUILD_ID_OFFSET value. */
+ struct {
+ struct perf_build_id *bid;
+ union {
+ u64 offset;
+ u64 ip;
+ };
+ u64 deferred_storage[3];
+ } bid;
+ /**
+ * @callchain_bids: Build IDs for the callchain when
+ * PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET is set.
+ */
+ struct perf_sample_build_id_offset *callchain_bids;
+ /** @callchain_bids_nr: Number of entries in callchain_bids. */
+ u32 callchain_bids_nr;
/** @file_offset: Byte offset of this event in the perf.data file. */
u64 file_offset;
/** @flags: Extra flag data from auxiliary events like intel-pt. */
@@ -205,11 +223,21 @@ struct perf_sample {
* user callchain marker was encountered.
*/
bool deferred_callchain;
+ /**
+ * @deferred_bid: When processing PERF_SAMPLE_BUILD_ID_OFFSET a deferred
+ * user marker was encountered.
+ */
+ bool deferred_bid;
/**
* @merged_callchain: A synthesized merged callchain that is allocated
* and needs freeing.
*/
bool merged_callchain;
+ /**
+ * @merged_callchain_bids: A synthesized merged callchain_bids that is
+ * allocated and needs freeing.
+ */
+ bool merged_callchain_bids;
/**
* @deferred_cookie: Identifier of the deferred callchain in the later
* PERF_RECORD_CALLCHAIN_DEFERRED event.
diff --git a/tools/perf/util/session.c b/tools/perf/util/session.c
index 5e113070b3c6..d952e186dcf6 100644
--- a/tools/perf/util/session.c
+++ b/tools/perf/util/session.c
@@ -21,6 +21,7 @@
#include "map_symbol.h"
#include "branch.h"
+#include "build-id.h"
#include "debug.h"
#include "dwarf-regs.h"
#include "env.h"
@@ -1326,15 +1327,17 @@ static void callchain__lbr_callstack_printf(struct perf_sample *sample)
struct ip_callchain *callchain = sample->callchain;
struct branch_stack *lbr_stack = sample->branch_stack;
struct branch_entry *entries = perf_sample__branch_entries(sample);
- u64 kernel_callchain_nr = callchain->nr;
+ u64 kernel_callchain_nr = callchain ? callchain->nr : sample->callchain_bids_nr;
unsigned int i;
for (i = 0; i < kernel_callchain_nr; i++) {
- if (callchain->ips[i] == PERF_CONTEXT_USER)
+ u64 ip = callchain ? callchain->ips[i] : sample->callchain_bids[i].offset;
+
+ if (ip == PERF_CONTEXT_USER)
break;
}
- if ((i != kernel_callchain_nr) && lbr_stack->nr) {
+ if ((i != kernel_callchain_nr) && lbr_stack && lbr_stack->nr) {
u64 total_nr;
/*
* LBR callstack can only get user call chain,
@@ -1357,9 +1360,11 @@ static void callchain__lbr_callstack_printf(struct perf_sample *sample)
printf("... LBR call chain: nr:%" PRIu64 "\n", total_nr);
- for (i = 0; i < kernel_callchain_nr; i++)
- printf("..... %2d: %016" PRIx64 "\n",
- i, callchain->ips[i]);
+ for (i = 0; i < kernel_callchain_nr; i++) {
+ u64 ip = callchain ? callchain->ips[i] : sample->callchain_bids[i].offset;
+
+ printf("..... %2d: %016" PRIx64 "\n", i, ip);
+ }
printf("..... %2d: %016" PRIx64 "\n",
(int)(kernel_callchain_nr), entries[0].to);
@@ -1400,12 +1405,24 @@ static void callchain__printf(struct evsel *evsel,
if (evsel__has_branch_callstack(evsel))
callchain__lbr_callstack_printf(sample);
- printf("... FP chain: nr:%" PRIu64 "\n", callchain->nr);
+ if (callchain) {
+ printf("... FP chain: nr:%" PRIu64 "\n", callchain->nr);
- for (i = 0; i < callchain->nr; i++)
- printf("..... %2d: %016" PRIx64 "%s\n",
- i, callchain->ips[i],
- callchain_context_str(callchain->ips[i]));
+ for (i = 0; i < callchain->nr; i++)
+ printf("..... %2d: %016" PRIx64 "%s\n",
+ i, callchain->ips[i],
+ callchain_context_str(callchain->ips[i]));
+ } else if (sample->callchain_bids) {
+ printf("... FP chain (build ID + offset): nr:%u\n",
+ sample->callchain_bids_nr);
+
+ for (i = 0; i < sample->callchain_bids_nr; i++) {
+ u64 offset = sample->callchain_bids[i].offset;
+
+ printf("..... %2d: %016" PRIx64 "%s\n",
+ i, offset, callchain_context_str(offset));
+ }
+ }
if (sample->deferred_callchain)
printf("...... (deferred)\n");
@@ -1730,6 +1747,17 @@ static void dump_sample(struct machine *machine, union perf_event *event,
event->header.misc, sample->pid, sample->tid, sample->ip,
sample->period, sample->addr);
+ if (sample_type & PERF_SAMPLE_BUILD_ID_OFFSET) {
+ struct build_id bid;
+ char sbuild_id[SBUILD_ID_SIZE] = "";
+
+ if (sample->bid.bid) {
+ build_id__init(&bid, sample->bid.bid->data, sample->bid.bid->size);
+ build_id__snprintf(&bid, sbuild_id, sizeof(sbuild_id));
+ }
+ printf("... build_id: %s offset: %#" PRIx64 "\n", sbuild_id, sample->bid.offset);
+ }
+
if (evsel__has_callchain(evsel))
callchain__printf(evsel, sample);
@@ -1783,7 +1811,7 @@ static void dump_deferred_callchain(union perf_event *event, struct perf_sample
printf("(IP, 0x%x): %d/%d: %#" PRIx64 "\n",
event->header.misc, sample->pid, sample->tid, sample->deferred_cookie);
- if (evsel__has_callchain(evsel))
+ if (evsel__has_callchain(evsel) || sample->callchain_bids)
callchain__printf(evsel, sample);
}
diff --git a/tools/perf/util/synthetic-events.c b/tools/perf/util/synthetic-events.c
index 4d552150bc04..e1c04f4ee76e 100644
--- a/tools/perf/util/synthetic-events.c
+++ b/tools/perf/util/synthetic-events.c
@@ -1711,6 +1711,13 @@ size_t perf_event__sample_event_size(const struct perf_sample *sample, u64 type,
result += sample->aux_sample.size;
}
+ if (type & PERF_SAMPLE_BUILD_ID_OFFSET)
+ result += sizeof(struct perf_build_id) + sizeof(u64);
+
+ if (type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET)
+ result += sizeof(u64) + sample->callchain_bids_nr *
+ (sizeof(struct perf_build_id) + sizeof(u64));
+
return result;
}
@@ -1967,6 +1974,27 @@ int perf_event__synthesize_sample(union perf_event *event, u64 type, u64 read_fo
array = (void *)array + sz;
}
+ if (type & PERF_SAMPLE_BUILD_ID_OFFSET) {
+ if (sample->bid.bid) {
+ memcpy(array, sample->bid.bid, sizeof(struct perf_build_id));
+ array += sizeof(struct perf_build_id) / sizeof(u64);
+ *array++ = sample->bid.offset;
+ } else {
+ memset(array, 0, sizeof(struct perf_build_id));
+ array += sizeof(struct perf_build_id) / sizeof(u64);
+ *array++ = sample->ip;
+ }
+ }
+
+ if (type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ *array++ = sample->callchain_bids_nr;
+ if (sample->callchain_bids && sample->callchain_bids_nr) {
+ sz = sample->callchain_bids_nr * sizeof(*sample->callchain_bids);
+ memcpy(array, sample->callchain_bids, sz);
+ array += sz / sizeof(u64);
+ }
+ }
+
return 0;
}
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* [PATCH v2 6/8] perf inject: Extend perf inject to support bid_offset conversion
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
` (4 preceding siblings ...)
2026-10-02 17:38 ` [PATCH v2 5/8] perf tools: Internal support for BUILD_ID_OFFSET Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 7/8] perf record: Add --buildid-offset option Ian Rogers
2026-10-02 17:38 ` [PATCH v2 8/8] perf tests: Add build_id_offset test coverage Ian Rogers
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
This adds the --sample-buildids option to perf inject, allowing it to
drop MMAP events and rewrite samples to use build IDs and offsets
instead of virtual addresses.
Signed-off-by: Ian Rogers <irogers@google.com>
---
tools/perf/builtin-inject.c | 70 ++-
tools/perf/util/Build | 1 +
tools/perf/util/inject_bid_offset.c | 680 ++++++++++++++++++++++++++++
tools/perf/util/inject_bid_offset.h | 15 +
4 files changed, 759 insertions(+), 7 deletions(-)
create mode 100644 tools/perf/util/inject_bid_offset.c
create mode 100644 tools/perf/util/inject_bid_offset.h
diff --git a/tools/perf/builtin-inject.c b/tools/perf/builtin-inject.c
index 42ed367d729f..bdca4a62ad39 100644
--- a/tools/perf/builtin-inject.c
+++ b/tools/perf/builtin-inject.c
@@ -9,6 +9,7 @@
#include "builtin.h"
#include "util/aslr.h"
+#include "util/inject_bid_offset.h"
#include "util/color.h"
#include "util/dso.h"
#include "util/vdso.h"
@@ -46,6 +47,7 @@
#include <errno.h>
#include <signal.h>
#include <inttypes.h>
+#include <stdlib.h>
struct guest_event {
struct perf_sample sample;
@@ -112,6 +114,7 @@ enum build_id_rewrite_style {
BID_RWS__INJECT_HEADER_ALL,
BID_RWS__MMAP2_BUILDID_ALL,
BID_RWS__MMAP2_BUILDID_LAZY,
+ BID_RWS__SAMPLE_BUILDID,
};
struct perf_inject {
@@ -243,12 +246,18 @@ static int perf_event__repipe_attr(const struct perf_tool *tool,
if (ret)
return ret;
- if (inject->aslr) {
+ if (inject->aslr || inject->build_id_style == BID_RWS__SAMPLE_BUILDID) {
aslr_event = malloc(event->header.size);
if (!aslr_event)
return -ENOMEM;
memcpy(aslr_event, event, event->header.size);
- aslr_tool__strip_attr_event(aslr_event, *pevlist);
+ if (inject->aslr)
+ aslr_tool__strip_attr_event(aslr_event, *pevlist);
+ if (inject->build_id_style == BID_RWS__SAMPLE_BUILDID) {
+ ret = perf_event__rewrite_attr_for_build_id_offset(&aslr_event->attr.attr);
+ if (ret)
+ goto out;
+ }
event = aslr_event;
}
@@ -2571,7 +2580,8 @@ static int __cmd_inject(struct perf_inject *inject)
};
if (inject->build_id_style == BID_RWS__INJECT_HEADER_LAZY ||
- inject->build_id_style == BID_RWS__INJECT_HEADER_ALL)
+ inject->build_id_style == BID_RWS__INJECT_HEADER_ALL ||
+ inject->build_id_style == BID_RWS__SAMPLE_BUILDID)
perf_header__set_feat(&session->header, HEADER_BUILD_ID);
/*
* Keep all buildids when there is unprocessed AUX data because
@@ -2626,6 +2636,17 @@ static int __cmd_inject(struct perf_inject *inject)
if (inject->aslr)
aslr_tool__strip_evlist(inject->session->tool, session->evlist);
+ if (inject->build_id_style == BID_RWS__SAMPLE_BUILDID) {
+ struct evsel *evsel;
+
+ evlist__for_each_entry(session->evlist, evsel) {
+ struct perf_event_attr *attr = &evsel->core.attr;
+
+ ret = perf_event__rewrite_attr_for_build_id_offset(attr);
+ if (ret)
+ return ret;
+ }
+ }
session->header.data_offset = output_data_offset;
session->header.data_size = inject->bytes_written;
@@ -2684,6 +2705,7 @@ int cmd_inject(int argc, const char **argv)
bool build_id_all = false;
bool mmap2_build_ids = false;
bool mmap2_build_id_all = false;
+ bool build_id_sample = false;
struct option options[] = {
OPT_BOOLEAN('b', "build-ids", &build_ids,
@@ -2692,8 +2714,11 @@ int cmd_inject(int argc, const char **argv)
"Inject build-ids of all DSOs into the output stream"),
OPT_BOOLEAN('B', "mmap2-buildids", &mmap2_build_ids,
"Drop unused mmap events, make others mmap2 with build IDs"),
+
OPT_BOOLEAN(0, "mmap2-buildid-all", &mmap2_build_id_all,
"Rewrite all mmap events as mmap2 events with build IDs"),
+ OPT_BOOLEAN('S', "sample-buildids", &build_id_sample,
+ "Drop all mmap events and rewrite samples to use build ID + offset"),
OPT_STRING(0, "known-build-ids", &known_build_ids,
"buildid path [,buildid path...]",
"build-ids to use for given paths"),
@@ -2766,8 +2791,8 @@ int cmd_inject(int argc, const char **argv)
if (argc)
usage_with_options(inject_usage, options);
- if (inject.aslr && inject.convert_callchain) {
- pr_err("Error: --aslr and --convert-callchain are mutually exclusive features.\n");
+ if ((inject.aslr + inject.convert_callchain + build_id_sample) > 1) {
+ pr_err("Error: --aslr, --convert-callchain and --sample-buildids are mutually exclusive features.\n");
return -EINVAL;
}
@@ -2812,14 +2837,18 @@ int cmd_inject(int argc, const char **argv)
inject.build_id_style = BID_RWS__MMAP2_BUILDID_ALL;
if (build_ids)
inject.build_id_style = BID_RWS__INJECT_HEADER_LAZY;
+
if (build_id_all)
inject.build_id_style = BID_RWS__INJECT_HEADER_ALL;
+ if (build_id_sample)
+ inject.build_id_style = BID_RWS__SAMPLE_BUILDID;
data.path = inject.input_name;
ordered_events = inject.jit_mode || inject.sched_stat ||
inject.build_id_style == BID_RWS__INJECT_HEADER_LAZY ||
- inject.build_id_style == BID_RWS__MMAP2_BUILDID_LAZY;
+ inject.build_id_style == BID_RWS__MMAP2_BUILDID_LAZY ||
+ inject.build_id_style == BID_RWS__SAMPLE_BUILDID;
perf_tool__init(&inject.tool, ordered_events);
inject.tool.sample = perf_event__repipe_sample;
inject.tool.read = perf_event__repipe_sample;
@@ -2870,6 +2899,12 @@ int cmd_inject(int argc, const char **argv)
ret = -ENOMEM;
goto out_close_output;
}
+ } else if (inject.build_id_style == BID_RWS__SAMPLE_BUILDID) {
+ tool = inject_bid_offset_tool__new(&inject.tool);
+ if (!tool) {
+ ret = -ENOMEM;
+ goto out_close_output;
+ }
}
inject.session = __perf_session__new(&data, tool,
/*trace_event_repipe=*/inject.output.is_pipe,
@@ -2877,8 +2912,11 @@ int cmd_inject(int argc, const char **argv)
if (IS_ERR(inject.session)) {
ret = PTR_ERR(inject.session);
+
if (inject.aslr)
aslr_tool__delete(tool);
+ if (inject.build_id_style == BID_RWS__SAMPLE_BUILDID)
+ inject_bid_offset_tool__delete(tool);
goto out_close_output;
}
@@ -2918,8 +2956,19 @@ int cmd_inject(int argc, const char **argv)
* the input.
*/
if (!data.is_pipe) {
+ struct evsel *evsel;
+
if (inject.aslr)
aslr_tool__strip_evlist(tool, inject.session->evlist);
+ if (inject.build_id_style == BID_RWS__SAMPLE_BUILDID) {
+ evlist__for_each_entry(inject.session->evlist, evsel) {
+ struct perf_event_attr *attr = &evsel->core.attr;
+
+ ret = perf_event__rewrite_attr_for_build_id_offset(attr);
+ if (ret)
+ goto out_delete;
+ }
+ }
ret = perf_event__synthesize_for_pipe(&inject.tool,
inject.session,
@@ -2928,6 +2977,10 @@ int cmd_inject(int argc, const char **argv)
if (inject.aslr)
aslr_tool__restore_evlist(tool, inject.session->evlist);
+ if (inject.build_id_style == BID_RWS__SAMPLE_BUILDID) {
+ evlist__for_each_entry(inject.session->evlist, evsel)
+ perf_event__rewrite_attr_for_sample_ip(&evsel->core.attr);
+ }
if (ret < 0)
goto out_delete;
@@ -2935,7 +2988,8 @@ int cmd_inject(int argc, const char **argv)
}
if (inject.build_id_style == BID_RWS__INJECT_HEADER_LAZY ||
- inject.build_id_style == BID_RWS__MMAP2_BUILDID_LAZY) {
+ inject.build_id_style == BID_RWS__MMAP2_BUILDID_LAZY ||
+ inject.build_id_style == BID_RWS__SAMPLE_BUILDID) {
/*
* to make sure the mmap records are ordered correctly
* and so that the correct especially due to jitted code
@@ -3002,6 +3056,8 @@ int cmd_inject(int argc, const char **argv)
perf_session__delete(inject.session);
if (inject.aslr)
aslr_tool__delete(tool);
+ if (inject.build_id_style == BID_RWS__SAMPLE_BUILDID)
+ inject_bid_offset_tool__delete(tool);
out_close_output:
if (!inject.in_place_update)
perf_data__close(&inject.output);
diff --git a/tools/perf/util/Build b/tools/perf/util/Build
index 512f2ca0cd1a..70732c79d6ac 100644
--- a/tools/perf/util/Build
+++ b/tools/perf/util/Build
@@ -7,6 +7,7 @@ perf-util-y += addr2line.o
perf-util-y += addr_location.o
perf-util-y += annotate.o
perf-util-y += aslr.o
+perf-util-y += inject_bid_offset.o
perf-util-y += blake2s.o
perf-util-y += block-info.o
perf-util-y += block-range.o
diff --git a/tools/perf/util/inject_bid_offset.c b/tools/perf/util/inject_bid_offset.c
new file mode 100644
index 000000000000..25e8f21a2d5a
--- /dev/null
+++ b/tools/perf/util/inject_bid_offset.c
@@ -0,0 +1,680 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "inject_bid_offset.h"
+
+#include <errno.h>
+#include <stdlib.h>
+#include <string.h>
+
+#include <linux/compiler.h>
+#include <linux/kernel.h>
+#include <linux/string.h>
+#include <linux/zalloc.h>
+
+#include "addr_location.h"
+#include "debug.h"
+#include "dso.h"
+#include "event.h"
+#include "evlist.h"
+#include "evsel.h"
+#include "machine.h"
+#include "map.h"
+#include "session.h"
+#include "synthetic-events.h"
+#include "thread.h"
+#include "tool.h"
+
+struct inject_bid_offset_tool {
+ struct delegate_tool tool;
+ char event_copy[PERF_SAMPLE_MAX_SIZE] __aligned(8);
+};
+
+int perf_event__rewrite_attr_for_build_id_offset(struct perf_event_attr *attr)
+{
+ if (attr->sample_type & (PERF_SAMPLE_BUILD_ID_OFFSET |
+ PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET)) {
+ /*
+ * Expect to add build ID information from virtual address, if
+ * it is already present then things would be confused so fail.
+ */
+ return -1;
+ }
+ if (attr->sample_type & PERF_SAMPLE_IP) {
+ attr->sample_type &= ~PERF_SAMPLE_IP;
+ attr->sample_type |= PERF_SAMPLE_BUILD_ID_OFFSET;
+ }
+ if (attr->sample_type & PERF_SAMPLE_CALLCHAIN) {
+ attr->sample_type &= ~PERF_SAMPLE_CALLCHAIN;
+ attr->sample_type |= PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET;
+ }
+ return 0;
+}
+
+int perf_event__rewrite_attr_for_sample_ip(struct perf_event_attr *attr)
+{
+ if (attr->sample_type & (PERF_SAMPLE_IP | PERF_SAMPLE_CALLCHAIN)) {
+ /*
+ * Expect to remove build ID information for virtual address, if
+ * it is already present then things would be confused so fail.
+ */
+ return -1;
+ }
+ if (attr->sample_type & PERF_SAMPLE_BUILD_ID_OFFSET) {
+ attr->sample_type &= ~PERF_SAMPLE_BUILD_ID_OFFSET;
+ attr->sample_type |= PERF_SAMPLE_IP;
+ }
+ if (attr->sample_type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ attr->sample_type &= ~PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET;
+ attr->sample_type |= PERF_SAMPLE_CALLCHAIN;
+ }
+ return 0;
+}
+
+static void perf_event__inject_sample_buildid_array(struct thread *thread,
+ u64 ip, u8 cpumode,
+ __u64 *array)
+{
+ struct perf_build_id bid = { .size = 0 };
+ u64 offset = ip;
+ struct addr_location al;
+ struct dso *dso;
+ const struct build_id *dso_bid;
+
+ struct perf_sample ps = { .ip = ip, .cpumode = cpumode };
+
+ addr_location__init(&al);
+
+ if (!thread)
+ goto write_bid;
+
+ if (!thread__find_map(thread, &ps, &al))
+ goto write_bid;
+
+ dso = al.map ? dso__get(map__dso(al.map)) : NULL;
+ if (!dso)
+ goto write_bid;
+ dso_bid = dso__bid(dso);
+ if (!dso_bid || dso_bid->size == 0 ||
+ map__mapping_type(al.map) != MAPPING_TYPE__DSO) {
+ dso__put(dso);
+ goto write_bid;
+ }
+
+ bid.size = dso_bid->size;
+ if (bid.size > sizeof(bid.data))
+ bid.size = sizeof(bid.data);
+ memcpy(bid.data, &dso_bid->data, bid.size);
+ offset = map__dso_map_ip(al.map, offset);
+ dso__put(dso);
+
+write_bid:
+ compiletime_assert(sizeof(struct perf_build_id) == 3 * sizeof(u64),
+ "Unexpected perf_build_id size");
+ memcpy(&array[0], &bid, 3 * sizeof(u64));
+ array[3] = offset;
+ addr_location__exit(&al);
+}
+
+static int inject_bid_offset_tool__process_build_id(const struct perf_tool *tool,
+ union perf_event *event,
+ struct perf_sample *sample __maybe_unused,
+ struct machine *machine __maybe_unused)
+{
+ return tool->build_id(tool, NULL, event);
+}
+
+static void mark_dso_hit(const struct perf_tool *tool,
+ struct perf_sample *sample, struct machine *machine,
+ struct thread *thread, u64 ip, u8 cpumode)
+{
+ struct addr_location al;
+ struct dso *dso;
+
+ struct perf_sample ps = { .ip = ip, .cpumode = cpumode };
+
+ addr_location__init(&al);
+
+ if (thread__find_map(thread, &ps, &al)) {
+ dso = al.map ? dso__get(map__dso(al.map)) : NULL;
+ if (dso) {
+ if (!dso__hit(dso)) {
+ const struct build_id *bid = dso__bid(dso);
+
+ dso__set_hit(dso);
+ if (bid && bid->size > 0) {
+ perf_event__synthesize_build_id(
+ tool, sample, machine,
+ inject_bid_offset_tool__process_build_id,
+ dso__kernel(dso) ?
+ PERF_RECORD_MISC_KERNEL :
+ PERF_RECORD_MISC_USER,
+ bid, dso->long_name);
+ }
+ }
+ dso__put(dso);
+ }
+ }
+ addr_location__exit(&al);
+}
+
+static int inject_bid_offset_tool__sample(const struct perf_tool *tool,
+ union perf_event *event,
+ struct perf_sample *sample,
+ struct machine *machine)
+{
+ struct delegate_tool *dt =
+ container_of(tool, struct delegate_tool, tool);
+ struct inject_bid_offset_tool *ibo =
+ container_of(dt, struct inject_bid_offset_tool, tool);
+ union perf_event *ev;
+ struct perf_sample new_sample;
+ struct evsel *evsel = sample->evsel;
+ __u64 i = 0, j = 0;
+ __u64 *in_array, *out_array;
+ __u64 sample_type = evsel->core.attr.sample_type;
+ const __u64 max_i = (event->header.size - sizeof(event->header)) / sizeof(__u64);
+ struct thread *thread;
+ size_t max_size = event->header.size;
+ int orig_sample_size, ret;
+
+ if ((sample_type & (PERF_SAMPLE_IP | PERF_SAMPLE_CALLCHAIN)) == 0)
+ return ibo->tool.delegate->sample(ibo->tool.delegate, event,
+ sample, machine);
+
+ if (symbol_conf.guest_code && !machine__is_host(machine))
+ thread = machine__findnew_guest_code(machine, sample->pid);
+ else
+ thread = machine__findnew_thread(machine, sample->pid,
+ sample->tid);
+
+ if (sample_type & PERF_SAMPLE_IP)
+ max_size += sizeof(struct perf_build_id) + sizeof(u64) -
+ sizeof(u64);
+
+ if (sample_type & PERF_SAMPLE_CALLCHAIN) {
+ max_size +=
+ sample->callchain->nr * (sizeof(struct perf_build_id) +
+ sizeof(u64) - sizeof(u64));
+ }
+
+ if (max_size >= PERF_SAMPLE_MAX_SIZE) {
+ pr_debug("Insufficient space to copy event\n");
+ thread__put(thread);
+ return -E2BIG;
+ }
+
+ ev = (union perf_event *)ibo->event_copy;
+ ev->sample.header =
+ (struct perf_event_header){ .type = event->header.type,
+ .misc = event->header.misc,
+ .size = max_size };
+
+ in_array = &event->sample.array[0];
+ out_array = &ev->sample.array[0];
+
+ if (sample_type & PERF_SAMPLE_IDENTIFIER) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_IP) {
+ if (i >= max_i)
+ goto err;
+ i++;
+ if (evsel && thread)
+ mark_dso_hit(ibo->tool.delegate, sample, machine,
+ thread, sample->ip, sample->cpumode);
+ }
+ if (sample_type & PERF_SAMPLE_TID) {
+ union {
+ u64 val64;
+ u32 val32[2];
+ } u;
+
+ if (i >= max_i)
+ goto err;
+ u.val32[0] = sample->pid;
+ u.val32[1] = sample->tid;
+ out_array[j++] = u.val64;
+ i++;
+ }
+ if (sample_type & PERF_SAMPLE_TIME) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_ADDR) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_ID) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_STREAM_ID) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_CPU) {
+ union {
+ u64 val64;
+ u32 val32[2];
+ } u;
+
+ if (i >= max_i)
+ goto err;
+ u.val32[0] = sample->cpu;
+ u.val32[1] = 0;
+ out_array[j++] = u.val64;
+ i++;
+ }
+ if (sample_type & PERF_SAMPLE_PERIOD) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_READ) {
+ if ((evsel->core.attr.read_format & PERF_FORMAT_GROUP) == 0) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ if (evsel->core.attr.read_format &
+ PERF_FORMAT_TOTAL_TIME_ENABLED) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (evsel->core.attr.read_format &
+ PERF_FORMAT_TOTAL_TIME_RUNNING) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (evsel->core.attr.read_format & PERF_FORMAT_ID) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (evsel->core.attr.read_format & PERF_FORMAT_LOST) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ } else {
+ u64 nr;
+
+ if (i >= max_i)
+ goto err;
+ nr = out_array[j++] = in_array[i++];
+ if (evsel->core.attr.read_format &
+ PERF_FORMAT_TOTAL_TIME_ENABLED) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (evsel->core.attr.read_format &
+ PERF_FORMAT_TOTAL_TIME_RUNNING) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ for (u64 cntr = 0; cntr < nr; cntr++) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ if (evsel->core.attr.read_format &
+ PERF_FORMAT_ID) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (evsel->core.attr.read_format &
+ PERF_FORMAT_LOST) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ }
+ }
+ }
+ if (sample_type & PERF_SAMPLE_CALLCHAIN) {
+ if (i >= max_i || sample->callchain->nr > max_i - (i + 1))
+ goto err;
+ i++;
+ if (evsel && thread) {
+ u8 cpumode = sample->cpumode;
+
+ for (u64 x = 0; x < sample->callchain->nr; x++) {
+ u64 ip = sample->callchain->ips[x];
+
+ if (ip >= PERF_CONTEXT_MAX) {
+ switch (ip) {
+ case PERF_CONTEXT_HV:
+ cpumode = PERF_RECORD_MISC_HYPERVISOR;
+ break;
+ case PERF_CONTEXT_KERNEL:
+ cpumode = PERF_RECORD_MISC_KERNEL;
+ break;
+ case PERF_CONTEXT_USER:
+ cpumode = PERF_RECORD_MISC_USER;
+ break;
+ case PERF_CONTEXT_GUEST:
+ case PERF_CONTEXT_GUEST_KERNEL:
+ cpumode = PERF_RECORD_MISC_GUEST_KERNEL;
+ break;
+ case PERF_CONTEXT_GUEST_USER:
+ cpumode = PERF_RECORD_MISC_GUEST_USER;
+ break;
+ case PERF_CONTEXT_USER_DEFERRED:
+ cpumode = PERF_RECORD_MISC_USER;
+ x++;
+ break;
+ default:
+ break;
+ }
+ continue;
+ }
+ mark_dso_hit(ibo->tool.delegate, sample,
+ machine, thread, ip, cpumode);
+ }
+ }
+ i += sample->callchain->nr;
+ }
+ if (sample_type & PERF_SAMPLE_RAW) {
+ size_t bytes = sizeof(u32) + sample->raw_size;
+ u64 words = DIV_ROUND_UP(bytes, sizeof(u64));
+ u32 *out_raw = (u32 *)&out_array[j];
+
+ if (i > max_i || words > max_i - i)
+ goto err;
+ out_array[j + words - 1] = 0;
+ *out_raw = sample->raw_size;
+ memcpy(out_raw + 1, sample->raw_data, sample->raw_size);
+ i += words;
+ j += words;
+ }
+ if (sample_type & PERF_SAMPLE_BRANCH_STACK) {
+ u64 nr = sample->branch_stack->nr;
+
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ if (evsel__has_branch_hw_idx(evsel)) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (i > max_i || (nr * 3) > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i],
+ nr * 3 * sizeof(u64));
+ i += nr * 3;
+ j += nr * 3;
+ if (sample->branch_stack_cntr) {
+ if (i > max_i || nr > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i],
+ nr * sizeof(u64));
+ i += nr;
+ j += nr;
+ }
+ }
+ if (sample_type & PERF_SAMPLE_REGS_USER) {
+ u64 abi;
+
+ if (i >= max_i)
+ goto err;
+ abi = out_array[j++] = in_array[i++];
+ if (abi != PERF_SAMPLE_REGS_ABI_NONE) {
+ u64 nr = hweight64(evsel->core.attr.sample_regs_user);
+
+ if (i > max_i || nr > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i], nr * sizeof(u64));
+ i += nr;
+ j += nr;
+ if (abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ u64 nr_vectors, vector_qwords, nr_pred, pred_qwords, simd_nr;
+
+ if (i > max_i || 4 > max_i - i)
+ goto err;
+ nr_vectors = out_array[j++] = in_array[i++];
+ vector_qwords = out_array[j++] = in_array[i++];
+ nr_pred = out_array[j++] = in_array[i++];
+ pred_qwords = out_array[j++] = in_array[i++];
+ simd_nr = nr_vectors * vector_qwords + nr_pred * pred_qwords;
+ if (i > max_i || simd_nr > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i], simd_nr * sizeof(u64));
+ i += simd_nr;
+ j += simd_nr;
+ }
+ }
+ }
+ if (sample_type & PERF_SAMPLE_STACK_USER) {
+ u64 size;
+
+ if (i >= max_i)
+ goto err;
+ size = out_array[j++] = in_array[i++];
+ if (size > 0) {
+ u64 words = DIV_ROUND_UP(size, sizeof(u64));
+
+ if (i > max_i || words > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i], words * sizeof(u64));
+ i += words;
+ j += words;
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ }
+ if (sample_type & PERF_SAMPLE_WEIGHT_TYPE) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_DATA_SRC) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_TRANSACTION) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_REGS_INTR) {
+ u64 abi;
+
+ if (i >= max_i)
+ goto err;
+ abi = out_array[j++] = in_array[i++];
+ if (abi != PERF_SAMPLE_REGS_ABI_NONE) {
+ u64 nr = hweight64(evsel->core.attr.sample_regs_intr);
+
+ if (i > max_i || nr > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i], nr * sizeof(u64));
+ i += nr;
+ j += nr;
+ if (abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ u64 nr_vectors, vector_qwords, nr_pred, pred_qwords, simd_nr;
+
+ if (i > max_i || 4 > max_i - i)
+ goto err;
+ nr_vectors = out_array[j++] = in_array[i++];
+ vector_qwords = out_array[j++] = in_array[i++];
+ nr_pred = out_array[j++] = in_array[i++];
+ pred_qwords = out_array[j++] = in_array[i++];
+ simd_nr = nr_vectors * vector_qwords + nr_pred * pred_qwords;
+ if (i > max_i || simd_nr > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i], simd_nr * sizeof(u64));
+ i += simd_nr;
+ j += simd_nr;
+ }
+ }
+ }
+ if (sample_type & PERF_SAMPLE_PHYS_ADDR) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_CGROUP) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_DATA_PAGE_SIZE) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_CODE_PAGE_SIZE) {
+ if (i >= max_i)
+ goto err;
+ out_array[j++] = in_array[i++];
+ }
+ if (sample_type & PERF_SAMPLE_AUX) {
+ u64 size, words;
+
+ if (i >= max_i)
+ goto err;
+ size = out_array[j++] = in_array[i++];
+ words = DIV_ROUND_UP(size, sizeof(u64));
+ if (i > max_i || words > max_i - i)
+ goto err;
+ memcpy(&out_array[j], &in_array[i], words * sizeof(u64));
+ i += words;
+ j += words;
+ }
+
+ if (sample_type & PERF_SAMPLE_IP) {
+ perf_event__inject_sample_buildid_array(
+ thread, sample->ip, sample->cpumode, &out_array[j]);
+ j += 4;
+ }
+ if (sample_type & PERF_SAMPLE_CALLCHAIN) {
+ u8 cpumode = sample->cpumode;
+
+ out_array[j++] = sample->callchain->nr;
+ for (u64 x = 0; x < sample->callchain->nr; x++) {
+ u64 ip = sample->callchain->ips[x];
+
+ if (ip >= PERF_CONTEXT_MAX) {
+ switch (ip) {
+ case PERF_CONTEXT_HV:
+ cpumode = PERF_RECORD_MISC_HYPERVISOR;
+ break;
+ case PERF_CONTEXT_KERNEL:
+ cpumode = PERF_RECORD_MISC_KERNEL;
+ break;
+ case PERF_CONTEXT_USER:
+ cpumode = PERF_RECORD_MISC_USER;
+ break;
+ case PERF_CONTEXT_GUEST:
+ case PERF_CONTEXT_GUEST_KERNEL:
+ cpumode = PERF_RECORD_MISC_GUEST_KERNEL;
+ break;
+ case PERF_CONTEXT_GUEST_USER:
+ cpumode = PERF_RECORD_MISC_GUEST_USER;
+ break;
+ case PERF_CONTEXT_USER_DEFERRED:
+ cpumode = PERF_RECORD_MISC_USER;
+ memset(&out_array[j], 0, 3 * sizeof(u64));
+ out_array[j + 3] = ip;
+ j += 4;
+ if (x + 1 < sample->callchain->nr) {
+ x++;
+ memset(&out_array[j], 0, 3 * sizeof(u64));
+ out_array[j + 3] = sample->callchain->ips[x];
+ j += 4;
+ }
+ continue;
+ default:
+ break;
+ }
+ memset(&out_array[j], 0, 3 * sizeof(u64));
+ out_array[j + 3] = ip;
+ } else {
+ perf_event__inject_sample_buildid_array(
+ thread, ip, cpumode, &out_array[j]);
+ }
+ j += 4;
+ }
+ }
+
+ max_size = sizeof(event->header) + j * sizeof(__u64);
+ if (max_size >= PERF_SAMPLE_MAX_SIZE)
+ goto err;
+ ev->sample.header.size = max_size;
+
+ orig_sample_size = evsel->sample_size;
+ if (perf_event__rewrite_attr_for_build_id_offset(&evsel->core.attr))
+ goto err;
+ evsel->sample_size = __evsel__sample_size(evsel->core.attr.sample_type);
+ perf_sample__init(&new_sample, /*all=*/true);
+ ret = __evsel__parse_sample(evsel, ev, &new_sample, /*needs_swap=*/false);
+ if (!ret) {
+ new_sample.evsel = evsel;
+ ret = ibo->tool.delegate->sample(ibo->tool.delegate, ev,
+ &new_sample, machine);
+ }
+ perf_sample__exit(&new_sample);
+ evsel->core.attr.sample_type = sample_type;
+ evsel->sample_size = orig_sample_size;
+ thread__put(thread);
+ return ret;
+
+err:
+ thread__put(thread);
+ return -EFAULT;
+}
+
+static int
+inject_bid_offset_tool__mmap(const struct perf_tool *tool __maybe_unused,
+ union perf_event *event,
+ struct perf_sample *sample,
+ struct machine *machine)
+{
+ perf_event__process_mmap(tool, event, sample, machine);
+ return 0; // Drop mmap events from output stream
+}
+static int
+inject_bid_offset_tool__mmap2(const struct perf_tool *tool __maybe_unused,
+ union perf_event *event,
+ struct perf_sample *sample,
+ struct machine *machine)
+{
+ perf_event__process_mmap2(tool, event, sample, machine);
+ return 0; // Drop mmap2 events from output stream
+}
+
+struct perf_tool *inject_bid_offset_tool__new(struct perf_tool *delegate)
+{
+ struct inject_bid_offset_tool *ibo = zalloc(sizeof(*ibo));
+
+ if (!ibo)
+ return NULL;
+
+ delegate_tool__init(&ibo->tool, delegate);
+ ibo->tool.tool.sample = inject_bid_offset_tool__sample;
+ ibo->tool.tool.mmap = inject_bid_offset_tool__mmap;
+ ibo->tool.tool.mmap2 = inject_bid_offset_tool__mmap2;
+
+ return &ibo->tool.tool;
+}
+
+void inject_bid_offset_tool__delete(struct perf_tool *tool)
+{
+ struct delegate_tool *dt;
+
+ if (!tool)
+ return;
+ dt = container_of(tool, struct delegate_tool, tool);
+ free(container_of(dt, struct inject_bid_offset_tool, tool));
+}
diff --git a/tools/perf/util/inject_bid_offset.h b/tools/perf/util/inject_bid_offset.h
new file mode 100644
index 000000000000..4ed2b3c0d85f
--- /dev/null
+++ b/tools/perf/util/inject_bid_offset.h
@@ -0,0 +1,15 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __PERF_INJECT_BID_OFFSET_H
+#define __PERF_INJECT_BID_OFFSET_H
+
+#include <linux/perf_event.h>
+
+struct perf_tool;
+struct evlist;
+
+int perf_event__rewrite_attr_for_build_id_offset(struct perf_event_attr *attr);
+int perf_event__rewrite_attr_for_sample_ip(struct perf_event_attr *attr);
+struct perf_tool *inject_bid_offset_tool__new(struct perf_tool *delegate);
+void inject_bid_offset_tool__delete(struct perf_tool *tool);
+
+#endif /* __PERF_INJECT_BID_OFFSET_H */
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/8] perf inject: Extend perf inject to support bid_offset conversion
2026-10-02 17:38 ` [PATCH v2 6/8] perf inject: Extend perf inject to support bid_offset conversion Ian Rogers
@ 2026-10-03 1:33 ` sashiko-bot
0 siblings, 0 replies; 27+ messages in thread
From: sashiko-bot @ 2026-10-03 1:33 UTC (permalink / raw)
To: Ian Rogers; +Cc: linux-perf-users
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] perf: strict aliasing violation via incompatible u32* cast
--
commit e992f40013df71934d4af6ea3359426fe5b0f8a1
Author: Ian Rogers <irogers@google.com>
perf inject: Extend perf inject to support bid_offset conversion
This commit adds the --sample-buildids option to perf inject, allowing it to
drop MMAP events and rewrite samples to use build IDs and offsets instead
of virtual addresses.
> diff --git a/tools/perf/util/inject_bid_offset.c b/tools/perf/util/inject_bid_offset.c
> new file mode 100644
> index 0000000000000..25e8f21a2d5aa
> --- /dev/null
> +++ b/tools/perf/util/inject_bid_offset.c
[ ... ]
> @@ -0,0 +1,680 @@
[ ... ]
> + if (sample_type & PERF_SAMPLE_RAW) {
> + size_t bytes = sizeof(u32) + sample->raw_size;
> + u64 words = DIV_ROUND_UP(bytes, sizeof(u64));
> + u32 *out_raw = (u32 *)&out_array[j];
> +
> + if (i > max_i || words > max_i - i)
> + goto err;
> + out_array[j + words - 1] = 0;
> + *out_raw = sample->raw_size;
[Severity: High]
Does this pointer cast and dereference in inject_bid_offset_tool__sample()
violate strict aliasing rules?
Since out_raw is an incompatible u32 pointer cast from the underlying u64
array, the compiler is allowed to assume the stores do not overlap. This
means the compiler can reorder the u64 store (out_array[j + words - 1] = 0)
and the u32 store (*out_raw = sample->raw_size).
If words == 1 (for example, when sample->raw_size <= 4), both assignments
target the exact same 8-byte memory region. Under type-based alias analysis,
reordering these independent stores could result in raw_size being zeroed
out, corrupting the perf data output.
This occurs when using --sample-buildids on an input perf.data file
containing events with PERF_SAMPLE_RAW. Unlike core kernel space, the tools/
directory assumes -fstrict-aliasing by default.
> + memcpy(out_raw + 1, sample->raw_data, sample->raw_size);
> + i += words;
> + j += words;
> + }
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261002173848.3228217-1-irogers@google.com?part=6
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 7/8] perf record: Add --buildid-offset option
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
` (5 preceding siblings ...)
2026-10-02 17:38 ` [PATCH v2 6/8] perf inject: Extend perf inject to support bid_offset conversion Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
2026-10-02 17:38 ` [PATCH v2 8/8] perf tests: Add build_id_offset test coverage Ian Rogers
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Add the --buildid-offset flag to perf record to natively request
PERF_SAMPLE_BUILD_ID_OFFSET and PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET
from the kernel. This configures the evsel attributes by replacing
the standard IP and CALLCHAIN bits with the build ID offset equivalents.
Signed-off-by: Ian Rogers <irogers@google.com>
---
tools/perf/builtin-record.c | 2 ++
tools/perf/util/evsel.c | 55 ++++++++++++++++++++++++++++++++++++-
tools/perf/util/record.h | 1 +
3 files changed, 57 insertions(+), 1 deletion(-)
diff --git a/tools/perf/builtin-record.c b/tools/perf/builtin-record.c
index ad81458989b5..0bd051bf95e1 100644
--- a/tools/perf/builtin-record.c
+++ b/tools/perf/builtin-record.c
@@ -3625,6 +3625,8 @@ static struct option __record_options[] = {
"file", "vmlinux pathname"),
OPT_BOOLEAN(0, "buildid-all", &record.buildid_all,
"Record build-id of all DSOs regardless of hits"),
+ OPT_BOOLEAN(0, "buildid-offset", &record.opts.buildid_offset,
+ "Record build id offset instead of IP"),
OPT_BOOLEAN_SET(0, "buildid-mmap", &record.buildid_mmap, &record.buildid_mmap_set,
"Record build-id in mmap events and skip build-id processing."),
OPT_BOOLEAN(0, "timestamp-filename", &record.timestamp_filename,
diff --git a/tools/perf/util/evsel.c b/tools/perf/util/evsel.c
index 08e9e6567bed..7517e960c58e 100644
--- a/tools/perf/util/evsel.c
+++ b/tools/perf/util/evsel.c
@@ -1549,6 +1549,43 @@ bool evsel__is_offcpu_event(struct evsel *evsel)
evsel->core.attr.sample_type & PERF_SAMPLE_RAW;
}
+static bool evlist__has_callchain_config(struct evlist *evlist,
+ const struct callchain_param *callchain)
+{
+ struct evsel *pos;
+
+ if (callchain && callchain->enabled)
+ return true;
+ if (!evlist)
+ return false;
+
+ evlist__for_each_entry(evlist, pos) {
+ struct evsel_config_term *term;
+ const char *callgraph_buf = NULL;
+ u32 dump_size = 0;
+ int max_stack = 0;
+
+ if (evsel__has_callchain(pos))
+ return true;
+
+ list_for_each_entry(term, &pos->config_terms, list) {
+ if (term->type == EVSEL__CONFIG_TERM_CALLGRAPH)
+ callgraph_buf = term->val.str;
+ else if (term->type == EVSEL__CONFIG_TERM_STACK_USER)
+ dump_size = term->val.stack_user;
+ else if (term->type == EVSEL__CONFIG_TERM_MAX_STACK)
+ max_stack = term->val.max_stack;
+ }
+ if (callgraph_buf) {
+ if (strcmp(callgraph_buf, "no") != 0)
+ return true;
+ } else if (dump_size > 0 || max_stack > 0) {
+ return true;
+ }
+ }
+ return false;
+}
+
/*
* The enable_on_exec/disabled value strategy:
*
@@ -1745,7 +1782,7 @@ void evsel__config(struct evsel *evsel, const struct record_opts *opts,
attr->mmap2 = track && !perf_missing_features.mmap2;
attr->comm = track;
attr->build_id = track && opts->build_id;
- attr->defer_output = track && callchain && callchain->defer;
+ attr->defer_output = track && ((callchain && callchain->defer) || opts->buildid_offset);
/*
* ksymbol is tracked separately with text poke because it needs to be
@@ -1831,6 +1868,22 @@ void evsel__config(struct evsel *evsel, const struct record_opts *opts,
*/
evsel__apply_config_terms(evsel, opts, track);
+ if (opts->buildid_offset && !evsel__is_offcpu_event(evsel)) {
+ if ((evsel->core.attr.sample_type & PERF_SAMPLE_IP) || track) {
+ evsel__reset_sample_bit(evsel, IP);
+ evsel__set_sample_bit(evsel, BUILD_ID_OFFSET);
+ if (!attr->exclude_user)
+ attr->defer_callchain = 1;
+ }
+ if ((evsel->core.attr.sample_type & PERF_SAMPLE_CALLCHAIN) ||
+ (track && evlist__has_callchain_config(evsel->evlist, callchain))) {
+ evsel__reset_sample_bit(evsel, CALLCHAIN);
+ evsel__set_sample_bit(evsel, CALLCHAIN_BUILD_ID_OFFSET);
+ if (!attr->exclude_callchain_user)
+ attr->defer_callchain = 1;
+ }
+ }
+
evsel->ignore_missing_thread = opts->ignore_missing_thread;
/* The --period option takes the precedence. */
diff --git a/tools/perf/util/record.h b/tools/perf/util/record.h
index 411bb7276ad7..6c9568616177 100644
--- a/tools/perf/util/record.h
+++ b/tools/perf/util/record.h
@@ -55,6 +55,7 @@ struct record_opts {
bool kcore;
bool text_poke;
bool build_id;
+ bool buildid_offset;
unsigned int freq;
unsigned int mmap_pages;
unsigned int auxtrace_mmap_pages;
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread* [PATCH v2 8/8] perf tests: Add build_id_offset test coverage
2026-10-02 17:38 ` [PATCH v2 0/8] perf/core, perf/tools: Add PERF_SAMPLE_BUILD_ID_OFFSET support Ian Rogers
` (6 preceding siblings ...)
2026-10-02 17:38 ` [PATCH v2 7/8] perf record: Add --buildid-offset option Ian Rogers
@ 2026-10-02 17:38 ` Ian Rogers
2026-10-03 1:33 ` sashiko-bot
7 siblings, 1 reply; 27+ messages in thread
From: Ian Rogers @ 2026-10-02 17:38 UTC (permalink / raw)
To: irogers, acme, adrian.hunter, mingo, namhyung, peterz
Cc: ak, alexander.shishkin, atrajeev, dvyukov, fzczx123, james.clark,
jolsa, kjain, krzysztof.m.lopatowski, leo.yan, lihuafei1,
linux-kernel, linux-perf-users, linux, m.liska, mark.rutland,
martin.liska, mpetlan, quic_zhonhan, scclevenger, sesse,
stephen.s.brennan, thomas.falcon, yangyicong
Test the kernel interface using sys_perf_event_open, test perf inject
--sample-buildids with pipe mode and callchains, and test perf record
--buildid-offset across multiple sampling modes.
Signed-off-by: Ian Rogers <irogers@google.com>
---
tools/perf/tests/Build | 1 +
tools/perf/tests/bid-offset.c | 60 +++++++
tools/perf/tests/builtin-test.c | 1 +
tools/perf/tests/mmap-thread-lookup.c | 42 ++++-
tools/perf/tests/sample-parsing.c | 68 +++++++-
tools/perf/tests/shell/inject_bid_offset.sh | 175 ++++++++++++++++++++
tools/perf/tests/shell/record.sh | 69 ++++++++
tools/perf/tests/tests.h | 1 +
8 files changed, 414 insertions(+), 3 deletions(-)
create mode 100644 tools/perf/tests/bid-offset.c
create mode 100755 tools/perf/tests/shell/inject_bid_offset.sh
diff --git a/tools/perf/tests/Build b/tools/perf/tests/Build
index 8fecd5390124..6a9563cc5533 100644
--- a/tools/perf/tests/Build
+++ b/tools/perf/tests/Build
@@ -64,6 +64,7 @@ perf-test-y += expand-cgroup.o
perf-test-y += perf-time-to-tsc.o
perf-test-y += dlfilter-test.o
perf-test-y += sigtrap.o
+perf-test-y += bid-offset.o
perf-test-y += event_groups.o
perf-test-y += hybrid-merge.o
perf-test-y += symbols.o
diff --git a/tools/perf/tests/bid-offset.c b/tools/perf/tests/bid-offset.c
new file mode 100644
index 000000000000..8c19d818f0b0
--- /dev/null
+++ b/tools/perf/tests/bid-offset.c
@@ -0,0 +1,60 @@
+// SPDX-License-Identifier: GPL-2.0
+#include "tests.h"
+#include "debug.h"
+#include "evlist.h"
+#include "evsel.h"
+#include <linux/hw_breakpoint.h>
+#include <linux/perf_event.h>
+#include <sys/syscall.h>
+#include <unistd.h>
+#include <errno.h>
+#include "../perf-sys.h"
+
+static int test__bid_offset(struct test_suite *test __maybe_unused, int subtest __maybe_unused)
+{
+ struct perf_event_attr attr = {
+ .type = PERF_TYPE_SOFTWARE,
+ .config = PERF_COUNT_SW_DUMMY,
+ .size = sizeof(attr),
+ .sample_type = PERF_SAMPLE_BUILD_ID_OFFSET,
+ .exclude_kernel = 1,
+ .exclude_hv = 1,
+ };
+ int fd;
+
+ fd = sys_perf_event_open(&attr, /*pid=*/0, /*cpu=*/-1, /*group_fd=*/-1, /*flags=*/0);
+ if (fd < 0) {
+ pr_debug("sys_perf_event_open failed with %d\n", errno);
+ if (errno == EINVAL || errno == EACCES || errno == EPERM) {
+ pr_debug("Kernel does not support or permit PERF_SAMPLE_BUILD_ID_OFFSET\n");
+ return TEST_SKIP;
+ }
+ return TEST_FAIL;
+ }
+ close(fd);
+
+ attr.sample_type = PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET;
+ fd = sys_perf_event_open(&attr, /*pid=*/0, /*cpu=*/-1, /*group_fd=*/-1, /*flags=*/0);
+ if (fd < 0) {
+ pr_debug("sys_perf_event_open with callchain failed with %d\n", errno);
+ if (errno == EINVAL || errno == EACCES || errno == EPERM) {
+ pr_debug("Kernel does not support or permit PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET\n");
+ return TEST_SKIP;
+ }
+ return TEST_FAIL;
+ }
+ close(fd);
+
+ return TEST_OK;
+}
+
+static struct test_case bid_offset_tests[] = {
+ TEST_CASE_REASON("Test PERF_SAMPLE_BUILD_ID_OFFSET", bid_offset,
+ "not supported by kernel"),
+ { .name = NULL, }
+};
+
+struct test_suite suite__bid_offset = {
+ .desc = "Test PERF_SAMPLE_BUILD_ID_OFFSET",
+ .test_cases = bid_offset_tests,
+};
diff --git a/tools/perf/tests/builtin-test.c b/tools/perf/tests/builtin-test.c
index d2f594921e25..ee7109049a79 100644
--- a/tools/perf/tests/builtin-test.c
+++ b/tools/perf/tests/builtin-test.c
@@ -151,6 +151,7 @@ static struct test_suite *generic_tests[] = {
&suite__perf_time_to_tsc,
&suite__dlfilter,
&suite__sigtrap,
+ &suite__bid_offset,
&suite__event_groups,
&suite__hybrid_merge,
&suite__symbols,
diff --git a/tools/perf/tests/mmap-thread-lookup.c b/tools/perf/tests/mmap-thread-lookup.c
index dca8bbfe780a..7af0f8697431 100644
--- a/tools/perf/tests/mmap-thread-lookup.c
+++ b/tools/perf/tests/mmap-thread-lookup.c
@@ -7,6 +7,7 @@
#include <pthread.h>
#include <stdlib.h>
#include <stdio.h>
+#include <string.h>
#include "debug.h"
#include "env.h"
#include "event.h"
@@ -17,7 +18,10 @@
#include "symbol.h"
#include "util/synthetic-events.h"
#include "thread.h"
+#include "dso.h"
+#include "build-id.h"
#include <internal/lib.h> // page_size
+#include <internal/rc_check.h>
#define THREADS 4
@@ -200,17 +204,51 @@ static int mmap_events(synth_cb synth)
.ip = (unsigned long) (td->map + 1),
}, &al);
- thread__put(thread);
-
if (!al.map) {
pr_debug("failed, couldn't find map\n");
err = -1;
+ thread__put(thread);
addr_location__exit(&al);
break;
}
pr_debug("map %p, addr %" PRIx64 "\n", al.map, map__start(al.map));
+ /* Verify lookup by build ID and offset */
+ {
+ struct dso *dso = dso__get(map__dso(al.map));
+
+ if (dso) {
+ const struct build_id *bid = dso__bid(dso);
+
+ if (bid && bid->size > 0) {
+ struct addr_location al2;
+ struct perf_build_id pbid = { .size = bid->size };
+ u64 offset = map__dso_map_ip(al.map,
+ (unsigned long)(td->map + 1));
+
+ memcpy(pbid.data, bid->data, bid->size);
+ addr_location__init(&al2);
+ thread__find_map(thread, &(struct perf_sample){
+ .cpumode = PERF_RECORD_MISC_USER,
+ .ip = 0,
+ .bid = {
+ .bid = &pbid,
+ .offset = offset,
+ },
+ }, &al2);
+
+ if (!RC_CHK_EQUAL(al2.map, al.map)) {
+ pr_debug("failed, couldn't find map by build id and offset\n");
+ err = -1;
+ }
+ addr_location__exit(&al2);
+ }
+ dso__put(dso);
+ }
+ }
+
+ thread__put(thread);
addr_location__exit(&al);
if (err)
break;
diff --git a/tools/perf/tests/sample-parsing.c b/tools/perf/tests/sample-parsing.c
index bd30f6d4c31b..d95bbd3587ab 100644
--- a/tools/perf/tests/sample-parsing.c
+++ b/tools/perf/tests/sample-parsing.c
@@ -218,6 +218,27 @@ static bool samples_same(struct perf_sample *s1,
}
}
+ if (type & PERF_SAMPLE_BUILD_ID_OFFSET) {
+ if (!s1->bid.bid || !s2->bid.bid ||
+ memcmp(s1->bid.bid, s2->bid.bid, sizeof(*s1->bid.bid))) {
+ pr_debug("Samples differ at 'bid.bid'\n");
+ return false;
+ }
+ COMP(bid.offset);
+ }
+
+ if (type & PERF_SAMPLE_CALLCHAIN_BUILD_ID_OFFSET) {
+ const size_t entry_sz = sizeof(struct perf_sample_build_id_offset);
+
+ COMP(callchain_bids_nr);
+ if (!s1->callchain_bids || !s2->callchain_bids ||
+ memcmp(s1->callchain_bids, s2->callchain_bids,
+ s1->callchain_bids_nr * entry_sz)) {
+ pr_debug("Samples differ at 'callchain_bids'\n");
+ return false;
+ }
+ }
+
return true;
}
@@ -247,6 +268,45 @@ static int do_test(u64 sample_type, u64 sample_regs, u64 read_format)
const u32 raw_data[] = {0x12345678, 0x0a0b0c0d, 0x11020304, 0x05060708, 0 };
const u64 data[] = {0x2211443366558877ULL, 0, 0xaabbccddeeff4321ULL};
const u64 aux_data[] = {0xa55a, 0, 0xeeddee, 0x0282028202820282};
+ const struct perf_build_id bid = {
+ .size = 20,
+ .data = {
+ '0', '1', '2', '3', '4', '5', '6', '7', '8', '9',
+ '0', '1', '2', '3', '4', '5', '6', '7', '8', '9',
+ }
+ };
+ const struct perf_sample_build_id_offset callchain_bids[] = {
+ {
+ .bid = {
+ .size = 20,
+ .data = {
+ '9', '8', '7', '6', '5', '4', '3', '2', '1', '0',
+ '9', '8', '7', '6', '5', '4', '3', '2', '1', '0',
+ }
+ },
+ .offset = 201,
+ },
+ {
+ .bid = {
+ .size = 20,
+ .data = {
+ 'a', 'b', 'c', 'd', 'e', 'f', 'a', 'b', 'c', 'd',
+ 'e', 'f', 'a', 'b', 'c', 'd', 'e', 'f', 'a', 'b',
+ }
+ },
+ .offset = 202,
+ },
+ {
+ .bid = {
+ .size = 20,
+ .data = {
+ 'f', 'e', 'd', 'c', 'b', 'a', 'f', 'e', 'd', 'c',
+ 'b', 'a', 'f', 'e', 'd', 'c', 'b', 'a', 'f', 'e',
+ }
+ },
+ .offset = 203,
+ },
+ };
struct regs_dump user_regs = {
.abi = PERF_SAMPLE_REGS_ABI_64,
.mask = sample_regs,
@@ -295,6 +355,12 @@ static int do_test(u64 sample_type, u64 sample_regs, u64 read_format)
.size = sizeof(aux_data),
.data = (void *)aux_data,
},
+ .bid = {
+ .bid = (struct perf_build_id *)&bid,
+ .offset = 119,
+ },
+ .callchain_bids_nr = ARRAY_SIZE(callchain_bids),
+ .callchain_bids = (struct perf_sample_build_id_offset *)callchain_bids,
};
struct sample_read_value values[] = {{1, 5, 0}, {9, 3, 0}, {2, 7, 0}, {6, 4, 1},};
struct sample_read_value packed_values[ARRAY_SIZE(values)];
@@ -547,7 +613,7 @@ static int test__sample_parsing(struct test_suite *test __maybe_unused, int subt
* were added. Please actually update the test rather than just change
* the condition below.
*/
- if (PERF_SAMPLE_MAX > PERF_SAMPLE_WEIGHT_STRUCT << 1) {
+ if (PERF_SAMPLE_WEIGHT_STRUCT << 3 < PERF_SAMPLE_MAX) {
pr_debug("sample format has changed, some new PERF_SAMPLE_ bit was introduced - test needs updating\n");
return -1;
}
diff --git a/tools/perf/tests/shell/inject_bid_offset.sh b/tools/perf/tests/shell/inject_bid_offset.sh
new file mode 100755
index 000000000000..a930f9b92f7e
--- /dev/null
+++ b/tools/perf/tests/shell/inject_bid_offset.sh
@@ -0,0 +1,175 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0
+# perf inject --sample-buildids test
+
+set -e
+set -o pipefail
+
+shelldir=$(dirname "$0")
+# shellcheck source=lib/perf_has_symbol.sh
+. "${shelldir}"/lib/perf_has_symbol.sh
+
+sym="noploop"
+
+skip_test_missing_symbol ${sym}
+
+temp_dir=$(mktemp -d /tmp/perf-test-bid-offset.XXXXXXXXXX)
+
+prog="perf test -w noploop"
+[ "$(uname -m)" = "s390x" ] && prog="$prog 3"
+err=0
+
+cleanup() {
+ local exit_code=${1:-$?}
+ trap - EXIT TERM INT
+ if [ "${exit_code}" -ne 0 ] || [ "${err}" -ne 0 ]; then
+ echo "Test failed! Preserving temp directory: ${temp_dir}"
+ exit 1
+ fi
+ if [[ "${temp_dir}" =~ ^/tmp/perf-test-bid-offset\. ]]; then
+ rm -rf "${temp_dir}"
+ fi
+ exit 0
+}
+
+trap_cleanup() {
+ local exit_code=$?
+ echo "Unexpected signal in ${FUNCNAME[1]}"
+ cleanup ${exit_code}
+}
+trap trap_cleanup EXIT TERM INT
+
+compare_script() {
+ local orig=$1
+ local bid=$2
+ local msg=$3
+
+ perf script -F -ip -i "${orig}" > "${temp_dir}/script_orig.txt"
+ perf script -F -ip -i "${bid}" > "${temp_dir}/script_bid.txt"
+
+ if ! diff -q "${temp_dir}/script_orig.txt" "${temp_dir}/script_bid.txt"; then
+ echo "${msg} [Failed - perf script output differs]"
+ err=1
+ else
+ echo "${msg} [Success]"
+ fi
+}
+
+compare_report() {
+ local orig=$1
+ local bid=$2
+ local msg=$3
+
+ perf report -i "${orig}" --stdio | grep -v '^#' > "${temp_dir}/report_orig.txt"
+ perf report -i "${bid}" --stdio | grep -v '^#' > "${temp_dir}/report_bid.txt"
+
+ if ! diff -q "${temp_dir}/report_orig.txt" "${temp_dir}/report_bid.txt"; then
+ echo "${msg} [Failed - perf report output differs]"
+ err=1
+ else
+ echo "${msg} [Success]"
+ fi
+}
+
+test_basic() {
+ echo "Test basic build id offset injection"
+ local data="${temp_dir}/perf.data.basic"
+ local data2="${temp_dir}/perf.data2.basic"
+
+ perf record -e task-clock:u -o "${data}" ${prog} >/dev/null 2>&1
+ perf inject --sample-buildids -i "${data}" -o "${data2}" >/dev/null 2>&1
+
+ compare_script "${data}" "${data2}" "Basic injection"
+}
+
+test_callchain() {
+ echo "Test Callchain build id offset injection"
+ local data="${temp_dir}/perf.data.callchain"
+ local data2="${temp_dir}/perf.data2.callchain"
+
+ perf record -g -e task-clock:u -o "${data}" ${prog} >/dev/null 2>&1
+ perf inject --sample-buildids -i "${data}" -o "${data2}" >/dev/null 2>&1
+
+ compare_script "${data}" "${data2}" "Callchain injection"
+}
+
+test_pipe() {
+ echo "Test pipe mode build id offset injection"
+ local data="${temp_dir}/perf.data.pipe"
+ local data2="${temp_dir}/perf.data2.pipe"
+
+ perf record -e task-clock:u -o - ${prog} 2>/dev/null | tee "${data}" | \
+ perf inject --sample-buildids -o "${data2}" >/dev/null 2>&1
+
+ compare_script "${data}" "${data2}" "Pipe mode injection"
+}
+
+test_report() {
+ echo "Test perf report consistency"
+ local data="${temp_dir}/perf.data.report"
+ local data2="${temp_dir}/perf.data2.report"
+
+ perf record -e task-clock:u -o "${data}" ${prog} >/dev/null 2>&1
+ perf inject --sample-buildids -i "${data}" -o "${data2}" >/dev/null 2>&1
+
+ compare_report "${data}" "${data2}" "Report consistency"
+}
+
+test_pipe_report() {
+ echo "Test pipe mode perf report consistency"
+ local data="${temp_dir}/perf.data.pipe_report"
+ local data2="${temp_dir}/perf.data2.pipe_report"
+
+ perf record -e task-clock:u -o - ${prog} 2>/dev/null | tee "${data}" | \
+ perf inject --sample-buildids -o "${data2}" >/dev/null 2>&1
+
+ compare_report "${data}" "${data2}" "Pipe Report consistency"
+}
+
+test_kernel() {
+ echo "Test kernel build id offset injection"
+ local data="${temp_dir}/perf.data.kernel"
+ local data2="${temp_dir}/perf.data2.kernel"
+
+ # Not all systems allow kernel profiling, so we check if it succeeds first
+ if ! perf record -e task-clock:k -o "${data}" ${prog} >/dev/null 2>&1; then
+ echo "Kernel injection [Skipped - permissions or support missing]"
+ return
+ fi
+ perf inject --sample-buildids -i "${data}" -o "${data2}" >/dev/null 2>&1
+
+ compare_script "${data}" "${data2}" "Kernel injection"
+}
+
+test_record_buildid_offset() {
+ echo "Test perf record --buildid-offset consistency"
+ local data="${temp_dir}/perf.data.record_orig"
+ local data2="${temp_dir}/perf.data2.record_bid"
+
+ perf record -e task-clock:u -o "${data}" ${prog} >/dev/null 2>&1
+ # We must have the binaries in .build-id cache, which the first record does natively.
+ # Now record with --buildid-offset.
+ if ! perf record --buildid-offset -e task-clock:u -o "${data2}" \
+ ${prog} >/dev/null 2>&1; then
+ echo "Record --buildid-offset consistency [Skipped - not supported by kernel]"
+ return
+ fi
+
+ # Verify symbolization resolves the workload symbol
+ if ! perf report -i "${data2}" --stdio | grep -q "${sym}"; then
+ echo "Record --buildid-offset consistency [Failed - missing ${sym} in perf report]"
+ err=1
+ else
+ echo "Record --buildid-offset consistency [Success]"
+ fi
+}
+
+test_basic
+test_callchain
+test_pipe
+test_report
+test_pipe_report
+test_kernel
+test_record_buildid_offset
+
+cleanup ${err}
diff --git a/tools/perf/tests/shell/record.sh b/tools/perf/tests/shell/record.sh
index 01dc3e22fa73..bbb3fd0d2a73 100755
--- a/tools/perf/tests/shell/record.sh
+++ b/tools/perf/tests/shell/record.sh
@@ -740,6 +740,74 @@ test_ratio_to_prev() {
echo "Basic ratio-to-prev record test [Success]"
}
+check_buildid_offset() {
+ perf report --header -D -i "${perfdata}" 2>/dev/null | grep -q "BUILD_ID_OFFSET"
+}
+
+test_buildid_offset() {
+ echo "buildid-offset test"
+ local ret=0
+ perf_record_with_retry "${perfdata}" "check_buildid_offset" "perf test -w thloop" \
+ --buildid-offset || ret=$?
+ if [ $ret -eq 2 ]; then
+ echo "buildid-offset test [Skipped not supported]"
+ return
+ elif [ $ret -eq 1 ]; then
+ echo "buildid-offset test [Failed missing output]"
+ err=1
+ return
+ fi
+
+ # Expand test coverage
+ perf_record_with_retry "${perfdata}" "check_per_thread" "perf test -w thloop" \
+ --buildid-offset --per-thread || ret=$?
+ if [ $ret -ne 0 ]; then
+ echo "buildid-offset test [Failed per-thread]"
+ err=1
+ return
+ fi
+
+ perf_record_with_retry "${perfdata}" "check_system_wide" "perf test -w thloop" \
+ --buildid-offset -aB --synth=no || ret=$?
+ if [ $ret -ne 0 ]; then
+ echo "buildid-offset test [Failed system-wide]"
+ err=1
+ return
+ fi
+
+ perf_record_with_retry "${perfdata}" "check_workload" "perf test -w thloop" \
+ --buildid-offset || ret=$?
+ if [ $ret -ne 0 ]; then
+ echo "buildid-offset test [Failed workload]"
+ err=1
+ return
+ fi
+
+ case $(uname -m)
+ in s390x)
+ cmd_flags="--call-graph dwarf -e cpu-clock";;
+ *)
+ cmd_flags="-g";;
+ esac
+
+ if ! perf record --buildid-offset -o "${perfdata}" $cmd_flags \
+ perf test -w brstack > /dev/null 2>&1
+ then
+ echo "buildid-offset test [Failed callgraph record]"
+ err=1
+ return
+ fi
+
+ if ! perf report -i "${perfdata}" 2>&1 | grep -q "${testsym2}"
+ then
+ echo "buildid-offset test [Failed callgraph missing symbol]"
+ err=1
+ return
+ fi
+
+ echo "buildid-offset test [Success]"
+}
+
# raise the limit of file descriptors to minimum
if [[ $default_fd_limit -lt $min_fd_limit ]]; then
ulimit -Sn $min_fd_limit
@@ -760,6 +828,7 @@ test_precise_max
test_callgraph
test_acr_sampling
test_ratio_to_prev
+test_buildid_offset
# restore the default value
ulimit -Sn $default_fd_limit
diff --git a/tools/perf/tests/tests.h b/tools/perf/tests/tests.h
index 9c96f33483d1..c0a1026c3978 100644
--- a/tools/perf/tests/tests.h
+++ b/tools/perf/tests/tests.h
@@ -176,6 +176,7 @@ DECLARE_SUITE(expand_cgroup_events);
DECLARE_SUITE(perf_time_to_tsc);
DECLARE_SUITE(dlfilter);
DECLARE_SUITE(sigtrap);
+DECLARE_SUITE(bid_offset);
DECLARE_SUITE(event_groups);
DECLARE_SUITE(hybrid_merge);
DECLARE_SUITE(symbols);
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply related [flat|nested] 27+ messages in thread