* [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole
@ 2026-09-11 4:09 Yonghong Song
2026-09-11 4:10 ` [PATCH dwarves 2/2] tests: tests: Add test for 16-byte aligned arguments on arm64 Yonghong Song
` (2 more replies)
0 siblings, 3 replies; 5+ messages in thread
From: Yonghong Song @ 2026-09-11 4:09 UTC (permalink / raw)
To: Alan Maguire, Arnaldo Carvalho de Melo, dwarves
Cc: Alexei Starovoitov, Andrii Nakryiko, bpf, kernel-team
arm64 requires an argument whose alignment is twice the register size to
start on an even-numbered argument register, so such an argument arriving
when the next free register is an odd one leaves that register unused:
u64 f_odd(u64 a, __int128 v, u64 b);
passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4. The x86-64 ABI
has no such rule and packs v into rsi:rdx instead.
Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
---
dwarf_loader.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++---
dwarves.h | 1 +
2 files changed, 51 insertions(+), 3 deletions(-)
diff --git a/dwarf_loader.c b/dwarf_loader.c
index 61ef52f..81c2076 100644
--- a/dwarf_loader.c
+++ b/dwarf_loader.c
@@ -1501,6 +1501,23 @@ static bool arch__agg_use_two_regs(const GElf_Ehdr *ehdr)
}
}
+/*
+ * Some ABIs require an argument whose alignment is twice the register size to
+ * start on an even-numbered argument register, leaving a hole when the next
+ * free register is an odd one. For example, on arm64,
+ * u64 f(u64 a, __int128 v, u64 b)
+ * passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4.
+ */
+static bool arch__arg_align_two_regs(const GElf_Ehdr *ehdr)
+{
+ switch (ehdr->e_machine) {
+ case EM_AARCH64:
+ return true;
+ default:
+ return false;
+ }
+}
+
static struct template_type_param *template_type_param__new(Dwarf_Die *die, struct cu *cu, struct conf_load *conf)
{
struct template_type_param *ttparm = tag__alloc(cu, sizeof(*ttparm));
@@ -3634,6 +3651,28 @@ static int parameter__abi_slots(const struct parameter *parm, const struct cu *c
return slots > 0 ? slots : 1;
}
+static int parameter__abi_reg_align(const struct parameter *parm, const struct cu *cu)
+{
+ struct tag *type;
+
+ if (!cu->arg_align_two_regs || parm->type_byte_size <= cu->addr_size)
+ return 1;
+
+ type = tag__strip_typedefs_and_modifiers(&parm->tag, cu);
+ if (type == NULL)
+ return 1;
+
+ return tag__natural_alignment(type, cu) > cu->addr_size ? 2 : 1;
+}
+
+static int parameter__align_reg_idx(const struct parameter *parm, int reg_idx,
+ const struct cu *cu)
+{
+ int align = parameter__abi_reg_align(parm, cu);
+
+ return (reg_idx + align - 1) & ~(align - 1);
+}
+
static bool parameter__has_piece_info(const struct parameter *parm)
{
return parm->first_reg_fields || parm->second_reg_fields;
@@ -3653,7 +3692,7 @@ static bool ftype__next_parameter_preserves_slots(struct ftype *ftype, struct pa
if (!next || next->loc_reg == PARAMETER_UNKNOWN_REG)
return false;
- next_reg_idx = reg_idx + slots;
+ next_reg_idx = parameter__align_reg_idx(next, reg_idx + slots, cu);
return next_reg_idx < cu->nr_register_params &&
next->loc_reg == cu->register_params[next_reg_idx];
}
@@ -3702,6 +3741,7 @@ static void function__match_clang_parameter_locations(struct ftype *ftype, struc
if (pos->passed_in_memory)
continue;
+ reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
if (reg_idx >= cu->nr_register_params)
break;
@@ -3732,11 +3772,17 @@ static void function__analyze_parameter_locations(struct function *fn, struct cu
ftype__for_each_parameter(ftype, pos) {
bool consumes_register = true;
- bool regs_available = reg_idx < cu->nr_register_params;
+ bool regs_available;
int slots = parameter__abi_slots(pos, cu);
- int expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
+ int expected_reg;
int reg_slots = pos->passed_in_memory ? 1 : slots;
+ if (!pos->passed_in_memory)
+ reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
+
+ regs_available = reg_idx < cu->nr_register_params;
+ expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
+
if (pos->has_loc) {
if (true_sig_enabled && pos->loc_const_value) {
pos->optimized = 1;
@@ -4467,6 +4513,7 @@ static int cu__set_common(struct cu *cu, struct conf_load *conf,
cu->little_endian = ehdr.e_ident[EI_DATA] == ELFDATA2LSB;
cu->nr_register_params = arch__nr_register_params(&ehdr);
cu->agg_use_two_regs = arch__agg_use_two_regs(&ehdr);
+ cu->arg_align_two_regs = arch__arg_align_two_regs(&ehdr);
arch__set_register_params(&ehdr, cu);
return 0;
}
diff --git a/dwarves.h b/dwarves.h
index df77f1e..70adbbf 100644
--- a/dwarves.h
+++ b/dwarves.h
@@ -304,6 +304,7 @@ struct cu {
uint8_t little_endian:1;
uint8_t producer_clang:1;
uint8_t agg_use_two_regs:1; /* An aggregate like {long a; long b;} */
+ uint8_t arg_align_two_regs:1; /* An over-aligned arg starts on an even register */
uint8_t nr_register_params;
int register_params[ARCH_MAX_REGISTER_PARAMS];
int functions_saved;
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 5+ messages in thread
* [PATCH dwarves 2/2] tests: tests: Add test for 16-byte aligned arguments on arm64
2026-09-11 4:09 [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
@ 2026-09-11 4:10 ` Yonghong Song
2026-09-11 16:02 ` [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
2026-09-15 11:40 ` Alan Maguire
2 siblings, 0 replies; 5+ messages in thread
From: Yonghong Song @ 2026-09-11 4:10 UTC (permalink / raw)
To: Alan Maguire, Arnaldo Carvalho de Melo, dwarves
Cc: Alexei Starovoitov, Andrii Nakryiko, bpf, kernel-team
Cover the arm64 even-register rule for 16-byte aligned arguments:
f_even() and f_stack() where the rule does not bite, f_odd() and
f_odd_tail() where it leaves a register hole, and f_box() where the
alignment comes from a struct member rather than from the parameter
type itself.
The test results on arm64:
$ VERBOSE=1 ./clang_parm_align16.sh
Validation of BTF encoding of over-aligned arguments.
BTF: u64 f_even(u64 a, u64 b, __int128 v);
BTF: u64 f_odd(u64 a, __int128 v, u64 b);
BTF: u64 f_odd_tail(u64 a, __int128 v, u64 b, u64 c, u64 d);
BTF: u64 f_stack(u64 a, u64 b, u64 c, u64 d, u64 e, u64 f, u64 g, __int128 v);
BTF: u64 f_box(u64 a, struct box s, u64 b);
Test ./clang_parm_align16.sh passed
Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
---
tests/clang_parm_align16.sh | 77 +++++++++++++++++++++++++++++++++++++
tests/test_lib.sh | 3 +-
2 files changed, 78 insertions(+), 2 deletions(-)
create mode 100755 tests/clang_parm_align16.sh
diff --git a/tests/clang_parm_align16.sh b/tests/clang_parm_align16.sh
new file mode 100755
index 0000000..484aeb6
--- /dev/null
+++ b/tests/clang_parm_align16.sh
@@ -0,0 +1,77 @@
+#!/bin/bash
+# SPDX-License-Identifier: GPL-2.0-only
+
+source test_lib.sh
+
+outdir=$(make_tmpdir)
+
+# Comment this out to save test data.
+trap cleanup EXIT
+
+title_log "Validation of BTF encoding of over-aligned arguments."
+
+align16="${outdir}/align16"
+CC=$(which clang 2>/dev/null)
+
+if [[ -z "$CC" ]]; then
+ info_log "skip: clang not available"
+ test_skip
+fi
+
+arch=$(uname -m)
+if [[ "$arch" != "aarch64" ]]; then
+ info_log "skip: test is arm64 only, running on $arch"
+ test_skip
+fi
+
+# arm64 makes an argument whose alignment is 16 start on an even-numbered
+# argument register, so f_odd() passes a in x0, v in x2:x3 -- leaving x1 as a
+# hole -- and b in x4. pahole has to account for that hole, otherwise every
+# parameter after v looks like it is in an unexpected register and the whole
+# function is dropped from BTF.
+cat > ${align16}.c << EOF
+typedef unsigned long long u64;
+struct box { __int128 v; };
+
+__attribute__((noinline)) u64 f_even(u64 a, u64 b, __int128 v)
+{ return a + b + (u64)v; }
+
+__attribute__((noinline)) u64 f_odd(u64 a, __int128 v, u64 b)
+{ return a + b + (u64)v; }
+
+__attribute__((noinline)) u64 f_odd_tail(u64 a, __int128 v, u64 b, u64 c, u64 d)
+{ return a + b + c + d + (u64)v; }
+
+__attribute__((noinline)) u64 f_stack(u64 a, u64 b, u64 c, u64 d, u64 e, u64 f,
+ u64 g, __int128 v)
+{ return a + b + c + d + e + f + g + (u64)v; }
+
+__attribute__((noinline)) u64 f_box(u64 a, struct box s, u64 b)
+{ return a + b + (u64)s.v; }
+
+u64 (*keep[])() = { (u64(*)())f_even, (u64(*)())f_odd, (u64(*)())f_odd_tail,
+ (u64(*)())f_stack, (u64(*)())f_box };
+EOF
+
+${CC} -g -O2 -c -o ${align16}.o ${align16}.c 2>/dev/null
+if [[ $? -ne 0 ]]; then
+ info_log "skip: clang could not compile ${align16}.c"
+ test_skip
+fi
+
+LLVM_OBJCOPY=objcopy pahole -J --btf_features=consistent_func ${align16}.o
+if [[ $? -ne 0 ]]; then
+ error_log "Could not encode BTF for ${align16}.o"
+ test_fail
+fi
+
+for fn in f_even f_odd f_odd_tail f_stack f_box; do
+ encoded=$(pfunct --all --format_path=btf ${align16}.o | grep " ${fn}(")
+ verbose_log "BTF: $encoded"
+ if [[ -z "$encoded" ]]; then
+ error_log "${fn}() is missing from BTF"
+ test_fail
+ fi
+done
+
+test_pass
diff --git a/tests/test_lib.sh b/tests/test_lib.sh
index e2a7218..e2d0b0c 100755
--- a/tests/test_lib.sh
+++ b/tests/test_lib.sh
@@ -661,8 +661,7 @@ check_bpftool_btf_support()
cleanup()
{
if [ -n "$outdir" ] && [ -d "$outdir" ]; then
- rm ${outdir}/*
- rmdir $outdir
+ rm -rf "$outdir"
fi
return 0
}
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 5+ messages in thread
* Re: [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole
2026-09-11 4:09 [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
2026-09-11 4:10 ` [PATCH dwarves 2/2] tests: tests: Add test for 16-byte aligned arguments on arm64 Yonghong Song
@ 2026-09-11 16:02 ` Yonghong Song
2026-09-15 11:40 ` Alan Maguire
2 siblings, 0 replies; 5+ messages in thread
From: Yonghong Song @ 2026-09-11 16:02 UTC (permalink / raw)
To: Alan Maguire, Arnaldo Carvalho de Melo, dwarves
Cc: Alexei Starovoitov, Andrii Nakryiko, bpf, kernel-team
On 9/10/26 9:09 PM, Yonghong Song wrote:
> arm64 requires an argument whose alignment is twice the register size to
> start on an even-numbered argument register, so such an argument arriving
> when the next free register is an odd one leaves that register unused:
>
> u64 f_odd(u64 a, __int128 v, u64 b);
>
> passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4. The x86-64 ABI
> has no such rule and packs v into rsi:rdx instead.
The kernel patch (bpf: Support by-value struct and __int128 arguments)
https://lore.kernel.org/bpf/20260911154914.2004336-1-yonghong.song@linux.dev/
depends on this pahole patch. See kernel patch 15.
It would be great if this patch can be reviewed soon.
>
> Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
> ---
> dwarf_loader.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++---
> dwarves.h | 1 +
> 2 files changed, 51 insertions(+), 3 deletions(-)
>
> diff --git a/dwarf_loader.c b/dwarf_loader.c
> index 61ef52f..81c2076 100644
> --- a/dwarf_loader.c
> +++ b/dwarf_loader.c
> @@ -1501,6 +1501,23 @@ static bool arch__agg_use_two_regs(const GElf_Ehdr *ehdr)
> }
> }
>
> +/*
> + * Some ABIs require an argument whose alignment is twice the register size to
> + * start on an even-numbered argument register, leaving a hole when the next
> + * free register is an odd one. For example, on arm64,
> + * u64 f(u64 a, __int128 v, u64 b)
> + * passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4.
> + */
> +static bool arch__arg_align_two_regs(const GElf_Ehdr *ehdr)
> +{
> + switch (ehdr->e_machine) {
> + case EM_AARCH64:
> + return true;
> + default:
> + return false;
> + }
> +}
> +
> static struct template_type_param *template_type_param__new(Dwarf_Die *die, struct cu *cu, struct conf_load *conf)
> {
> struct template_type_param *ttparm = tag__alloc(cu, sizeof(*ttparm));
> @@ -3634,6 +3651,28 @@ static int parameter__abi_slots(const struct parameter *parm, const struct cu *c
> return slots > 0 ? slots : 1;
> }
>
> +static int parameter__abi_reg_align(const struct parameter *parm, const struct cu *cu)
> +{
> + struct tag *type;
> +
> + if (!cu->arg_align_two_regs || parm->type_byte_size <= cu->addr_size)
> + return 1;
> +
> + type = tag__strip_typedefs_and_modifiers(&parm->tag, cu);
> + if (type == NULL)
> + return 1;
> +
> + return tag__natural_alignment(type, cu) > cu->addr_size ? 2 : 1;
> +}
> +
> +static int parameter__align_reg_idx(const struct parameter *parm, int reg_idx,
> + const struct cu *cu)
> +{
> + int align = parameter__abi_reg_align(parm, cu);
> +
> + return (reg_idx + align - 1) & ~(align - 1);
> +}
> +
> static bool parameter__has_piece_info(const struct parameter *parm)
> {
> return parm->first_reg_fields || parm->second_reg_fields;
> @@ -3653,7 +3692,7 @@ static bool ftype__next_parameter_preserves_slots(struct ftype *ftype, struct pa
> if (!next || next->loc_reg == PARAMETER_UNKNOWN_REG)
> return false;
>
> - next_reg_idx = reg_idx + slots;
> + next_reg_idx = parameter__align_reg_idx(next, reg_idx + slots, cu);
> return next_reg_idx < cu->nr_register_params &&
> next->loc_reg == cu->register_params[next_reg_idx];
> }
> @@ -3702,6 +3741,7 @@ static void function__match_clang_parameter_locations(struct ftype *ftype, struc
> if (pos->passed_in_memory)
> continue;
>
> + reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
> if (reg_idx >= cu->nr_register_params)
> break;
>
> @@ -3732,11 +3772,17 @@ static void function__analyze_parameter_locations(struct function *fn, struct cu
>
> ftype__for_each_parameter(ftype, pos) {
> bool consumes_register = true;
> - bool regs_available = reg_idx < cu->nr_register_params;
> + bool regs_available;
> int slots = parameter__abi_slots(pos, cu);
> - int expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
> + int expected_reg;
> int reg_slots = pos->passed_in_memory ? 1 : slots;
>
> + if (!pos->passed_in_memory)
> + reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
> +
> + regs_available = reg_idx < cu->nr_register_params;
> + expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
> +
> if (pos->has_loc) {
> if (true_sig_enabled && pos->loc_const_value) {
> pos->optimized = 1;
> @@ -4467,6 +4513,7 @@ static int cu__set_common(struct cu *cu, struct conf_load *conf,
> cu->little_endian = ehdr.e_ident[EI_DATA] == ELFDATA2LSB;
> cu->nr_register_params = arch__nr_register_params(&ehdr);
> cu->agg_use_two_regs = arch__agg_use_two_regs(&ehdr);
> + cu->arg_align_two_regs = arch__arg_align_two_regs(&ehdr);
> arch__set_register_params(&ehdr, cu);
> return 0;
> }
> diff --git a/dwarves.h b/dwarves.h
> index df77f1e..70adbbf 100644
> --- a/dwarves.h
> +++ b/dwarves.h
> @@ -304,6 +304,7 @@ struct cu {
> uint8_t little_endian:1;
> uint8_t producer_clang:1;
> uint8_t agg_use_two_regs:1; /* An aggregate like {long a; long b;} */
> + uint8_t arg_align_two_regs:1; /* An over-aligned arg starts on an even register */
> uint8_t nr_register_params;
> int register_params[ARCH_MAX_REGISTER_PARAMS];
> int functions_saved;
^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole
2026-09-11 4:09 [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
2026-09-11 4:10 ` [PATCH dwarves 2/2] tests: tests: Add test for 16-byte aligned arguments on arm64 Yonghong Song
2026-09-11 16:02 ` [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
@ 2026-09-15 11:40 ` Alan Maguire
2026-09-17 14:17 ` Yonghong Song
2 siblings, 1 reply; 5+ messages in thread
From: Alan Maguire @ 2026-09-15 11:40 UTC (permalink / raw)
To: Yonghong Song, Arnaldo Carvalho de Melo, dwarves
Cc: Alexei Starovoitov, Andrii Nakryiko, bpf, kernel-team
On 11/09/2026 05:09, Yonghong Song wrote:
> arm64 requires an argument whose alignment is twice the register size to
> start on an even-numbered argument register, so such an argument arriving
> when the next free register is an odd one leaves that register unused:
>
> u64 f_odd(u64 a, __int128 v, u64 b);
>
> passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4. The x86-64 ABI
> has no such rule and packs v into rsi:rdx instead.
>
> Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
> ---
> dwarf_loader.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++---
> dwarves.h | 1 +
> 2 files changed, 51 insertions(+), 3 deletions(-)
>
> diff --git a/dwarf_loader.c b/dwarf_loader.c
> index 61ef52f..81c2076 100644
> --- a/dwarf_loader.c
> +++ b/dwarf_loader.c
> @@ -1501,6 +1501,23 @@ static bool arch__agg_use_two_regs(const GElf_Ehdr *ehdr)
> }
> }
>
> +/*
> + * Some ABIs require an argument whose alignment is twice the register size to
> + * start on an even-numbered argument register, leaving a hole when the next
> + * free register is an odd one. For example, on arm64,
> + * u64 f(u64 a, __int128 v, u64 b)
> + * passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4.
> + */
> +static bool arch__arg_align_two_regs(const GElf_Ehdr *ehdr)
> +{
> + switch (ehdr->e_machine) {
> + case EM_AARCH64:
> + return true;
> + default:
> + return false;
> + }
> +}
> +
> static struct template_type_param *template_type_param__new(Dwarf_Die *die, struct cu *cu, struct conf_load *conf)
> {
> struct template_type_param *ttparm = tag__alloc(cu, sizeof(*ttparm));
> @@ -3634,6 +3651,28 @@ static int parameter__abi_slots(const struct parameter *parm, const struct cu *c
> return slots > 0 ? slots : 1;
> }
>
> +static int parameter__abi_reg_align(const struct parameter *parm, const struct cu *cu)
> +{
> + struct tag *type;
> +
> + if (!cu->arg_align_two_regs || parm->type_byte_size <= cu->addr_size)
> + return 1;
> +
> + type = tag__strip_typedefs_and_modifiers(&parm->tag, cu);
> + if (type == NULL)
> + return 1;
> +
> + return tag__natural_alignment(type, cu) > cu->addr_size ? 2 : 1;
> +}
There is a problem here (identified by AI) which, while not impacting on kernel
signatures would I think be worth fixing. It stems from the fact that we can
have non general-purpose registers > pointer size in a function signatures,
and if they are present they do not advance the general purpose register number by 2
in the way that dedicating 2 general-purpose registers would.
Example signature:
u64 f_fp(u64 a, long double v, u64 b);
To catch this, have a test in parameter__abi_reg_align()
if (!parameter__uses_gpr_bank(parm, cu))
return 1;
static bool parameter__uses_gpr_bank(const struct parameter *parm, const struct cu *cu)
{
struct tag *type = tag__strip_typedefs_and_modifiers(&parm->tag, cu);
if (!type)
return false;
/* Scalars represented by FP/SIMD registers have their own allocation bank. */
if (tag__is_base_type(type) && base_type__is_float(tag__base_type(type)))
return false;
/* Also exclude vector/SIMD types if dwarves represents them distinctly. */
if (tag__is_vector(type))
return false;
return true;
}
So given that the fix is small, I think this would be worth doing.
> +
> +static int parameter__align_reg_idx(const struct parameter *parm, int reg_idx,
> + const struct cu *cu)
> +{
> + int align = parameter__abi_reg_align(parm, cu);
> +
> + return (reg_idx + align - 1) & ~(align - 1);
> +}
> +
> static bool parameter__has_piece_info(const struct parameter *parm)
> {
> return parm->first_reg_fields || parm->second_reg_fields;
> @@ -3653,7 +3692,7 @@ static bool ftype__next_parameter_preserves_slots(struct ftype *ftype, struct pa
> if (!next || next->loc_reg == PARAMETER_UNKNOWN_REG)
> return false;
>
> - next_reg_idx = reg_idx + slots;
> + next_reg_idx = parameter__align_reg_idx(next, reg_idx + slots, cu);
> return next_reg_idx < cu->nr_register_params &&
> next->loc_reg == cu->register_params[next_reg_idx];
> }
> @@ -3702,6 +3741,7 @@ static void function__match_clang_parameter_locations(struct ftype *ftype, struc
> if (pos->passed_in_memory)
> continue;
>
> + reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
> if (reg_idx >= cu->nr_register_params)
> break;
>
> @@ -3732,11 +3772,17 @@ static void function__analyze_parameter_locations(struct function *fn, struct cu
>
> ftype__for_each_parameter(ftype, pos) {
> bool consumes_register = true;
> - bool regs_available = reg_idx < cu->nr_register_params;
> + bool regs_available;
> int slots = parameter__abi_slots(pos, cu);
> - int expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
> + int expected_reg;
> int reg_slots = pos->passed_in_memory ? 1 : slots;
>
> + if (!pos->passed_in_memory)
> + reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
> +
> + regs_available = reg_idx < cu->nr_register_params;
> + expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
> +
> if (pos->has_loc) {
> if (true_sig_enabled && pos->loc_const_value) {
> pos->optimized = 1;
> @@ -4467,6 +4513,7 @@ static int cu__set_common(struct cu *cu, struct conf_load *conf,
> cu->little_endian = ehdr.e_ident[EI_DATA] == ELFDATA2LSB;
> cu->nr_register_params = arch__nr_register_params(&ehdr);
> cu->agg_use_two_regs = arch__agg_use_two_regs(&ehdr);
> + cu->arg_align_two_regs = arch__arg_align_two_regs(&ehdr);
> arch__set_register_params(&ehdr, cu);
> return 0;
> }
> diff --git a/dwarves.h b/dwarves.h
> index df77f1e..70adbbf 100644
> --- a/dwarves.h
> +++ b/dwarves.h
> @@ -304,6 +304,7 @@ struct cu {
> uint8_t little_endian:1;
> uint8_t producer_clang:1;
> uint8_t agg_use_two_regs:1; /* An aggregate like {long a; long b;} */
> + uint8_t arg_align_two_regs:1; /* An over-aligned arg starts on an even register */
> uint8_t nr_register_params;
> int register_params[ARCH_MAX_REGISTER_PARAMS];
> int functions_saved;
^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole
2026-09-15 11:40 ` Alan Maguire
@ 2026-09-17 14:17 ` Yonghong Song
0 siblings, 0 replies; 5+ messages in thread
From: Yonghong Song @ 2026-09-17 14:17 UTC (permalink / raw)
To: Alan Maguire, Arnaldo Carvalho de Melo, dwarves
Cc: Alexei Starovoitov, Andrii Nakryiko, bpf, kernel-team
On 9/15/26 4:40 AM, Alan Maguire wrote:
> On 11/09/2026 05:09, Yonghong Song wrote:
>> arm64 requires an argument whose alignment is twice the register size to
>> start on an even-numbered argument register, so such an argument arriving
>> when the next free register is an odd one leaves that register unused:
>>
>> u64 f_odd(u64 a, __int128 v, u64 b);
>>
>> passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4. The x86-64 ABI
>> has no such rule and packs v into rsi:rdx instead.
>>
>> Signed-off-by: Yonghong Song <yonghong.song@linux.dev>
>> ---
>> dwarf_loader.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++---
>> dwarves.h | 1 +
>> 2 files changed, 51 insertions(+), 3 deletions(-)
>>
>> diff --git a/dwarf_loader.c b/dwarf_loader.c
>> index 61ef52f..81c2076 100644
>> --- a/dwarf_loader.c
>> +++ b/dwarf_loader.c
>> @@ -1501,6 +1501,23 @@ static bool arch__agg_use_two_regs(const GElf_Ehdr *ehdr)
>> }
>> }
>>
>> +/*
>> + * Some ABIs require an argument whose alignment is twice the register size to
>> + * start on an even-numbered argument register, leaving a hole when the next
>> + * free register is an odd one. For example, on arm64,
>> + * u64 f(u64 a, __int128 v, u64 b)
>> + * passes a in x0, v in x2:x3 -- skipping x1 -- and b in x4.
>> + */
>> +static bool arch__arg_align_two_regs(const GElf_Ehdr *ehdr)
>> +{
>> + switch (ehdr->e_machine) {
>> + case EM_AARCH64:
>> + return true;
>> + default:
>> + return false;
>> + }
>> +}
>> +
>> static struct template_type_param *template_type_param__new(Dwarf_Die *die, struct cu *cu, struct conf_load *conf)
>> {
>> struct template_type_param *ttparm = tag__alloc(cu, sizeof(*ttparm));
>> @@ -3634,6 +3651,28 @@ static int parameter__abi_slots(const struct parameter *parm, const struct cu *c
>> return slots > 0 ? slots : 1;
>> }
>>
>> +static int parameter__abi_reg_align(const struct parameter *parm, const struct cu *cu)
>> +{
>> + struct tag *type;
>> +
>> + if (!cu->arg_align_two_regs || parm->type_byte_size <= cu->addr_size)
>> + return 1;
>> +
>> + type = tag__strip_typedefs_and_modifiers(&parm->tag, cu);
>> + if (type == NULL)
>> + return 1;
>> +
>> + return tag__natural_alignment(type, cu) > cu->addr_size ? 2 : 1;
>> +}
>
> There is a problem here (identified by AI) which, while not impacting on kernel
> signatures would I think be worth fixing. It stems from the fact that we can
> have non general-purpose registers > pointer size in a function signatures,
> and if they are present they do not advance the general purpose register number by 2
> in the way that dedicating 2 general-purpose registers would.
>
> Example signature:
>
> u64 f_fp(u64 a, long double v, u64 b);
>
> To catch this, have a test in parameter__abi_reg_align()
>
> if (!parameter__uses_gpr_bank(parm, cu))
> return 1;
>
>
> static bool parameter__uses_gpr_bank(const struct parameter *parm, const struct cu *cu)
> {
> struct tag *type = tag__strip_typedefs_and_modifiers(&parm->tag, cu);
>
> if (!type)
> return false;
>
> /* Scalars represented by FP/SIMD registers have their own allocation bank. */
> if (tag__is_base_type(type) && base_type__is_float(tag__base_type(type)))
> return false;
>
> /* Also exclude vector/SIMD types if dwarves represents them distinctly. */
> if (tag__is_vector(type))
> return false;
>
> return true;
> }
>
> So given that the fix is small, I think this would be worth doing.
Okay, thanks for the suggestion. I will add these in the next revision.
BTW, AI found another couple of places where parameter__uses_gpr_bank()
should be used. I will fold them too.
>
>> +
>> +static int parameter__align_reg_idx(const struct parameter *parm, int reg_idx,
>> + const struct cu *cu)
>> +{
>> + int align = parameter__abi_reg_align(parm, cu);
>> +
>> + return (reg_idx + align - 1) & ~(align - 1);
>> +}
>> +
>> static bool parameter__has_piece_info(const struct parameter *parm)
>> {
>> return parm->first_reg_fields || parm->second_reg_fields;
>> @@ -3653,7 +3692,7 @@ static bool ftype__next_parameter_preserves_slots(struct ftype *ftype, struct pa
>> if (!next || next->loc_reg == PARAMETER_UNKNOWN_REG)
>> return false;
>>
>> - next_reg_idx = reg_idx + slots;
>> + next_reg_idx = parameter__align_reg_idx(next, reg_idx + slots, cu);
>> return next_reg_idx < cu->nr_register_params &&
>> next->loc_reg == cu->register_params[next_reg_idx];
>> }
>> @@ -3702,6 +3741,7 @@ static void function__match_clang_parameter_locations(struct ftype *ftype, struc
>> if (pos->passed_in_memory)
>> continue;
>>
>> + reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
>> if (reg_idx >= cu->nr_register_params)
>> break;
>>
>> @@ -3732,11 +3772,17 @@ static void function__analyze_parameter_locations(struct function *fn, struct cu
>>
>> ftype__for_each_parameter(ftype, pos) {
>> bool consumes_register = true;
>> - bool regs_available = reg_idx < cu->nr_register_params;
>> + bool regs_available;
>> int slots = parameter__abi_slots(pos, cu);
>> - int expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
>> + int expected_reg;
>> int reg_slots = pos->passed_in_memory ? 1 : slots;
>>
>> + if (!pos->passed_in_memory)
>> + reg_idx = parameter__align_reg_idx(pos, reg_idx, cu);
>> +
>> + regs_available = reg_idx < cu->nr_register_params;
>> + expected_reg = regs_available ? cu->register_params[reg_idx] : -1;
>> +
>> if (pos->has_loc) {
>> if (true_sig_enabled && pos->loc_const_value) {
>> pos->optimized = 1;
>> @@ -4467,6 +4513,7 @@ static int cu__set_common(struct cu *cu, struct conf_load *conf,
>> cu->little_endian = ehdr.e_ident[EI_DATA] == ELFDATA2LSB;
>> cu->nr_register_params = arch__nr_register_params(&ehdr);
>> cu->agg_use_two_regs = arch__agg_use_two_regs(&ehdr);
>> + cu->arg_align_two_regs = arch__arg_align_two_regs(&ehdr);
>> arch__set_register_params(&ehdr, cu);
>> return 0;
>> }
>> diff --git a/dwarves.h b/dwarves.h
>> index df77f1e..70adbbf 100644
>> --- a/dwarves.h
>> +++ b/dwarves.h
>> @@ -304,6 +304,7 @@ struct cu {
>> uint8_t little_endian:1;
>> uint8_t producer_clang:1;
>> uint8_t agg_use_two_regs:1; /* An aggregate like {long a; long b;} */
>> + uint8_t arg_align_two_regs:1; /* An over-aligned arg starts on an even register */
>> uint8_t nr_register_params;
>> int register_params[ARCH_MAX_REGISTER_PARAMS];
>> int functions_saved;
^ permalink raw reply [flat|nested] 5+ messages in thread
end of thread, other threads:[~2026-09-17 14:17 UTC | newest]
Thread overview: 5+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-11 4:09 [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
2026-09-11 4:10 ` [PATCH dwarves 2/2] tests: tests: Add test for 16-byte aligned arguments on arm64 Yonghong Song
2026-09-11 16:02 ` [PATCH dwarves 1/2] dwarf_loader: Skip the argument register the arm64 ABI leaves as a hole Yonghong Song
2026-09-15 11:40 ` Alan Maguire
2026-09-17 14:17 ` Yonghong Song
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox;
as well as URLs for NNTP newsgroup(s).