Re: [PATCH v5 16/26] perf annotate-arm64: Support load instruction tracking
From: Tengda Wu
Date: Thu Sep 10 2026 - 09:35:24 EST
On 2026/9/8 21:05, Tengda Wu wrote:
> Extend update_insn_state_arm64() to handle load instructions, tracking
> register state changes when data is loaded from memory to registers.
>
> Two main categories are considered: standard load (e.g, ldr variants)
> and load pair (ldp), for example:
>
> ldr dst, [src, #imm]
> ldp dst1, dst2, [src, #imm]
>
> For data type propagation, a load pair can be treated as two standard
> loads combined, first loading dst1 and then dst2. The difference is
> that when loading dst2, an additional memory offset 'mem_offset' after
> the first load must be added. This mem_offset can be derived from the
> instruction mnemonic and the size of the dst register. Each load operation
> is handled by introducing propagate_load_reg_state().
>
> When processing a load pair, the case where dst and src registers are
> the same (e.g., ldp x0, x1, [x0]) also needs to be considered. Therefore,
> before propagating data types, a snapshot of the src register's state
> must be saved and then passed to propagate_load_reg_state().
>
> Finally, handle the side effects of pre-index and post-index addressing
> via adjust_reg_index_state().
>
> A real-world example is shown below:
>
> ffff80008011f5b0 <pick_task_stop>:
> ffff80008011f5b8: ldr x0, [x0, #2712] // x0: struct rq* -> task_struct*
> * ffff80008011f5c0: ldr w1, [x0, #104]
>
> Before this commit, the type of x0 was incorrectly inferred as 'struct rq':
>
> find data type for 0x68(reg0) at pick_task_stop+0x10
> var [8] reg0 offset 0 type='struct rq*'
> chk [10] reg0 offset=0x68 ok=1 kind=1 (struct rq*) : Good!
> final result: type='struct rq'
>
> After this commit, the type of x0 is correctly inferred as 'struct task_struct':
>
> find data type for 0x68(reg0) at pick_task_stop+0x10
> var [8] reg0 offset 0 type='struct rq*'
> ldr [8] 0xa98(reg0) -> reg0 type='struct task_struct*'
> chk [10] reg0 offset=0x68 ok=1 kind=1 (struct task_struct*) : Good!
> final result: type='struct task_struct'
>
> Signed-off-by: Li Huafei <lihuafei1@xxxxxxxxxx>
> Signed-off-by: Tengda Wu <wutengda@xxxxxxxxxxxxxxx>
> ---
> .../perf/util/annotate-arch/annotate-arm64.c | 202 +++++++++++++++++-
> 1 file changed, 201 insertions(+), 1 deletion(-)
>
> diff --git a/tools/perf/util/annotate-arch/annotate-arm64.c b/tools/perf/util/annotate-arch/annotate-arm64.c
> index 9a3226526522..c239ad9473dc 100644
> --- a/tools/perf/util/annotate-arch/annotate-arm64.c
> +++ b/tools/perf/util/annotate-arch/annotate-arm64.c
> @@ -455,11 +455,206 @@ static bool is_readonly_branch_or_cmp(const char *name)
> !strcmp(name, "ccmp") || !strcmp(name, "ccmn");
> }
>
> +static int arm64__reg_size(const char *reg)
> +{
> + if (!reg || !*reg || !arm64__is_reg(reg))
> + return -1;
> +
> + if (reg[0] == 'w')
> + return 4;
> +
> + if (reg[0] == 'x' || !strncmp(reg, "sp", 2))
> + return 8;
> +
> + return -1;
> +}
> +
> +/* Apply addressing mode (pre-index, post-index) to register state */
> +static void adjust_reg_index_state(struct type_state *state,
> + struct data_loc_info *dloc,
> + struct disasm_line *dl,
> + struct annotated_op_loc *op_loc)
> +{
> + struct type_state_reg *tsr;
> + int reg = op_loc->reg1;
> + int offset;
> +
> + if (op_loc->addr_mode != PERF_AAM_PRE_INDEX &&
> + op_loc->addr_mode != PERF_AAM_POST_INDEX)
> + return;
> +
> + if (!has_reg_type(state, reg) || !state->regs[reg].ok)
> + return;
> +
> + tsr = &state->regs[reg];
> + tsr->copied_from = -1;
> +
> + if (arch_get_reg_offset(dloc->arch, op_loc, reg, state, true, &offset)) {
> + invalidate_reg_state(tsr);
> + return;
> + }
> +
> + tsr->offset += offset;
> +
> + pr_debug_dtp("%s [%x] %s-index %#x(reg%d) -> reg%d", dl->ins.name,
> + (u32) dl->al.offset, op_loc->addr_mode == PERF_AAM_PRE_INDEX ?
> + "pre" : "post", offset, reg, reg);
> + pr_debug_type_name(&tsr->type, tsr->kind);
> +}
> +
> +/*
> + * Match standard load variants (ldr, ldur, ldar, ldp) that follow
> + * straightforward load semantics: memory source on the right
> + * and target registers on the left (written). Excludes exclusive loads
> + * (ldxr, ldxp, etc.) and acquire/exclusive combination variants.
> + */
> +static bool is_standard_load_insn(const char *name)
> +{
> + return !strncmp(name, "ldr", 3) || /* ldr, ldrb, ldrh, ldrsb, ldrsh, ldrsw */
> + !strncmp(name, "ldur", 4) || /* ldur, ldurb, ldurh, ldursb, ldursh, ldursw */
> + !strncmp(name, "ldar", 4) || /* ldar, ldarb, ldarh */
> + !strncmp(name, "ldp", 3); /* ldp, ldpsw */
> +}
> +
> +/*
> + * For load insns: propagate type from source reg state @src_states to @dreg.
> + *
> + * @mem_offset accounts for additional memory byte offset when handling multi-reg
> + * loads (e.g. second target reg in 'ldp'), which is added to the reg offset.
> + */
> +static int propagate_load_reg_state(struct type_state *state,
> + struct data_loc_info *dloc,
> + struct disasm_line *dl, int dreg,
> + struct annotated_op_loc *src,
> + struct type_state_reg **src_states,
> + int mem_offset)
> +{
> + struct type_state_reg *tsr;
> + struct type_state_reg *src_tsr = src_states[0];
> + Dwarf_Die type_die;
> + u32 insn_offset = dl->al.offset;
> + int sreg = src->reg1;
> + int reg_offset;
> +
> + if (!has_reg_type(state, dreg))
> + return -1;
> +
> + tsr = &state->regs[dreg];
> + tsr->copied_from = -1;
> +
> +retry:
> + if (arch_get_reg_offset(dloc->arch, src, sreg, state, false, ®_offset))
> + return -1;
> +
> + reg_offset += mem_offset;
> +
> + if (!src_tsr || !src_tsr->ok)
> + return -1;
> +
> + /* Dereference the pointer if it has one */
> + if (src_tsr->kind == TSR_KIND_TYPE &&
> + die_deref_ptr_type(&src_tsr->type,
> + src_tsr->offset + reg_offset, &type_die)) {
> + tsr->type = type_die;
> + tsr->kind = TSR_KIND_TYPE;
> + tsr->offset = 0;
> + tsr->ok = true;
> +
> + if (src->multi_regs) {
> + pr_debug_dtp("%s [%x] %#x(reg%d, reg%d) -> reg%d",
> + dl->ins.name, insn_offset, reg_offset,
> + src->reg1, src->reg2, dreg);
> + } else {
> + pr_debug_dtp("%s [%x] %#x(reg%d) -> reg%d",
> + dl->ins.name, insn_offset, reg_offset,
> + sreg, dreg);
> + }
> + pr_debug_type_name(&tsr->type, tsr->kind);
> + return 0;
> + }
> +
> + /* Or try another register if any */
> + if (src->multi_regs && src->reg1 != src->reg2 && sreg != src->reg2 &&
> + !(src->extend_type || src->shift_type)) {
> + sreg = src->reg2;
> + src_tsr = src_states[1];
> + goto retry;
> + }
> +
> + return -1;
> +}
> +
> +static void update_load_insn_state(struct type_state *state,
> + struct data_loc_info *dloc,
> + struct disasm_line *dl,
> + struct annotated_op_loc *src,
> + struct annotated_op_loc *dst)
> +{
> + struct type_state_reg snapshots[2];
> + struct type_state_reg *src_states[2] = {};
> +
> + if (!src->mem_ref || /* exclude PC-relative loads */
> + !has_reg_type(state, dst->reg1) ||
> + (dst->multi_regs && !has_reg_type(state, dst->reg2)))
> + goto out_err_adjust;
> +
Just realized the condition here is not quite right -- rejecting too early
here leaves no chance for the type of dst->reg1 to be propagated. For example,
consider the following instruction:
ldp x1, xzr, [x2]
It would be more appropriate to change it to a check on src->reg1:
if (!src->mem_ref || /* exclude PC-relative loads */
!has_reg_type(state, src->reg1))
goto out_invalidate_all;
After changing it to this, the issue raised by Sashiko in the subsequent
patch 0021, where both sreg and fbreg are -1, can also be resolved.
Thanks,
Tengda
> + /*
> + * Snapshot source register states before updating destination registers.
> + * This avoids state corruption in aliased instructions (e.g., ldp x0, x1, [x0]),
> + * ensuring the second destination load still uses the original base
> + * register state.
> + */
> + if (has_reg_type(state, src->reg1)) {
> + snapshots[0] = state->regs[src->reg1];
> + src_states[0] = &snapshots[0];
> + }
> + if (src->multi_regs && has_reg_type(state, src->reg2)) {
> + snapshots[1] = state->regs[src->reg2];
> + src_states[1] = &snapshots[1];
> + }
> +
> + /* Handle the first destination register */
> + if (propagate_load_reg_state(state, dloc, dl, dst->reg1,
> + src, src_states, /*mem_offset=*/0))
> + goto out_err_adjust;
> +
> + /* Handle the second destination register (if any) */
> + if (dst->multi_regs) {
> + int mem_offset;
> +
> + /*
> + * ldpsw loads two 32-bit signed words into 64-bit registers.
> + * Memory spacing between elements is 4 bytes, not the register size.
> + */
> + if (!strcmp(dl->ins.name, "ldpsw"))
> + mem_offset = 4;
> + else
> + mem_offset = arm64__reg_size(dl->ops.target.raw);
> +
> + if (mem_offset < 0 ||
> + propagate_load_reg_state(state, dloc, dl, dst->reg2,
> + src, src_states, mem_offset))
> + goto out_err_adjust;
> + }
> +
> +out_adjust:
> + adjust_reg_index_state(state, dloc, dl, src);
> + return;
> +
> +out_err_adjust:
> + if (has_reg_type(state, dst->reg1))
> + invalidate_reg_state(&state->regs[dst->reg1]);
> + if (dst->multi_regs && has_reg_type(state, dst->reg2))
> + invalidate_reg_state(&state->regs[dst->reg2]);
> + goto out_adjust;
> +}
> +
> static void update_insn_state_arm64(struct type_state *state,
> struct data_loc_info *dloc, Dwarf_Die *cu_die,
> struct disasm_line *dl)
> {
> struct annotated_insn_loc loc;
> + struct annotated_op_loc *src = &loc.ops[INSN_OP_SOURCE];
> struct annotated_op_loc *dst = &loc.ops[INSN_OP_TARGET];
> u32 insn_offset = dl->al.offset;
>
> @@ -519,7 +714,8 @@ static void update_insn_state_arm64(struct type_state *state,
> * Invalidate destination register(s) for unsupported instructions to
> * prevent stale type info from propagating to subsequent instructions.
> */
> - if (has_reg_type(state, dst->reg1) && !dst->mem_ref) {
> + if (has_reg_type(state, dst->reg1) && !dst->mem_ref &&
> + !is_standard_load_insn(dl->ins.name)) {
> pr_debug_dtp("%s [%x] invalidate reg%d",
> dl->ins.name, insn_offset, dst->reg1);
> invalidate_reg_state(&state->regs[dst->reg1]);
> @@ -530,6 +726,10 @@ static void update_insn_state_arm64(struct type_state *state,
> pr_debug_dtp("\n");
> return;
> }
> +
> + /* Memory to register transfers */
> + if (is_standard_load_insn(dl->ins.name))
> + update_load_insn_state(state, dloc, dl, src, dst);
> }
> #endif
>