Re: [Regression] [PATCH v5 2/4] vdso: Switch get/put unaligned from packed struct to memcpy
From: Marc Kleine-Budde
Date: Tue Oct 06 2026 - 08:32:16 EST
Cc+ Arnd
On 28.09.2026 17:39:15, Stefan Kerkmann wrote:
> Hi Ian,
>
> On 10/16/25 22:51, Ian Rogers wrote:
> > Type punning is necessary for get/put unaligned but the use of a
> > packed struct violates strict aliasing rules, requiring
> > -fno-strict-aliasing to be passed to the C compiler. Switch to using
> > memcpy so that -fno-strict-aliasing isn't necessary.
> >
> > Signed-off-by: Ian Rogers <irogers@xxxxxxxxxx>
> > ---
> > include/vdso/unaligned.h | 41 ++++++++++++++++++++++++++++++++++------
> > 1 file changed, 35 insertions(+), 6 deletions(-)
> >
> > diff --git a/include/vdso/unaligned.h b/include/vdso/unaligned.h
> > index ff0c06b6513e..9076483c9fbb 100644
> > --- a/include/vdso/unaligned.h
> > +++ b/include/vdso/unaligned.h
> > @@ -2,14 +2,43 @@
> > #ifndef __VDSO_UNALIGNED_H
> > #define __VDSO_UNALIGNED_H
> >
> > -#define __get_unaligned_t(type, ptr) ({ \
> > - const struct { type x; } __packed * __get_pptr = (typeof(__get_pptr))(ptr); \
> > - __get_pptr->x; \
> > +#include <linux/compiler_types.h>
> > +
> > +/**
> > + * __get_unaligned_t - read an unaligned value from memory.
> > + * @type: the type to load from the pointer.
> > + * @ptr: the pointer to load from.
> > + *
> > + * Use memcpy to affect an unaligned type sized load avoiding undefined behavior
> > + * from approaches like type punning that require -fno-strict-aliasing in order
> > + * to be correct. As type may be const, use __unqual_scalar_typeof to map to a
> > + * non-const type - you can't memcpy into a const type. The
> > + * __get_unaligned_ctrl_type gives __unqual_scalar_typeof its required
> > + * expression rather than type, a pointer is used to avoid warnings about mixing
> > + * the use of 0 and NULL. The void* cast silences ubsan warnings.
> > + */
> > +#define __get_unaligned_t(type, ptr) ({ \
> > + type *__get_unaligned_ctrl_type __always_unused = NULL; \
> > + __unqual_scalar_typeof(*__get_unaligned_ctrl_type) __get_unaligned_val; \
> > + __builtin_memcpy(&__get_unaligned_val, (void *)(ptr), \
> > + sizeof(__get_unaligned_val)); \
> > + __get_unaligned_val; \
> > })
> >
> > -#define __put_unaligned_t(type, val, ptr) do { \
> > - struct { type x; } __packed * __put_pptr = (typeof(__put_pptr))(ptr); \
> > - __put_pptr->x = (val); \
> > +/**
> > + * __put_unaligned_t - write an unaligned value to memory.
> > + * @type: the type of the value to store.
> > + * @val: the value to store.
> > + * @ptr: the pointer to store to.
> > + *
> > + * Use memcpy to affect an unaligned type sized store avoiding undefined
> > + * behavior from approaches like type punning that require -fno-strict-aliasing
> > + * in order to be correct. The void* cast silences ubsan warnings.
> > + */
> > +#define __put_unaligned_t(type, val, ptr) do { \
> > + type __put_unaligned_val = (val); \
> > + __builtin_memcpy((void *)(ptr), &__put_unaligned_val, \
> > + sizeof(__put_unaligned_val)); \
> > } while (0)
> >
> > #endif /* __VDSO_UNALIGNED_H */
>
> commit a339671db64b ("vdso: Switch get/put_unaligned() from packed struct to
> memcpy()"), which landed in 7.0, causes a performance regression on an NXP
> i.MX25 (ARMv5TE) SoC.
>
> I found it while updating a client's board from 6.12 to 7.0. A fio 4k randwrite
> benchmark on a NAND storage with UBI and UBIFS filesystem was the only workload
> that showed a clear regression between those two versions, so I bisected with
> it:
>
> perf stat -e irq:irq_handler_entry --filter 'irq == 49' -a \
> -- \
> fio --name=rw \
> --filename=/var/stat/testfile \
> --size=8M \
> --rw=randwrite \
> --bs=4k \
> --direct=0 \
> --fsync=1 \
> --numjobs=4 \
> --group_reporting
>
> | kernel | irq_handler_entry | fio bw |
> | ------ | ----------------- | -------- |
> | 6.12 | 155783 | 401KiB/s |
> | 6.13 | 162317 | 395KiB/s |
> | 6.14 | 168741 | 392KiB/s |
> | 6.15 | 168556 | 401KiB/s |
> | 6.16 | 166090 | 403KiB/s |
> | 6.17 | 162541 | 385KiB/s |
> | 6.18 | 157527 | 386KiB/s |
> | 6.19 | 183675 | 381KiB/s |
> | 7.0 | 190372 | 297KiB/s |
>
> The bisect targeted the large drop between 6.19 and 7.0; the smaller 6.17
> regression predates this commit and is unrelated. Reverting a339671db64b
> restores throughput to the 6.17 level (~385 KiB/s). The commit is still
> present in 7.3-rc5, and the same codegen problem reproduces there.
>
> Digging deeper, I built 7.3-rc5 with my config and GCC 16.2, with and without
> the commit, and compared the object files: 114 of them differ. As
> <vdso/unaligned.h> is included by <linux/unaligned.h>, every
> get/put_unaligned() call site depends on it transitively. GCC did not inline
> __builtin_memcpy() and turned it into a function call, e.g. in crypto/crc32c.c
> (__chksum_finup(), inlined into chksum_digest()):
>
> Without the commit:
>
> <chksum_digest>:
> str lr, [sp, #-0x4]!
> sub sp, sp, #12
> str lr, [sp, #-0x4]!
> bl 0xc0 <chksum_digest+0xc> @ imm = #-0x8
> R_ARM_CALL __gnu_mcount_nc
> ldr r0, [r0]
> str r3, [sp, #0x4]
> ldr r0, [r0, #0x20]
> bl 0xd0 <chksum_digest+0x1c> @ imm = #-0x8
> R_ARM_CALL crc32c
> mvn r2, r0
> mov r0, #0
> ldr r3, [sp, #0x4]
> lsr r12, r2, #8
> lsr r1, r2, #16
> strb r2, [r3]
> lsr r2, r2, #24
> strb r12, [r3, #0x1]
> strb r1, [r3, #0x2]
> strb r2, [r3, #0x3]
> add sp, sp, #12
> ldr pc, [sp], #4
>
> With the commit:
>
> <chksum_digest>:
> push {r4, lr}
> sub sp, sp, #8
> str lr, [sp, #-0x4]!
> bl 0x124 <chksum_digest+0xc> @ imm = #-0x8
> R_ARM_CALL __gnu_mcount_nc
> ldr r0, [r0]
> ldr r12, [pc, #0x54] @ 0x188 <chksum_digest+0x70>
> ldr r0, [r0, #0x20]
> mov r4, r3
> ldr r12, [r12]
> str r12, [sp, #0x4]
> mov r12, #0
> bl 0x144 <chksum_digest+0x2c> @ imm = #-0x8
> R_ARM_CALL crc32c
> mvn r3, r0
> mov r2, #4
> mov r0, r4
> mov r1, sp
> str r3, [sp]
> bl 0x15c <chksum_digest+0x44> @ imm = #-0x8
> R_ARM_CALL memcpy
> ldr r3, [pc, #0x20] @ 0x188 <chksum_digest+0x70>
> ldr r2, [r3]
> ldr r3, [sp, #0x4]
> eors r2, r3, r2
> mov r3, #0
> bne 0x184 <chksum_digest+0x6c> @ imm = #0x8
> mov r0, #0
> add sp, sp, #8
> pop {r4, pc}
> bl 0x184 <chksum_digest+0x6c> @ imm = #-0x8
> R_ARM_CALL __stack_chk_fail
> 188: 00 00 00 00 .word 0x00000000
> R_ARM_ABS32 __stack_chk_guard
>
> Is this an accepted trade-off? My understanding is that the kernel is always
> built with -fno-strict-aliasing, so the packed-struct type punning was well
> defined there, and the __packed annotation is what lets GCC generate valid
> code for the unaligned access.
>
> Best regards,
> Stefan
>
> --
> Pengutronix e.K. | Stefan Kerkmann |
> Steuerwalder Str. 21 | https://www.pengutronix.de/ |
> 31137 Hildesheim, Germany | Phone: +49-5121-206917-128 |
> Amtsgericht Hildesheim, HRA 2686 | Fax: +49-5121-206917-9 |
>
>
--
Pengutronix e.K. | Marc Kleine-Budde |
Embedded Linux | https://www.pengutronix.de |
Vertretung Nürnberg | Phone: +49-5121-206917-129 |
Amtsgericht Hildesheim, HRA 2686 | Fax: +49-5121-206917-9 |
Attachment:
signature.asc
Description: PGP signature