Re: [PATCH] alpha: add 128-bit shift helpers
From: Magnus Lindholm
Date: Sat Oct 10 2026 - 11:56:32 EST
Hi Matt,
On Sat, Oct 10, 2026 at 4:20 AM Matt Turner <mattst88@xxxxxxxxx> wrote:
>
> Selecting ARCH_SUPPORTS_INT128 lets generic code shift __int128 values by
> a variable count. gcc open-codes such a shift when optimizing for speed,
> but with CONFIG_CC_OPTIMIZE_FOR_SIZE=y it calls the libgcc routines
> __ashlti3, __ashrti3 and __lshrti3 instead. The kernel does not link
> against libgcc and Alpha does not provide them, so the link fails:
>
> alpha-linux-ld: lib/ubsan.o: in function `get_signed_val':
> (.text+0x1c0): undefined reference to `__ashlti3'
> (.text+0x1f8): undefined reference to `__ashrti3'
> alpha-linux-ld: kernel/time/timekeeping.o: in function `delta_to_ns_safe':
> (.text+0x2ac): undefined reference to `__lshrti3'
>
> Provide the three helpers in C, as s390 does, and export them for
> modules. They are built into the kernel proper rather than lib.a so the
> exports are present even when nothing built in references them.
>
> The 64x64 to 128-bit multiply is always expanded inline with mulq/umulh,
> so no __multi3 is needed.
>
> Fixes: fd04433a5bf3 ("alpha: select ARCH_SUPPORTS_INT128")
> Reported-by: kernel test robot <lkp@xxxxxxxxx>
> Closes: https://lore.kernel.org/oe-kbuild-all/202610090831.DSxrGQRc-lkp@xxxxxxxxx/
> Assisted-by: Claude:claude-opus-5-5
> Signed-off-by: Matt Turner <mattst88@xxxxxxxxx>
> ---
> Magnus, this is an alternative to dropping the select. It applies on top
> of for-next (496e328c6107). If you would rather drop "alpha: select
> ARCH_SUPPORTS_INT128" for now, I will resend this without the Fixes tag,
> followed by the select, as a two-patch series.
>
I'll apply this on top of my for-next, no need for a separate series
> Tested with gcc 16: defconfig plus CONFIG_CC_OPTIMIZE_FOR_SIZE=y and
> CONFIG_UBSAN=y builds and links (vmlinux and modules, W=1), and the
> helpers match the compiler's inline shifts for every count from 0 to 127
> in a userspace test. Not boot tested.
>
> arch/alpha/include/asm/asm-prototypes.h | 6 +++
> arch/alpha/lib/Makefile | 2 +
> arch/alpha/lib/tishift.c | 70 +++++++++++++++++++++++++
> 3 files changed, 78 insertions(+)
> create mode 100644 arch/alpha/lib/tishift.c
>
> diff --git a/arch/alpha/include/asm/asm-prototypes.h b/arch/alpha/include/asm/asm-prototypes.h
> index c8ae46fc2e74..672fea15032f 100644
> --- a/arch/alpha/include/asm/asm-prototypes.h
> +++ b/arch/alpha/include/asm/asm-prototypes.h
> @@ -17,3 +17,9 @@ extern void __remlu(void);
> extern void __divqu(void);
> extern void __remqu(void);
> extern unsigned long __udiv_qrnnd(unsigned long *, unsigned long, unsigned long , unsigned long);
> +
> +#ifdef CONFIG_ARCH_SUPPORTS_INT128
> +extern __int128_t __ashlti3(__int128_t a, int shift);
> +extern __int128_t __ashrti3(__int128_t a, int shift);
> +extern __int128_t __lshrti3(__int128_t a, int shift);
> +#endif
> diff --git a/arch/alpha/lib/Makefile b/arch/alpha/lib/Makefile
> index 84046e730e6d..f77d4a46f2e6 100644
> --- a/arch/alpha/lib/Makefile
> +++ b/arch/alpha/lib/Makefile
> @@ -35,6 +35,8 @@ lib-y = __divqu.o __remqu.o __divlu.o __remlu.o \
> callback_srm.o srm_puts.o srm_printk.o \
> fls.o
>
> +obj-$(CONFIG_ARCH_SUPPORTS_INT128) += tishift.o
> +
> # The division routines are built from single source, with different defines.
> AFLAGS___divqu.o = -DDIV
> AFLAGS___remqu.o = -DREM
> diff --git a/arch/alpha/lib/tishift.c b/arch/alpha/lib/tishift.c
> new file mode 100644
> index 000000000000..488e1e483d48
> --- /dev/null
> +++ b/arch/alpha/lib/tishift.c
> @@ -0,0 +1,70 @@
> +// SPDX-License-Identifier: GPL-2.0
> +/*
> + * 128-bit shift helpers.
> + *
> + * gcc open-codes a variable 128-bit shift when optimizing for speed, but
> + * calls these libgcc routines when optimizing for size.
> + */
> +
> +#include <linux/export.h>
> +#include <linux/types.h>
> +#include <asm/asm-prototypes.h>
> +
> +union ti {
> + __int128_t val;
> + struct {
> + u64 low;
> + u64 high;
> + };
> +};
> +
> +__int128_t __ashlti3(__int128_t a, int shift)
> +{
> + union ti ti = { .val = a };
> +
> + if (!shift)
> + return ti.val;
> + if (shift < 64) {
> + ti.high = (ti.high << shift) | (ti.low >> (64 - shift));
> + ti.low <<= shift;
> + } else {
> + ti.high = ti.low << (shift - 64);
> + ti.low = 0;
> + }
> + return ti.val;
> +}
> +EXPORT_SYMBOL(__ashlti3);
> +
> +__int128_t __ashrti3(__int128_t a, int shift)
> +{
> + union ti ti = { .val = a };
> +
> + if (!shift)
> + return ti.val;
> + if (shift < 64) {
> + ti.low = (ti.low >> shift) | (ti.high << (64 - shift));
> + ti.high = (s64)ti.high >> shift;
> + } else {
> + ti.low = (s64)ti.high >> (shift - 64);
> + ti.high = (s64)ti.high >> 63;
> + }
> + return ti.val;
> +}
> +EXPORT_SYMBOL(__ashrti3);
> +
> +__int128_t __lshrti3(__int128_t a, int shift)
> +{
> + union ti ti = { .val = a };
> +
> + if (!shift)
> + return ti.val;
> + if (shift < 64) {
> + ti.low = (ti.low >> shift) | (ti.high << (64 - shift));
> + ti.high >>= shift;
> + } else {
> + ti.low = ti.high >> (shift - 64);
> + ti.high = 0;
> + }
> + return ti.val;
> +}
> +EXPORT_SYMBOL(__lshrti3);
> --
> 2.55.0
>
I build-tested this on top of the current Alpha for-next branch with
CONFIG_CC_OPTIMIZE_FOR_SIZE=y and CONFIG_UBSAN=y. Both vmlinux and the
configured modules built successfully, with no unresolved references
to the new 128-bit shift helpers.
Tested-by: Magnus Lindholm <linmag7@xxxxxxxxx>
Reviewed-by: Magnus Lindholm <linmag7@xxxxxxxxx>