[PATCH bpf-next v1 2/3] bpf, x86: Skip the PROBE_MEM address range check under SMAP

From: Kumar Kartikeya Dwivedi

Date: Thu Oct 08 2026 - 22:50:23 EST


The x86 JIT guards every PROBE_MEM load with a range check that keeps user
addresses, the guard page above TASK_SIZE_MAX and the vsyscall page away
from the load, because a kernel-mode fault on those addresses would oops
under SMAP instead of reaching the load's exception table entry. The check
is nine instructions and 39 bytes (32 when the offset is zero) in front of
a load of a few bytes. For "r7 = *(u64 *)(r0 + 2200)" on a 5-level paging
kernel, with VSYSCALL_ADDR and TASK_SIZE_MAX + PAGE_SIZE - VSYSCALL_ADDR as
the two constants:

movq $-10485760, %r10
movq %rax, %r11
addq $2200, %r11
subq %r10, %r11
movabsq $72057594048413696, %r10
cmpq %r10, %r11
ja load
xorl %edi, %edi
jmp done
load:
movq 2200(%rax), %rdi
done:

The previous patch made do_user_addr_fault() resolve the exception table
entries of BPF programs for faults on user addresses when SMAP is enabled.
On such kernels, emit the bare load with its exception table entry, as the
arm64, riscv, s390 and loongarch JITs already do. All BPF programs run with
SMAP active once the CPU feature is enabled, and the feature cannot change
after boot, so checking it at JIT time is sufficient. Kernels without SMAP,
including those booted with nosmap, keep the range check.

Measured with veristat over every object of the BPF selftests and over 466
production objects from Meta's fleet, on an x86-64 guest with SMAP, with
and without this series on top of bpf-next:

programs with PROBE_MEM reduction per program
mean median max
BPF selftests 3324 91 35.0% 37.5% 74.4%
Meta production programs 1710 228 7.0% 1.8% 69.2%

No program grows. The socket and task iterators of the selftests lose about
half of their code, dump_tcp6 goes from 4386 to 2124 bytes, and the
smallest production programs lose 60% to 69%. Loads through trusted
pointers and the probe_read helpers do not use PROBE_MEM, which is why most
programs are unaffected.

Signed-off-by: Kumar Kartikeya Dwivedi <memxor@xxxxxxxxx>
---
arch/x86/net/bpf_jit_comp.c | 21 ++++++++++++++++-----
1 file changed, 16 insertions(+), 5 deletions(-)

diff --git a/arch/x86/net/bpf_jit_comp.c b/arch/x86/net/bpf_jit_comp.c
index 083fcd6cf15b..793e7cd5a5c4 100644
--- a/arch/x86/net/bpf_jit_comp.c
+++ b/arch/x86/net/bpf_jit_comp.c
@@ -2133,6 +2133,7 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
const s32 imm32 = insn->imm;
u32 dst_reg = insn->dst_reg;
u32 src_reg = insn->src_reg;
+ bool probe_mem, bounds_check;
bool accesses_stack_only;
u8 b2 = 0, b3 = 0;
u8 *start_of_ldx;
@@ -2709,6 +2710,15 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
case BPF_LDX | BPF_PROBE_MEMSX | BPF_B:
case BPF_LDX | BPF_PROBE_MEMSX | BPF_H:
case BPF_LDX | BPF_PROBE_MEMSX | BPF_W:
+ probe_mem = BPF_MODE(insn->code) == BPF_PROBE_MEM ||
+ BPF_MODE(insn->code) == BPF_PROBE_MEMSX;
+ /*
+ * With SMAP enabled, a load from a user address faults and
+ * do_user_addr_fault() resolves the exception table entry of the
+ * program, as for an unmapped kernel address, so the address range
+ * check is only needed without SMAP.
+ */
+ bounds_check = probe_mem && !cpu_feature_enabled(X86_FEATURE_SMAP);
insn_off = insn->off;
if (src_reg == BPF_REG_PARAMS) {
if (insn_off == 8) {
@@ -2724,8 +2734,7 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
*/
}

- if (BPF_MODE(insn->code) == BPF_PROBE_MEM ||
- BPF_MODE(insn->code) == BPF_PROBE_MEMSX) {
+ if (bounds_check) {
/* Conservatively check that src_reg + insn->off is a kernel address:
* src_reg + insn->off > TASK_SIZE_MAX + PAGE_SIZE
* and
@@ -2772,6 +2781,8 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
/* populate jmp_offset for JAE above to jump to start_of_ldx */
start_of_ldx = prog;
end_of_jmp[-1] = start_of_ldx - end_of_jmp;
+ } else if (probe_mem) {
+ start_of_ldx = prog;
} else if (!accesses_stack_only) {
err = emit_kasan_check(env, &prog, src_reg,
insn_off,
@@ -2785,14 +2796,14 @@ static int do_jit(struct bpf_verifier_env *env, struct bpf_prog *bpf_prog, int *
emit_ldsx(&prog, BPF_SIZE(insn->code), dst_reg, src_reg, insn_off);
else
emit_ldx(&prog, BPF_SIZE(insn->code), dst_reg, src_reg, insn_off);
- if (BPF_MODE(insn->code) == BPF_PROBE_MEM ||
- BPF_MODE(insn->code) == BPF_PROBE_MEMSX) {
+ if (probe_mem) {
struct exception_table_entry *ex;
u8 *_insn = image + proglen + (start_of_ldx - temp);
s64 delta;

/* populate jmp_offset for JMP above */
- start_of_ldx[-1] = prog - start_of_ldx;
+ if (bounds_check)
+ start_of_ldx[-1] = prog - start_of_ldx;

if (!bpf_prog->aux->extable)
break;
--
2.53.0-Meta