[PATCH bpf-next v4 05/12] bpf, tracing: load the vmlinux BTF where tracefs and bpffs requests start

From: Jay Wang

Date: Thu Oct 01 2026 - 18:58:22 EST


With CONFIG_DEBUG_INFO_BTF=m, bpf_get_btf_vmlinux() and
bpf_find_btf_id() do not load the vmlinux BTF: loading waits for user
space, and their callers were not written for that. Besides the bpf()
system call and /sys/kernel/btf/vmlinux, some tracefs and bpffs requests
need the BTF. Load it at the start of those, with
bpf_load_btf_vmlinux(), and have the code that only uses the BTF if it
happens to be there peek:

- Reading a tracepoint's btf_ids file (events/*/btf_ids): built-in
events use the vmlinux BTF, and module BTF is only registered once
that is loaded. event_btf_ids_read() looks the BTF up under
event_mutex, which the trace notifier of the btf_vmlinux module
takes, so load it before taking the mutex, on the first read() of
the file (not again for the one that returns EOF).
- Probe events with BTF arguments ($argN, argument names, $retval,
$current, typecasts): the parser loads the BTF before looking a
function or struct up. It holds dyn_event_ops_mutex, which loading
a module never takes: besides event creation, only
dyn_event_register() takes it, from built-in init code. Loading
here also keeps a $retval from silently losing its type.
- The ftrace function argument printer (func-args, funcgraph-args)
runs in the trace output path, which includes ftrace_dump() with
interrupts disabled. It only prints the arguments if the BTF is
already loaded, and never loads it: that also avoids one modprobe
per trace line when the module is not installed.
- bpffs: parsing delegate_* mount options (fs_context) loads the BTF
when a value names commands or types, which are looked up in it;
"any" and numeric masks need no BTF and do not load it. Showing the
options in /proc/*/mountinfo runs under namespace_sem; it only uses
the names if the BTF is already there and falls back to hex, as it
already does without BTF.

With CONFIG_DEBUG_INFO_BTF=y bpf_load_btf_vmlinux() is
bpf_get_btf_vmlinux() and the BTF is parsed at boot, so nothing changes.

Signed-off-by: Jay Wang <wanjay@xxxxxxxxxx>
---
kernel/bpf/inode.c | 44 ++++++++++++++++++++++++-------------
kernel/trace/trace_events.c | 10 +++++++++
kernel/trace/trace_output.c | 7 ++++++
kernel/trace/trace_probe.c | 16 ++++++++++++++
4 files changed, 62 insertions(+), 15 deletions(-)

diff --git a/kernel/bpf/inode.c b/kernel/bpf/inode.c
index 7837968c0842..d05bbb61a593 100644
--- a/kernel/bpf/inode.c
+++ b/kernel/bpf/inode.c
@@ -658,7 +658,11 @@ struct bpffs_btf_enums {
const struct btf_type *attach_t;
};

-static int find_bpffs_btf_enums(struct bpffs_btf_enums *info)
+/*
+ * @load: load the vmlinux BTF if necessary (CONFIG_DEBUG_INFO_BTF=m), see
+ * bpf_load_btf_vmlinux(); otherwise only use it if it is already parsed.
+ */
+static int find_bpffs_btf_enums(struct bpffs_btf_enums *info, bool load)
{
struct {
const struct btf_type **type;
@@ -674,7 +678,7 @@ static int find_bpffs_btf_enums(struct bpffs_btf_enums *info)

memset(info, 0, sizeof(*info));

- btf = bpf_get_btf_vmlinux();
+ btf = load ? bpf_load_btf_vmlinux() : bpf_peek_btf_vmlinux();
if (IS_ERR(btf))
return PTR_ERR(btf);
if (!btf)
@@ -795,8 +799,11 @@ static int bpf_show_options(struct seq_file *m, struct dentry *root)
opts->delegate_progs || opts->delegate_attachs) {
struct bpffs_btf_enums info;

- /* ignore errors, fallback to hex */
- (void)find_bpffs_btf_enums(&info);
+ /*
+ * ignore errors, fallback to hex; this runs under
+ * namespace_sem, so do not load the BTF from here
+ */
+ (void)find_bpffs_btf_enums(&info, false);

mask = (1ULL << __MAX_BPF_CMD) - 1;
seq_print_delegate_opts(m, "delegate_cmds",
@@ -1052,35 +1059,33 @@ static int bpf_parse_param(struct fs_context *fc, struct fs_parameter *param)
case OPT_DELEGATE_MAPS:
case OPT_DELEGATE_PROGS:
case OPT_DELEGATE_ATTACHS: {
- struct bpffs_btf_enums info;
- const struct btf_type *enum_t;
+ struct bpffs_btf_enums info = {};
+ const struct btf_type **enum_t;
+ bool enums_tried = false;
const char *enum_pfx;
- u64 *delegate_msk, msk = 0;
+ u64 *delegate_msk, msk = 0, num;
char *p, *str;
int val;

- /* ignore errors, fallback to hex */
- (void)find_bpffs_btf_enums(&info);
-
switch (opt) {
case OPT_DELEGATE_CMDS:
delegate_msk = &opts->delegate_cmds;
- enum_t = info.cmd_t;
+ enum_t = &info.cmd_t;
enum_pfx = "BPF_";
break;
case OPT_DELEGATE_MAPS:
delegate_msk = &opts->delegate_maps;
- enum_t = info.map_t;
+ enum_t = &info.map_t;
enum_pfx = "BPF_MAP_TYPE_";
break;
case OPT_DELEGATE_PROGS:
delegate_msk = &opts->delegate_progs;
- enum_t = info.prog_t;
+ enum_t = &info.prog_t;
enum_pfx = "BPF_PROG_TYPE_";
break;
case OPT_DELEGATE_ATTACHS:
delegate_msk = &opts->delegate_attachs;
- enum_t = info.attach_t;
+ enum_t = &info.attach_t;
enum_pfx = "BPF_";
break;
default:
@@ -1089,9 +1094,18 @@ static int bpf_parse_param(struct fs_context *fc, struct fs_parameter *param)

str = param->string;
while ((p = strsep(&str, ":"))) {
+ /*
+ * Only names need the vmlinux BTF: "any" and numbers do
+ * not load it. Ignore errors, fallback to hex.
+ */
+ if (strcmp(p, "any") && kstrtou64(p, 0, &num) && !enums_tried) {
+ (void)find_bpffs_btf_enums(&info, true);
+ enums_tried = true;
+ }
+
if (strcmp(p, "any") == 0) {
msk |= ~0ULL;
- } else if (find_btf_enum_const(info.btf, enum_t, enum_pfx, p, &val)) {
+ } else if (find_btf_enum_const(info.btf, *enum_t, enum_pfx, p, &val)) {
msk |= 1ULL << val;
} else {
err = kstrtou64(p, 0, &msk);
diff --git a/kernel/trace/trace_events.c b/kernel/trace/trace_events.c
index 30c0ddf90887..c887ac6a4857 100644
--- a/kernel/trace/trace_events.c
+++ b/kernel/trace/trace_events.c
@@ -23,6 +23,7 @@
#include <linux/sort.h>
#include <linux/slab.h>
#include <linux/delay.h>
+#include <linux/bpf.h>
#include <linux/btf.h>

#include <trace/events/sched.h>
@@ -2245,6 +2246,15 @@ event_btf_ids_read(struct file *filp, char __user *ubuf, size_t cnt, loff_t *ppo
char buf[128];
int len;

+ /*
+ * Built-in events use the vmlinux BTF, and with CONFIG_DEBUG_INFO_BTF=m
+ * module BTF is only registered once that is loaded. Loading it loads
+ * a module, whose trace notifier takes event_mutex: load it before
+ * taking that, and only for the first read, not again for the EOF one.
+ */
+ if (!*ppos)
+ bpf_load_btf_vmlinux();
+
/* Module unload could free call->class and ids[] mid-read. */
scoped_guard(mutex, &event_mutex) {
file = event_file_file(filp);
diff --git a/kernel/trace/trace_output.c b/kernel/trace/trace_output.c
index a5ad76175d10..1f346e524ee6 100644
--- a/kernel/trace/trace_output.c
+++ b/kernel/trace/trace_output.c
@@ -739,6 +739,13 @@ void print_function_args(struct trace_seq *s, unsigned long *args,
if (lookup_symbol_name(func, name))
goto out;

+ /*
+ * This can run with interrupts disabled (ftrace_dump()): only use
+ * the vmlinux BTF if it is parsed, never load it from here.
+ */
+ if (IS_ERR_OR_NULL(bpf_peek_btf_vmlinux()))
+ goto out;
+
/* TODO: Pass module name here too */
t = btf_find_func_proto(name, &btf);
if (IS_ERR_OR_NULL(t))
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 804442b2f7d2..d53ee1ef820c 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -531,6 +531,19 @@ static const char *fetch_type_from_btf_type(struct btf *btf,
return NULL;
}

+/*
+ * Arguments described by BTF need the vmlinux BTF. With
+ * CONFIG_DEBUG_INFO_BTF=m it may not be loaded yet, so load it before looking
+ * anything up. Parsing holds dyn_event_ops_mutex, which loading a module
+ * never takes: besides event creation, only dyn_event_register() takes it,
+ * from built-in init code.
+ */
+static void trace_probe_load_btf(void)
+{
+ lockdep_assert_held(&dyn_event_ops_mutex);
+ bpf_load_btf_vmlinux();
+}
+
static int query_btf_context(struct traceprobe_parse_context *ctx)
{
const struct btf_param *param;
@@ -544,6 +557,7 @@ static int query_btf_context(struct traceprobe_parse_context *ctx)
if (!ctx->funcname)
return -EINVAL;

+ trace_probe_load_btf();
type = btf_find_func_proto(ctx->funcname, &btf);
if (!type)
return -ENOENT;
@@ -762,6 +776,7 @@ static int parse_btf_arg(char *varname,
if (!strcmp(varname, "$current")) {
code->op = FETCH_OP_CURRENT;
/* If no typecast is specified for $current, use task_struct by default */
+ trace_probe_load_btf();
ret = bpf_find_btf_id("task_struct", BTF_KIND_STRUCT, &ctx->struct_btf);
if (ret < 0) {
trace_probe_log_err(ctx->offset, NO_BTF_ENTRY);
@@ -890,6 +905,7 @@ static int query_btf_struct(const char *sname, struct traceprobe_parse_context *
ctx->struct_btf = NULL;
}

+ trace_probe_load_btf();
id = bpf_find_btf_id(sname, BTF_KIND_STRUCT, &btf);
if (id < 0)
return id;
--
2.47.3