[PATCH bpf-next v4 05/12] bpf, tracing: load the vmlinux BTF where tracefs and bpffs requests start
HOTtoday
From: Jay Wang <hidden>
Date: 2026-10-01 22:53:40
Also in:
bpf, linux-doc, linux-kbuild, linux-kselftest, linux-modules, linux-perf-users, linux-trace-kernel, lkml, rust-for-linux, sched-ext
Subsystem:
bpf [general] (safe dynamic programs and tools), the rest, tracing · Maintainers:
Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko, Eduard Zingerman, Kumar Kartikeya Dwivedi, Linus Torvalds, Steven Rostedt, Masami Hiramatsu
Revision v4 of 4 in this series.
Revisions (4)
- v1 [diff vs current]
- v2 [diff vs current]
- v3 [diff vs current]
- v4 current
With CONFIG_DEBUG_INFO_BTF=m, bpf_get_btf_vmlinux() and bpf_find_btf_id() do not load the vmlinux BTF: loading waits for user space, and their callers were not written for that. Besides the bpf() system call and /sys/kernel/btf/vmlinux, some tracefs and bpffs requests need the BTF. Load it at the start of those, with bpf_load_btf_vmlinux(), and have the code that only uses the BTF if it happens to be there peek: - Reading a tracepoint's btf_ids file (events/*/btf_ids): built-in events use the vmlinux BTF, and module BTF is only registered once that is loaded. event_btf_ids_read() looks the BTF up under event_mutex, which the trace notifier of the btf_vmlinux module takes, so load it before taking the mutex, on the first read() of the file (not again for the one that returns EOF). - Probe events with BTF arguments ($argN, argument names, $retval, $current, typecasts): the parser loads the BTF before looking a function or struct up. It holds dyn_event_ops_mutex, which loading a module never takes: besides event creation, only dyn_event_register() takes it, from built-in init code. Loading here also keeps a $retval from silently losing its type. - The ftrace function argument printer (func-args, funcgraph-args) runs in the trace output path, which includes ftrace_dump() with interrupts disabled. It only prints the arguments if the BTF is already loaded, and never loads it: that also avoids one modprobe per trace line when the module is not installed. - bpffs: parsing delegate_* mount options (fs_context) loads the BTF when a value names commands or types, which are looked up in it; "any" and numeric masks need no BTF and do not load it. Showing the options in /proc/*/mountinfo runs under namespace_sem; it only uses the names if the BTF is already there and falls back to hex, as it already does without BTF. With CONFIG_DEBUG_INFO_BTF=y bpf_load_btf_vmlinux() is bpf_get_btf_vmlinux() and the BTF is parsed at boot, so nothing changes. Signed-off-by: Jay Wang <redacted> --- kernel/bpf/inode.c | 44 ++++++++++++++++++++++++------------- kernel/trace/trace_events.c | 10 +++++++++ kernel/trace/trace_output.c | 7 ++++++ kernel/trace/trace_probe.c | 16 ++++++++++++++ 4 files changed, 62 insertions(+), 15 deletions(-)
diff --git a/kernel/bpf/inode.c b/kernel/bpf/inode.c
index 7837968c0842..d05bbb61a593 100644
--- a/kernel/bpf/inode.c
+++ b/kernel/bpf/inode.c@@ -658,7 +658,11 @@ struct bpffs_btf_enums { const struct btf_type *attach_t; }; -static int find_bpffs_btf_enums(struct bpffs_btf_enums *info) +/* + * @load: load the vmlinux BTF if necessary (CONFIG_DEBUG_INFO_BTF=m), see + * bpf_load_btf_vmlinux(); otherwise only use it if it is already parsed. + */ +static int find_bpffs_btf_enums(struct bpffs_btf_enums *info, bool load) { struct { const struct btf_type **type;
@@ -674,7 +678,7 @@ static int find_bpffs_btf_enums(struct bpffs_btf_enums *info) memset(info, 0, sizeof(*info)); - btf = bpf_get_btf_vmlinux(); + btf = load ? bpf_load_btf_vmlinux() : bpf_peek_btf_vmlinux(); if (IS_ERR(btf)) return PTR_ERR(btf); if (!btf)
@@ -795,8 +799,11 @@ static int bpf_show_options(struct seq_file *m, struct dentry *root) opts->delegate_progs || opts->delegate_attachs) { struct bpffs_btf_enums info; - /* ignore errors, fallback to hex */ - (void)find_bpffs_btf_enums(&info); + /* + * ignore errors, fallback to hex; this runs under + * namespace_sem, so do not load the BTF from here + */ + (void)find_bpffs_btf_enums(&info, false); mask = (1ULL << __MAX_BPF_CMD) - 1; seq_print_delegate_opts(m, "delegate_cmds",
@@ -1052,35 +1059,33 @@ static int bpf_parse_param(struct fs_context *fc, struct fs_parameter *param) case OPT_DELEGATE_MAPS: case OPT_DELEGATE_PROGS: case OPT_DELEGATE_ATTACHS: { - struct bpffs_btf_enums info; - const struct btf_type *enum_t; + struct bpffs_btf_enums info = {}; + const struct btf_type **enum_t; + bool enums_tried = false; const char *enum_pfx; - u64 *delegate_msk, msk = 0; + u64 *delegate_msk, msk = 0, num; char *p, *str; int val; - /* ignore errors, fallback to hex */ - (void)find_bpffs_btf_enums(&info); - switch (opt) { case OPT_DELEGATE_CMDS: delegate_msk = &opts->delegate_cmds; - enum_t = info.cmd_t; + enum_t = &info.cmd_t; enum_pfx = "BPF_"; break; case OPT_DELEGATE_MAPS: delegate_msk = &opts->delegate_maps; - enum_t = info.map_t; + enum_t = &info.map_t; enum_pfx = "BPF_MAP_TYPE_"; break; case OPT_DELEGATE_PROGS: delegate_msk = &opts->delegate_progs; - enum_t = info.prog_t; + enum_t = &info.prog_t; enum_pfx = "BPF_PROG_TYPE_"; break; case OPT_DELEGATE_ATTACHS: delegate_msk = &opts->delegate_attachs; - enum_t = info.attach_t; + enum_t = &info.attach_t; enum_pfx = "BPF_"; break; default:
@@ -1089,9 +1094,18 @@ static int bpf_parse_param(struct fs_context *fc, struct fs_parameter *param) str = param->string; while ((p = strsep(&str, ":"))) { + /* + * Only names need the vmlinux BTF: "any" and numbers do + * not load it. Ignore errors, fallback to hex. + */ + if (strcmp(p, "any") && kstrtou64(p, 0, &num) && !enums_tried) { + (void)find_bpffs_btf_enums(&info, true); + enums_tried = true; + } + if (strcmp(p, "any") == 0) { msk |= ~0ULL; - } else if (find_btf_enum_const(info.btf, enum_t, enum_pfx, p, &val)) { + } else if (find_btf_enum_const(info.btf, *enum_t, enum_pfx, p, &val)) { msk |= 1ULL << val; } else { err = kstrtou64(p, 0, &msk);
diff --git a/kernel/trace/trace_events.c b/kernel/trace/trace_events.c
index 30c0ddf90887..c887ac6a4857 100644
--- a/kernel/trace/trace_events.c
+++ b/kernel/trace/trace_events.c@@ -23,6 +23,7 @@ #include <linux/sort.h> #include <linux/slab.h> #include <linux/delay.h> +#include <linux/bpf.h> #include <linux/btf.h> #include <trace/events/sched.h>
@@ -2245,6 +2246,15 @@ event_btf_ids_read(struct file *filp, char __user *ubuf, size_t cnt, loff_t *ppo char buf[128]; int len; + /* + * Built-in events use the vmlinux BTF, and with CONFIG_DEBUG_INFO_BTF=m + * module BTF is only registered once that is loaded. Loading it loads + * a module, whose trace notifier takes event_mutex: load it before + * taking that, and only for the first read, not again for the EOF one. + */ + if (!*ppos) + bpf_load_btf_vmlinux(); + /* Module unload could free call->class and ids[] mid-read. */ scoped_guard(mutex, &event_mutex) { file = event_file_file(filp);
diff --git a/kernel/trace/trace_output.c b/kernel/trace/trace_output.c
index a5ad76175d10..1f346e524ee6 100644
--- a/kernel/trace/trace_output.c
+++ b/kernel/trace/trace_output.c@@ -739,6 +739,13 @@ void print_function_args(struct trace_seq *s, unsigned long *args, if (lookup_symbol_name(func, name)) goto out; + /* + * This can run with interrupts disabled (ftrace_dump()): only use + * the vmlinux BTF if it is parsed, never load it from here. + */ + if (IS_ERR_OR_NULL(bpf_peek_btf_vmlinux())) + goto out; + /* TODO: Pass module name here too */ t = btf_find_func_proto(name, &btf); if (IS_ERR_OR_NULL(t))
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 804442b2f7d2..d53ee1ef820c 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c@@ -531,6 +531,19 @@ static const char *fetch_type_from_btf_type(struct btf *btf, return NULL; } +/* + * Arguments described by BTF need the vmlinux BTF. With + * CONFIG_DEBUG_INFO_BTF=m it may not be loaded yet, so load it before looking + * anything up. Parsing holds dyn_event_ops_mutex, which loading a module + * never takes: besides event creation, only dyn_event_register() takes it, + * from built-in init code. + */ +static void trace_probe_load_btf(void) +{ + lockdep_assert_held(&dyn_event_ops_mutex); + bpf_load_btf_vmlinux(); +} + static int query_btf_context(struct traceprobe_parse_context *ctx) { const struct btf_param *param;
@@ -544,6 +557,7 @@ static int query_btf_context(struct traceprobe_parse_context *ctx) if (!ctx->funcname) return -EINVAL; + trace_probe_load_btf(); type = btf_find_func_proto(ctx->funcname, &btf); if (!type) return -ENOENT;
@@ -762,6 +776,7 @@ static int parse_btf_arg(char *varname, if (!strcmp(varname, "$current")) { code->op = FETCH_OP_CURRENT; /* If no typecast is specified for $current, use task_struct by default */ + trace_probe_load_btf(); ret = bpf_find_btf_id("task_struct", BTF_KIND_STRUCT, &ctx->struct_btf); if (ret < 0) { trace_probe_log_err(ctx->offset, NO_BTF_ENTRY);
@@ -890,6 +905,7 @@ static int query_btf_struct(const char *sname, struct traceprobe_parse_context * ctx->struct_btf = NULL; } + trace_probe_load_btf(); id = bpf_find_btf_id(sname, BTF_KIND_STRUCT, &btf); if (id < 0) return id;
--
2.47.3