Thread (16 messages) 16 messages, 2 authors, 1h ago

[PATCH bpf-next v4 06/12] bpf: defer vmlinux kfunc and struct_ops registrations

HOTtoday

From: Jay Wang <hidden>
Date: 2026-10-01 22:53:56
Also in: bpf, linux-doc, linux-kbuild, linux-kselftest, linux-modules, linux-perf-users, linux-trace-kernel, lkml, rust-for-linux, sched-ext
Subsystem: bpf [general] (safe dynamic programs and tools), the rest · Maintainers: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko, Eduard Zingerman, Kumar Kartikeya Dwivedi, Linus Torvalds

Revision v4 of 4 in this series.

Revisions (4)
  1. v1 [diff vs current]
  2. v2 [diff vs current]
  3. v3 [diff vs current]
  4. v4 current
With CONFIG_DEBUG_INFO_BTF=m the vmlinux BTF is loaded on first use.  For
that to save anything, nothing may pull it in at boot.  The verifier no
longer does since the previous patches, but register_btf_kfunc_id_set(),
register_btf_id_dtor_kfuncs() and register_bpf_struct_ops() for vmlinux
run from initcalls and need the parsed BTF.

Queue them instead (btf_defer_reg()) and apply them in
btf_parse_vmlinux(), before the BTF is published, so that no program can
see a vmlinux BTF without its kfuncs and struct_ops, not even by id: the
id is reserved before the queue is applied, so that nothing can fail once
it is, and installed afterwards (btf_reserve_id(), btf_install_id()).
Applying a struct_ops runs its ->init(), which registers the kfunc sets
of its hook; those land back on the queue, so it is drained in a loop
until a pass adds nothing, and only then are new registrations applied
directly.  The dtor arrays are copied: every caller in the tree builds
them on the stack of its initcall.

The queue has its own lock, btf_vmlinux_regs_mutex: it is drained under
btf_vmlinux_lock, and with its own lock btf_module_mutex, which the
module notifier takes, is never taken under btf_vmlinux_lock, so the two
stay unordered.

A parse failure other than -ENOMEM is remembered by
bpf_load_btf_vmlinux(), so nothing would ever apply the queue: it is then
freed and closed, and later registrations fail as they do with =y.

Module registrations are not queued yet; the next patch does that
together with deferring the module BTF itself.  With =y the BTF is
present from boot and nothing is queued.  Nothing here is reachable
until the Kconfig symbol becomes a tristate.

Signed-off-by: Jay Wang <redacted>
---
 kernel/bpf/btf.c | 271 ++++++++++++++++++++++++++++++++++++++++++++++-
 1 file changed, 268 insertions(+), 3 deletions(-)
diff --git a/kernel/bpf/btf.c b/kernel/bpf/btf.c
index 2aa9d4b3f438..96241dc62dc3 100644
--- a/kernel/bpf/btf.c
+++ b/kernel/bpf/btf.c
@@ -1786,13 +1786,13 @@ static int btf_add_type(struct btf_verifier_env *env, struct btf_type *t)
 	return 0;
 }
 
-static int btf_alloc_id(struct btf *btf)
+static int __btf_alloc_id(struct btf *btf, struct btf *entry)
 {
 	int id;
 
 	idr_preload(GFP_KERNEL);
 	spin_lock_bh(&btf_idr_lock);
-	id = idr_alloc_cyclic(&btf_idr, btf, 1, INT_MAX, GFP_ATOMIC);
+	id = idr_alloc_cyclic(&btf_idr, entry, 1, INT_MAX, GFP_ATOMIC);
 	if (id > 0)
 		btf->id = id;
 	spin_unlock_bh(&btf_idr_lock);
@@ -1804,6 +1804,29 @@ static int btf_alloc_id(struct btf *btf)
 	return id > 0 ? 0 : id;
 }
 
+static int btf_alloc_id(struct btf *btf)
+{
+	return __btf_alloc_id(btf, btf);
+}
+
+/*
+ * A BTF that gets registrations applied after it is parsed, with
+ * CONFIG_DEBUG_INFO_BTF=m, has its id reserved before, so that nothing can
+ * fail once they are applied, and installed after, so that nobody finds it
+ * by id in the meantime (btf_get_fd_by_id() and idr_get_next() skip NULL).
+ */
+static int btf_reserve_id(struct btf *btf)
+{
+	return __btf_alloc_id(btf, NULL);
+}
+
+static void btf_install_id(struct btf *btf)
+{
+	spin_lock_bh(&btf_idr_lock);
+	idr_replace(&btf_idr, btf, btf->id);
+	spin_unlock_bh(&btf_idr_lock);
+}
+
 static void btf_free_id(struct btf *btf)
 {
 	unsigned long flags;
@@ -6960,6 +6983,9 @@ static struct btf *btf_parse_base(struct btf_verifier_env *env, const char *name
 	return ERR_PTR(err);
 }
 
+static void btf_apply_deferred_vmlinux_regs(struct btf *btf);
+static void btf_drop_deferred_vmlinux_regs(void);
+
 struct btf *btf_parse_vmlinux(void)
 {
 	struct btf_verifier_env *env = NULL;
@@ -6986,13 +7012,29 @@ struct btf *btf_parse_vmlinux(void)
 
 	/* btf_parse_vmlinux() runs under btf_vmlinux_lock */
 	bpf_ctx_convert.t = btf_type_by_id(btf, bpf_ctx_convert_btf_id[0]);
-	err = btf_alloc_id(btf);
+	err = btf_reserve_id(btf);
 	if (err) {
 		bpf_ctx_convert.t = NULL;
 		btf_free(btf);
 		btf = ERR_PTR(err);
+		goto err_out;
 	}
+
+	/*
+	 * With CONFIG_DEBUG_INFO_BTF=m, kfunc, dtor kfunc and struct_ops
+	 * registrations for vmlinux made before the BTF was available were
+	 * queued; apply them now, before the BTF becomes visible to anyone,
+	 * by id or through bpf_get_btf_vmlinux().
+	 */
+	btf_apply_deferred_vmlinux_regs(btf);
+	btf_install_id(btf);
 err_out:
+	/*
+	 * Any failure but -ENOMEM is remembered by bpf_load_btf_vmlinux(), so
+	 * the queued registrations would never be applied.
+	 */
+	if (IS_ERR(btf) && PTR_ERR(btf) != -ENOMEM)
+		btf_drop_deferred_vmlinux_regs();
 	btf_verifier_env_free(env);
 	return btf;
 }
@@ -9110,6 +9152,33 @@ enum {
 #define BTF_MODULE_NOTIFIER 1
 #endif
 
+/*
+ * CONFIG_DEBUG_INFO_BTF=m: a kfunc, dtor kfunc or struct_ops registration
+ * made while the BTF it applies to is not available yet.  Kept until the BTF
+ * arrives, see btf_defer_reg().
+ */
+enum btf_deferred_reg_kind {
+	BTF_DEFERRED_KFUNC_SET,
+	BTF_DEFERRED_DTOR_KFUNCS,
+	BTF_DEFERRED_STRUCT_OPS,
+};
+
+struct btf_deferred_reg {
+	struct list_head list;
+	enum btf_deferred_reg_kind kind;
+	union {
+		struct {
+			enum btf_kfunc_hook hook;
+			const struct btf_kfunc_id_set *kset;
+		} kfunc;
+		struct {
+			const struct btf_id_dtor_kfunc *dtors;
+			u32 cnt;
+		} dtor;
+		struct bpf_struct_ops *st_ops;
+	};
+};
+
 #ifdef BTF_MODULE_NOTIFIER
 struct btf_module {
 	struct list_head list;
@@ -9905,12 +9974,22 @@ static int btf_kfunc_id_set_add(struct btf *btf, enum btf_kfunc_hook hook,
 	return btf_populate_kfunc_set(btf, hook, kset);
 }
 
+static int btf_defer_reg(struct module *owner, const struct btf_deferred_reg *tmpl);
+
 static int __register_btf_kfunc_id_set(enum btf_kfunc_hook hook,
 				       const struct btf_kfunc_id_set *kset)
 {
+	struct btf_deferred_reg tmpl = {
+		.kind = BTF_DEFERRED_KFUNC_SET,
+		.kfunc = { .hook = hook, .kset = kset },
+	};
 	struct btf *btf;
 	int ret;
 
+	ret = btf_defer_reg(kset->owner, &tmpl);
+	if (ret)
+		return ret > 0 ? 0 : ret;
+
 	btf = btf_get_module_btf(kset->owner);
 	if (!btf)
 		return check_btf_kconfigs(kset->owner, "kfunc");
@@ -10079,9 +10158,17 @@ static int btf_dtor_kfuncs_add(struct btf *btf, const struct btf_id_dtor_kfunc *
 int register_btf_id_dtor_kfuncs(const struct btf_id_dtor_kfunc *dtors, u32 add_cnt,
 				struct module *owner)
 {
+	struct btf_deferred_reg tmpl = {
+		.kind = BTF_DEFERRED_DTOR_KFUNCS,
+		.dtor = { .dtors = dtors, .cnt = add_cnt },
+	};
 	struct btf *btf;
 	int ret;
 
+	ret = btf_defer_reg(owner, &tmpl);
+	if (ret)
+		return ret > 0 ? 0 : ret;
+
 	btf = btf_get_module_btf(owner);
 	if (!btf)
 		return check_btf_kconfigs(owner, "dtor kfuncs");
@@ -10751,9 +10838,17 @@ static int btf_struct_ops_register(struct btf *btf, struct bpf_struct_ops *st_op
 
 int __register_bpf_struct_ops(struct bpf_struct_ops *st_ops)
 {
+	struct btf_deferred_reg tmpl = {
+		.kind = BTF_DEFERRED_STRUCT_OPS,
+		.st_ops = st_ops,
+	};
 	struct btf *btf;
 	int err;
 
+	err = btf_defer_reg(st_ops->owner, &tmpl);
+	if (err)
+		return err > 0 ? 0 : err;
+
 	btf = btf_get_module_btf(st_ops->owner);
 	if (!btf)
 		return check_btf_kconfigs(st_ops->owner, "struct_ops");
@@ -10765,8 +10860,178 @@ int __register_bpf_struct_ops(struct bpf_struct_ops *st_ops)
 	return err;
 }
 EXPORT_SYMBOL_GPL(__register_bpf_struct_ops);
+#elif defined(BTF_MODULE_NOTIFIER)
+static int btf_struct_ops_register(struct btf *btf, struct bpf_struct_ops *st_ops)
+{
+	return -EOPNOTSUPP;
+}
 #endif
 
+/*
+ * CONFIG_DEBUG_INFO_BTF=m: registrations for vmlinux made before its BTF is
+ * available wait in btf_vmlinux_deferred_regs until btf_parse_vmlinux()
+ * applies them.
+ */
+#ifdef BTF_MODULE_NOTIFIER
+/*
+ * The queue has its own lock: it is drained under btf_vmlinux_lock, and
+ * with its own lock btf_module_mutex, which the module notifier takes, is
+ * never taken under btf_vmlinux_lock, so the two stay unordered.
+ */
+static DEFINE_MUTEX(btf_vmlinux_regs_mutex);
+static LIST_HEAD(btf_vmlinux_deferred_regs);
+/* Set when the vmlinux BTF is parsed; new registrations apply directly */
+static bool btf_vmlinux_regs_closed;
+
+/*
+ * Queue @tmpl if the BTF for @owner is not available yet.  Returns 1 if the
+ * registration was queued and is to be considered done, 0 if the caller has
+ * to apply it, or -ENOMEM.  Only vmlinux registrations are queued so far.
+ */
+static int btf_defer_reg(struct module *owner, const struct btf_deferred_reg *tmpl)
+{
+	struct list_head *head = NULL;
+	struct btf_deferred_reg *reg;
+
+	if (!IS_MODULE(CONFIG_DEBUG_INFO_BTF) || owner)
+		return 0;
+
+	guard(mutex)(&btf_vmlinux_regs_mutex);
+	if (!btf_vmlinux_regs_closed)
+		head = &btf_vmlinux_deferred_regs;
+	if (!head)
+		return 0;
+
+	reg = kmemdup(tmpl, sizeof(*reg), GFP_KERNEL);
+	if (!reg)
+		return -ENOMEM;
+	/*
+	 * kfunc id sets and struct_ops are static data of their owner, but
+	 * the dtor arrays are commonly built on the stack of the initcall.
+	 */
+	if (reg->kind == BTF_DEFERRED_DTOR_KFUNCS) {
+		reg->dtor.dtors = kmemdup_array(tmpl->dtor.dtors, tmpl->dtor.cnt,
+						sizeof(*tmpl->dtor.dtors), GFP_KERNEL);
+		if (!reg->dtor.dtors) {
+			kfree(reg);
+			return -ENOMEM;
+		}
+	}
+	list_add_tail(&reg->list, head);
+	return 1;
+}
+
+static void btf_free_deferred_reg(struct btf_deferred_reg *reg)
+{
+	if (reg->kind == BTF_DEFERRED_DTOR_KFUNCS)
+		kfree(reg->dtor.dtors);
+	kfree(reg);
+}
+
+static const char *btf_deferred_reg_name(const struct btf_deferred_reg *reg)
+{
+	switch (reg->kind) {
+	case BTF_DEFERRED_KFUNC_SET:	return "kfunc set";
+	case BTF_DEFERRED_DTOR_KFUNCS:	return "dtor kfuncs";
+	case BTF_DEFERRED_STRUCT_OPS:	return "struct_ops";
+	}
+	return "?";
+}
+
+static int btf_apply_deferred_reg(struct btf *btf, const struct btf_deferred_reg *reg)
+{
+	switch (reg->kind) {
+	case BTF_DEFERRED_KFUNC_SET:
+		return btf_kfunc_id_set_add(btf, reg->kfunc.hook, reg->kfunc.kset);
+	case BTF_DEFERRED_DTOR_KFUNCS:
+		return btf_dtor_kfuncs_add(btf, reg->dtor.dtors, reg->dtor.cnt);
+	case BTF_DEFERRED_STRUCT_OPS:
+		return btf_struct_ops_register(btf, reg->st_ops);
+	}
+	return -EINVAL;
+}
+
+/* Apply and free the registrations in @regs to @btf. */
+static void btf_apply_deferred_regs(struct btf *btf, struct list_head *regs)
+{
+	struct btf_deferred_reg *reg, *tmp;
+	int err;
+
+	list_for_each_entry_safe(reg, tmp, regs, list) {
+		err = btf_apply_deferred_reg(btf, reg);
+		if (err)
+			pr_warn("failed to register deferred %s for [%s] BTF: %d\n",
+				btf_deferred_reg_name(reg), btf->name, err);
+		list_del(&reg->list);
+		btf_free_deferred_reg(reg);
+	}
+}
+
+/*
+ * The vmlinux BTF has just been parsed; apply the registrations that waited
+ * for it.  Runs under btf_vmlinux_lock, before @btf is published, so nothing
+ * can observe a vmlinux BTF without its kfuncs and struct_ops.
+ *
+ * Applying a registration can queue further ones: a struct_ops ->init()
+ * registers the kfuncs of its hook.  Those cannot be applied directly, the
+ * BTF is only published once we are done, so the queue stays open until a
+ * pass applies nothing new, and only then are registrations applied directly.
+ */
+static void btf_apply_deferred_vmlinux_regs(struct btf *btf)
+{
+	LIST_HEAD(regs);
+
+	if (!IS_MODULE(CONFIG_DEBUG_INFO_BTF))
+		return;
+
+	mutex_lock(&btf_vmlinux_regs_mutex);
+	while (!list_empty(&btf_vmlinux_deferred_regs)) {
+		list_splice_init(&btf_vmlinux_deferred_regs, &regs);
+		mutex_unlock(&btf_vmlinux_regs_mutex);
+		btf_apply_deferred_regs(btf, &regs);
+		mutex_lock(&btf_vmlinux_regs_mutex);
+	}
+	btf_vmlinux_regs_closed = true;
+	mutex_unlock(&btf_vmlinux_regs_mutex);
+}
+
+/*
+ * The vmlinux BTF will not become available: free the queued registrations
+ * and close the queue, so that later ones fail as they do with =y.
+ */
+static void btf_drop_deferred_vmlinux_regs(void)
+{
+	struct btf_deferred_reg *reg, *tmp;
+	LIST_HEAD(regs);
+
+	if (!IS_MODULE(CONFIG_DEBUG_INFO_BTF))
+		return;
+
+	mutex_lock(&btf_vmlinux_regs_mutex);
+	list_splice_init(&btf_vmlinux_deferred_regs, &regs);
+	btf_vmlinux_regs_closed = true;
+	mutex_unlock(&btf_vmlinux_regs_mutex);
+
+	list_for_each_entry_safe(reg, tmp, &regs, list) {
+		list_del(&reg->list);
+		btf_free_deferred_reg(reg);
+	}
+}
+#else
+static int btf_defer_reg(struct module *owner, const struct btf_deferred_reg *tmpl)
+{
+	return 0;
+}
+
+static void btf_apply_deferred_vmlinux_regs(struct btf *btf)
+{
+}
+
+static void btf_drop_deferred_vmlinux_regs(void)
+{
+}
+#endif /* BTF_MODULE_NOTIFIER */
+
 bool btf_param_match_suffix(const struct btf *btf,
 			    const struct btf_param *arg,
 			    const char *suffix)
-- 
2.47.3
Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help