Module BTF is split BTF against the vmlinux BTF and is parsed in the
module notifier.  With CONFIG_DEBUG_INFO_BTF=m the vmlinux BTF may not be
loaded yet when a module loads, and the notifier cannot load btf_vmlinux
(that would nest a module load in a module load).

So a module loaded before the vmlinux BTF keeps a copy of its .BTF and
.BTF.base and gets a list entry with btf == NULL; its kfunc, dtor kfunc
and struct_ops registrations wait on that entry.  Without a .BTF.base
the data is final and is exposed in /sys/kernel/btf right away (the raw
bytes need no parsing); with one, parsing relocates the data in place,
so its file is created once parsed, as with =y.  The next patch creates
that file earlier.

When the vmlinux BTF arrives, btf_parse_deferred_modules() parses the
kept copies (the copy is the one btf_parse_module() makes anyway, so an
existing sysfs file keeps pointing at valid data), applies the waiting
registrations and only then publishes the BTF, so nobody sees a module
BTF without its kfuncs; its id is reserved before the registrations are
applied and installed after.  Applying walks the module list
(btf_check_kfunc_name()) and so happens with btf_module_mutex dropped
and the module pinned, and it loops, as for vmlinux, because a struct_ops
->init() registers kfuncs in turn.  A module that is still initializing
when the vmlinux BTF arrives is only published; its init is still
queueing registrations, later ones queue behind them so that the order
is kept, and MODULE_STATE_LIVE applies them once init is done, before
the module counts as live, as with =y.  This also keeps a module whose
init fails from being touched after it is freed.

bpf_load_btf_vmlinux() returns only once the kept module BTF is
registered, also to callers that come while another one is doing it.
Until then a search of the module BTFs by name (bpf_find_btf_id(), and
in-kernel CO-RE candidate search) may miss a module that is loaded;
rather than taking a missing type for a local one, or caching the
candidates, it counts a vmlinux BTF miss and fails, and bpf() runs the
command again after waiting.

A module whose BTF cannot be kept for lack of memory fails to load, as
with =y when its BTF fails to parse, unless
CONFIG_MODULE_ALLOW_BTF_MISMATCH.

A module whose BTF turns out to mismatch at that point is already
running and keeps running without BTF, with a warning; its entry stays,
dead, until the module goes, and keeps the raw data a sysfs file may
serve.  With =y such a module would have been refused at load time
unless CONFIG_MODULE_ALLOW_BTF_MISMATCH; that check only applies to
modules loaded after the vmlinux BTF.  If the vmlinux BTF itself fails
to parse for good, every kept module becomes such a dead entry.

Walkers of the module BTF list skip entries whose BTF is not parsed yet.
With =y the BTF is present from boot and the notifier takes the existing
path.  Still nothing is reachable until the Kconfig symbol becomes a
tristate.

Signed-off-by: Jay Wang <[email protected]>
---
 include/linux/bpf.h   |   5 +
 include/linux/btf.h   |  10 ++
 kernel/bpf/btf.c      | 333 +++++++++++++++++++++++++++++++++++++++---
 kernel/bpf/verifier.c |  39 +++--
 4 files changed, 359 insertions(+), 28 deletions(-)

diff --git a/include/linux/bpf.h b/include/linux/bpf.h
index d812dbc683ae..f6634600467e 100644
--- a/include/linux/bpf.h
+++ b/include/linux/bpf.h
@@ -3187,11 +3187,16 @@ struct btf *bpf_peek_btf_vmlinux(void);
 struct btf *bpf_load_btf_vmlinux(void);
 #if IS_MODULE(CONFIG_DEBUG_INFO_BTF)
 unsigned int bpf_btf_vmlinux_misses(void);
+void bpf_btf_vmlinux_miss(void);
 #else
 static inline unsigned int bpf_btf_vmlinux_misses(void)
 {
        return 0;
 }
+
+static inline void bpf_btf_vmlinux_miss(void)
+{
+}
 #endif
 
 /* Map specifics */
diff --git a/include/linux/btf.h b/include/linux/btf.h
index 81e6c65fe5f6..b546531dd6f4 100644
--- a/include/linux/btf.h
+++ b/include/linux/btf.h
@@ -603,6 +603,16 @@ const char *btf_name_by_offset(const struct btf *btf, u32 
offset);
 const char *btf_str_by_offset(const struct btf *btf, u32 offset);
 struct btf *btf_parse_vmlinux(void);
 void *btf_vmlinux_data(u32 *size, bool load);
+#if IS_MODULE(CONFIG_DEBUG_INFO_BTF)
+void btf_parse_deferred_modules(void);
+bool btf_deferred_modules_pending(void);
+#else
+static inline void btf_parse_deferred_modules(void) {}
+static inline bool btf_deferred_modules_pending(void)
+{
+       return false;
+}
+#endif
 struct btf *bpf_prog_get_target_btf(const struct bpf_prog *prog);
 u32 *btf_kfunc_flags(const struct btf *btf, u32 kfunc_btf_id, const struct 
bpf_prog *prog);
 int btf_kfunc_check_flag(const struct btf *btf, u32 kfunc_btf_id, u32 flag);
diff --git a/kernel/bpf/btf.c b/kernel/bpf/btf.c
index 96241dc62dc3..4d51fb212218 100644
--- a/kernel/bpf/btf.c
+++ b/kernel/bpf/btf.c
@@ -719,6 +719,12 @@ s32 bpf_find_btf_id(const char *name, u32 kind, struct btf 
**btf_p)
                return ret;
        }
 
+       /* the modules loaded before the vmlinux BTF may not be registered yet 
*/
+       if (btf_deferred_modules_pending()) {
+               bpf_btf_vmlinux_miss();
+               return -EINVAL;
+       }
+
        /* If name is not found in vmlinux's BTF then search in module's BTFs */
        spin_lock_bh(&btf_idr_lock);
        idr_for_each_entry(&btf_idr, btf, id) {
@@ -9155,7 +9161,7 @@ enum {
 /*
  * CONFIG_DEBUG_INFO_BTF=m: a kfunc, dtor kfunc or struct_ops registration
  * made while the BTF it applies to is not available yet.  Kept until the BTF
- * arrives, see btf_defer_reg().
+ * arrives, see btf_defer_reg() and btf_apply_deferred_regs().
  */
 enum btf_deferred_reg_kind {
        BTF_DEFERRED_KFUNC_SET,
@@ -9180,12 +9186,30 @@ struct btf_deferred_reg {
 };
 
 #ifdef BTF_MODULE_NOTIFIER
+static void btf_free_deferred_regs(struct list_head *regs);
+static void btf_apply_deferred_regs(struct btf *btf, struct list_head *regs);
+
 struct btf_module {
        struct list_head list;
        struct module *module;
        struct btf *btf;
        struct bin_attribute *sysfs_attr;
        int flags;
+       /*
+        * CONFIG_DEBUG_INFO_BTF=m: a module loaded before the vmlinux BTF is
+        * available cannot have its BTF parsed yet.  Its .BTF and .BTF.base
+        * sections are copied here and parsed once the vmlinux BTF arrives
+        * (btf_parse_deferred_modules()); @btf is NULL until then.
+        * Registrations of the module's kfuncs, dtor kfuncs and struct_ops
+        * wait in @deferred_regs.
+        */
+       void *data;
+       void *base_data;
+       u32 data_size;
+       u32 base_data_size;
+       struct list_head deferred_regs;
+       /* the kept BTF turned out unusable; the entry stays until the module 
goes */
+       bool gone;
 };
 
 static LIST_HEAD(btf_modules);
@@ -9229,8 +9253,14 @@ static void btf_module_free(struct btf_module *btf_mod)
 {
        if (btf_mod->sysfs_attr)
                sysfs_remove_bin_file(btf_kobj, btf_mod->sysfs_attr);
-       purge_cand_cache(btf_mod->btf);
-       btf_put(btf_mod->btf);
+       if (btf_mod->btf) {
+               purge_cand_cache(btf_mod->btf);
+               btf_put(btf_mod->btf);
+       } else {
+               kvfree(btf_mod->data);
+               kvfree(btf_mod->base_data);
+       }
+       btf_free_deferred_regs(&btf_mod->deferred_regs);
        kfree(btf_mod->sysfs_attr);
        kfree(btf_mod);
 }
@@ -9280,6 +9310,66 @@ static bool btf_is_vmlinux_carrier(const struct module 
*mod)
 {
        return !strcmp(mod->name, btf_vmlinux_link.module_name);
 }
+
+/*
+ * The vmlinux BTF is not available yet and must not be loaded from the
+ * module notifier (that would nest a module load into a module load).  Keep
+ * the module's BTF for btf_parse_deferred_modules().
+ *
+ * Without a .BTF.base section the .BTF data is final and can be exposed in
+ * sysfs right away, it needs no parsing.  With one, parsing relocates the
+ * data in place against the vmlinux BTF, so the file is created afterwards,
+ * as with =y where it also only appears once the BTF is parsed.
+ */
+static int btf_module_defer(struct btf_module *btf_mod, struct module *mod)
+{
+       btf_mod->data = kvmemdup(mod->btf_data, mod->btf_data_size,
+                                GFP_KERNEL | __GFP_NOWARN);
+       if (!btf_mod->data)
+               return -ENOMEM;
+       btf_mod->data_size = mod->btf_data_size;
+
+       if (mod->btf_base_data) {
+               btf_mod->base_data = kvmemdup(mod->btf_base_data,
+                                             mod->btf_base_data_size,
+                                             GFP_KERNEL | __GFP_NOWARN);
+               if (!btf_mod->base_data) {
+                       kvfree(btf_mod->data);
+                       return -ENOMEM;
+               }
+               btf_mod->base_data_size = mod->btf_base_data_size;
+       } else {
+               /* not fatal, the module BTF is usable without the sysfs file */
+               btf_module_sysfs_add(btf_mod, mod->name, btf_mod->data,
+                                    btf_mod->data_size);
+       }
+
+       list_add(&btf_mod->list, &btf_modules);
+       return 0;
+}
+
+/*
+ * Apply the registrations queued for @btf_mod to @btf, and those that
+ * applying them queues in turn: a struct_ops ->init() registers the kfuncs
+ * of its hook.  Called and returns with btf_module_mutex held, which is
+ * dropped while applying, as that walks btf_modules
+ * (btf_check_kfunc_name()); the module is pinned meanwhile, so the entry
+ * stays.  A module that is going has its queue freed with the entry.
+ */
+static void btf_module_apply_regs(struct btf_module *btf_mod, struct btf *btf)
+{
+       LIST_HEAD(regs);
+
+       if (list_empty(&btf_mod->deferred_regs) || 
!try_module_get(btf_mod->module))
+               return;
+       while (!list_empty(&btf_mod->deferred_regs)) {
+               list_splice_init(&btf_mod->deferred_regs, &regs);
+               mutex_unlock(&btf_module_mutex);
+               btf_apply_deferred_regs(btf, &regs);
+               mutex_lock(&btf_module_mutex);
+       }
+       module_put(btf_mod->module);
+}
 #else
 static int btf_vmlinux_module_coming(struct module *mod)
 {
@@ -9290,6 +9380,15 @@ static bool btf_is_vmlinux_carrier(const struct module 
*mod)
 {
        return false;
 }
+
+static int btf_module_defer(struct btf_module *btf_mod, struct module *mod)
+{
+       return 0;
+}
+
+static void btf_module_apply_regs(struct btf_module *btf_mod, struct btf *btf)
+{
+}
 #endif
 
 static int btf_module_notify(struct notifier_block *nb, unsigned long op,
@@ -9321,6 +9420,26 @@ static int btf_module_notify(struct notifier_block *nb, 
unsigned long op,
                        goto out;
                }
                btf_mod->module = module;
+               INIT_LIST_HEAD(&btf_mod->deferred_regs);
+
+               if (IS_MODULE(CONFIG_DEBUG_INFO_BTF)) {
+                       mutex_lock(&btf_module_mutex);
+                       /* Pairs with the publication in bpf_load_btf_vmlinux() 
*/
+                       if (!smp_load_acquire(&btf_vmlinux)) {
+                               err = btf_module_defer(btf_mod, mod);
+                               mutex_unlock(&btf_module_mutex);
+                               if (err) {
+                                       pr_warn("failed to keep module [%s] 
BTF: %d\n",
+                                               mod->name, err);
+                                       kfree(btf_mod);
+                                       /* as a module BTF that fails to parse 
with =y */
+                                       if 
(IS_ENABLED(CONFIG_MODULE_ALLOW_BTF_MISMATCH))
+                                               err = 0;
+                               }
+                               goto out;
+                       }
+                       mutex_unlock(&btf_module_mutex);
+               }
 
                btf = btf_parse_module(mod->name, bpf_get_btf_vmlinux(),
                                       mod->btf_data, mod->btf_data_size, false,
@@ -9358,6 +9477,16 @@ static int btf_module_notify(struct notifier_block *nb, 
unsigned long op,
                        if (btf_mod->module != module)
                                continue;
 
+                       /*
+                        * The vmlinux BTF arrived while this module was
+                        * initializing: btf_parse_deferred_modules() parsed its
+                        * BTF but left the registrations its init queued to us.
+                        * Apply them before the module counts as live: with =y
+                        * all of them are done by then, and nothing may use the
+                        * module's kfuncs or struct_ops while they are added.
+                        */
+                       if (IS_MODULE(CONFIG_DEBUG_INFO_BTF) && btf_mod->btf)
+                               btf_module_apply_regs(btf_mod, btf_mod->btf);
                        btf_mod->flags |= BTF_MODULE_F_LIVE;
                        break;
                }
@@ -9375,7 +9504,8 @@ static int btf_module_notify(struct notifier_block *nb, 
unsigned long op,
                         * btf_try_get_module() on such BTFs will fail. This may
                         * be called again on btf_put(), but it's ok to do so.
                         */
-                       btf_free_id(btf_mod->btf);
+                       if (btf_mod->btf)
+                               btf_free_id(btf_mod->btf);
                        list_del(&btf_mod->list);
                        btf_module_free(btf_mod);
                        break;
@@ -9398,6 +9528,127 @@ static int __init btf_module_init(void)
 }
 
 fs_initcall(btf_module_init);
+
+#if IS_MODULE(CONFIG_DEBUG_INFO_BTF)
+/* The modules kept aside have been registered, see 
btf_parse_deferred_modules() */
+static bool btf_deferred_modules_done;
+
+/*
+ * A kept module whose BTF cannot be used after all.  The module is loaded
+ * and stays, so there is no way to reject it: the entry stays on the list,
+ * dead, until the module goes.  A sysfs file it has keeps serving the raw
+ * data, which is kept for that.
+ */
+static void btf_module_dead(struct btf_module *btf_mod, const char *what, int 
err)
+{
+       pr_warn("failed to %s module [%s] BTF: %d\n", what, 
btf_mod->module->name, err);
+       kvfree(btf_mod->base_data);
+       btf_mod->base_data = NULL;
+       btf_free_deferred_regs(&btf_mod->deferred_regs);
+       btf_mod->gone = true;
+}
+
+/*
+ * CONFIG_DEBUG_INFO_BTF=m: the vmlinux BTF has just become available.  Parse
+ * the BTF of the modules that were loaded before it, and apply the
+ * registrations that waited for them.  Called from bpf_load_btf_vmlinux()
+ * once btf_vmlinux is published, serialized by it, with no locks held.
+ *
+ * A module's BTF is published (btf_mod->btf set, id installed) only after
+ * its queued registrations are applied, so nobody sees a module BTF without
+ * its kfuncs and struct_ops, as with the vmlinux BTF; its id is reserved
+ * before, so that nothing can fail once they are.  Applying walks
+ * btf_modules (btf_check_kfunc_name()) and so needs the mutex dropped; the
+ * module is pinned for that, and the scan restarts afterwards.  A module
+ * that is still initializing is only published: its init is still queueing
+ * registrations, and MODULE_STATE_LIVE applies them once it is done.
+ *
+ * Until this has run, a search of the module BTFs may miss one of these
+ * modules: see btf_deferred_modules_pending().
+ */
+void btf_parse_deferred_modules(void)
+{
+       /* Pairs with the publication in bpf_load_btf_vmlinux() */
+       struct btf *vmlinux_btf = smp_load_acquire(&btf_vmlinux);
+       struct btf_module *btf_mod;
+       bool parsed = false;
+       struct btf *btf;
+       int err;
+
+       if (!vmlinux_btf || !btf_deferred_modules_pending())
+               return;
+
+       mutex_lock(&btf_module_mutex);
+       if (IS_ERR(vmlinux_btf)) {
+               /* remembered failure: the kept modules can never be parsed */
+               list_for_each_entry(btf_mod, &btf_modules, list) {
+                       if (!btf_mod->btf && !btf_mod->gone)
+                               btf_module_dead(btf_mod, "parse", 
PTR_ERR(vmlinux_btf));
+               }
+               mutex_unlock(&btf_module_mutex);
+               goto done;
+       }
+restart:
+       list_for_each_entry(btf_mod, &btf_modules, list) {
+               if (btf_mod->btf || btf_mod->gone)
+                       continue;
+
+               btf = btf_parse_module(btf_mod->module->name, vmlinux_btf,
+                                      btf_mod->data, btf_mod->data_size, true,
+                                      btf_mod->base_data, 
btf_mod->base_data_size);
+               if (IS_ERR(btf)) {
+                       /* on failure the caller keeps the data */
+                       btf_module_dead(btf_mod, "validate", PTR_ERR(btf));
+                       continue;
+               }
+               /* btf->data is btf_mod->data now, the sysfs file keeps 
pointing at valid data */
+               kvfree(btf_mod->base_data);
+               btf_mod->base_data = NULL;
+
+               err = btf_reserve_id(btf);
+               if (err) {
+                       /* give the data back to the entry, the sysfs file may 
serve it */
+                       btf->data = NULL;
+                       btf_free(btf);
+                       btf_module_dead(btf_mod, "register", err);
+                       continue;
+               }
+
+               if (btf_mod->flags & BTF_MODULE_F_LIVE)
+                       btf_module_apply_regs(btf_mod, btf);
+
+               /* modules with .BTF.base get their sysfs file now, the data is 
relocated */
+               if (!btf_mod->sysfs_attr)
+                       btf_module_sysfs_add(btf_mod, btf->name, btf->data, 
btf->data_size);
+               btf_mod->data = NULL;
+               btf_mod->btf = btf;
+               btf_install_id(btf);
+               parsed = true;
+               /* the list may have changed while the mutex was dropped */
+               goto restart;
+       }
+       mutex_unlock(&btf_module_mutex);
+
+       if (parsed)
+               purge_cand_cache(NULL);
+done:
+       /* Pairs with the smp_load_acquire() in btf_deferred_modules_pending() 
*/
+       smp_store_release(&btf_deferred_modules_done, true);
+}
+
+/*
+ * The vmlinux BTF is published before the BTF of the modules loaded ahead of
+ * it is registered by btf_parse_deferred_modules().  Until then, a search of
+ * the module BTFs may miss a module: such searches count a miss and fail, so
+ * that bpf() runs the command again after bpf_load_btf_vmlinux(), which
+ * waits for the modules.
+ */
+bool btf_deferred_modules_pending(void)
+{
+       /* Pairs with the smp_store_release() in btf_parse_deferred_modules() */
+       return !smp_load_acquire(&btf_deferred_modules_done);
+}
+#endif /* IS_MODULE(CONFIG_DEBUG_INFO_BTF) */
 #endif /* BTF_MODULE_NOTIFIER */
 
 struct module *btf_try_get_module(const struct btf *btf)
@@ -9450,8 +9701,11 @@ struct btf *btf_get_module_btf(const struct module 
*module)
                if (btf_mod->module != module)
                        continue;
 
-               btf_get(btf_mod->btf);
-               btf = btf_mod->btf;
+               /* NULL while waiting for the vmlinux BTF 
(CONFIG_DEBUG_INFO_BTF=m) */
+               if (btf_mod->btf) {
+                       btf_get(btf_mod->btf);
+                       btf = btf_mod->btf;
+               }
                break;
        }
        mutex_unlock(&btf_module_mutex);
@@ -9634,7 +9888,8 @@ static int btf_check_kfunc_name(struct btf *btf, const 
char *func_name, u32 kind
 #ifdef CONFIG_DEBUG_INFO_BTF_MODULES
        guard(mutex)(&btf_module_mutex);
        list_for_each_entry_safe(btf_mod, tmp, &btf_modules, list) {
-               if (btf_mod->btf == btf)
+               /* skip ourselves and, with CONFIG_DEBUG_INFO_BTF=m, unparsed 
BTF */
+               if (btf_mod->btf == btf || !btf_mod->btf)
                        continue;
                id = btf_find_by_name_kind(btf_mod->btf, func_name, kind);
                if (id >= 0) {
@@ -10492,6 +10747,12 @@ bpf_core_find_cands(struct bpf_core_ctx *ctx, u32 
local_type_id)
                return cc;
 
 check_modules:
+       /* the modules loaded before the vmlinux BTF may not be registered yet 
*/
+       if (btf_deferred_modules_pending()) {
+               bpf_btf_vmlinux_miss();
+               return ERR_PTR(-EINVAL);
+       }
+
        /* cands is a pointer to stack here and cands->cnt == 0 */
        cc = check_cand_cache(cands, module_cand_cache, MODULE_CAND_CACHE_SIZE);
        if (cc)
@@ -10868,15 +11129,17 @@ static int btf_struct_ops_register(struct btf *btf, 
struct bpf_struct_ops *st_op
 #endif
 
 /*
- * CONFIG_DEBUG_INFO_BTF=m: registrations for vmlinux made before its BTF is
- * available wait in btf_vmlinux_deferred_regs until btf_parse_vmlinux()
- * applies them.
+ * CONFIG_DEBUG_INFO_BTF=m: registrations made before the BTF they apply to
+ * is available.  Registrations for vmlinux wait in btf_vmlinux_deferred_regs
+ * until btf_parse_vmlinux() applies them; registrations for a module wait in
+ * its struct btf_module, under btf_module_mutex, until
+ * btf_parse_deferred_modules() does.
  */
 #ifdef BTF_MODULE_NOTIFIER
 /*
- * The queue has its own lock: it is drained under btf_vmlinux_lock, and
- * with its own lock btf_module_mutex, which the module notifier takes, is
- * never taken under btf_vmlinux_lock, so the two stay unordered.
+ * The vmlinux queue has its own lock: it is drained under btf_vmlinux_lock,
+ * and with its own lock btf_module_mutex, which the module notifier takes,
+ * is never taken under btf_vmlinux_lock, so the two stay unordered.
  */
 static DEFINE_MUTEX(btf_vmlinux_regs_mutex);
 static LIST_HEAD(btf_vmlinux_deferred_regs);
@@ -10886,19 +11149,38 @@ static bool btf_vmlinux_regs_closed;
 /*
  * Queue @tmpl if the BTF for @owner is not available yet.  Returns 1 if the
  * registration was queued and is to be considered done, 0 if the caller has
- * to apply it, or -ENOMEM.  Only vmlinux registrations are queued so far.
+ * to apply it, or -ENOMEM.
  */
 static int btf_defer_reg(struct module *owner, const struct btf_deferred_reg 
*tmpl)
 {
        struct list_head *head = NULL;
        struct btf_deferred_reg *reg;
+       struct btf_module *btf_mod;
 
-       if (!IS_MODULE(CONFIG_DEBUG_INFO_BTF) || owner)
+       if (!IS_MODULE(CONFIG_DEBUG_INFO_BTF))
                return 0;
 
-       guard(mutex)(&btf_vmlinux_regs_mutex);
-       if (!btf_vmlinux_regs_closed)
-               head = &btf_vmlinux_deferred_regs;
+       guard(mutex)(owner ? &btf_module_mutex : &btf_vmlinux_regs_mutex);
+       if (!owner) {
+               if (!btf_vmlinux_regs_closed)
+                       head = &btf_vmlinux_deferred_regs;
+       } else {
+               list_for_each_entry(btf_mod, &btf_modules, list) {
+                       if (btf_mod->module != owner)
+                               continue;
+                       /*
+                        * Wait until the entry's BTF is published, and after 
that
+                        * behind registrations that are still queued (a module
+                        * that was initializing when its BTF was published), so
+                        * that they are applied in order.  A dead entry has no 
BTF
+                        * to register with, as with =y.
+                        */
+                       if (!btf_mod->gone &&
+                           (!btf_mod->btf || 
!list_empty(&btf_mod->deferred_regs)))
+                               head = &btf_mod->deferred_regs;
+                       break;
+               }
+       }
        if (!head)
                return 0;
 
@@ -10951,7 +11233,10 @@ static int btf_apply_deferred_reg(struct btf *btf, 
const struct btf_deferred_reg
        return -EINVAL;
 }
 
-/* Apply and free the registrations in @regs to @btf. */
+/*
+ * Apply and free the registrations in @regs to @btf.  For a module BTF the
+ * caller holds a reference on @btf and makes sure the owning module stays.
+ */
 static void btf_apply_deferred_regs(struct btf *btf, struct list_head *regs)
 {
        struct btf_deferred_reg *reg, *tmp;
@@ -10967,6 +11252,16 @@ static void btf_apply_deferred_regs(struct btf *btf, 
struct list_head *regs)
        }
 }
 
+static void btf_free_deferred_regs(struct list_head *regs)
+{
+       struct btf_deferred_reg *reg, *tmp;
+
+       list_for_each_entry_safe(reg, tmp, regs, list) {
+               list_del(&reg->list);
+               btf_free_deferred_reg(reg);
+       }
+}
+
 /*
  * The vmlinux BTF has just been parsed; apply the registrations that waited
  * for it.  Runs under btf_vmlinux_lock, before @btf is published, so nothing
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 895e7feb6429..f6205894fe11 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -21935,18 +21935,22 @@ struct btf *bpf_peek_btf_vmlinux(void)
  * bpf_load_btf_vmlinux - get the vmlinux BTF, loading it if necessary
  *
  * Like bpf_get_btf_vmlinux(), but with CONFIG_DEBUG_INFO_BTF=m it loads the
- * btf_vmlinux module if the BTF is not there yet and parses it.  Loading the
- * module waits for user space (modprobe), and the notifiers of the module
- * load take locks of their own, event_mutex among them.  So this is only
- * called at the start of a request from user space, in process context,
- * holding no lock that loading a module may need; everything else uses
- * bpf_get_btf_vmlinux() or bpf_peek_btf_vmlinux().
+ * btf_vmlinux module if the BTF is not there yet, parses it, and registers
+ * the BTF of the modules that were loaded before it; it returns once all of
+ * that is done, also to a caller that comes while another one is doing it.
+ * Loading the module waits for user space (modprobe), and the notifiers of
+ * the module load take locks of their own, event_mutex among them.  So this
+ * is only called at the start of a request from user space, in process
+ * context, holding no lock that loading a module may need; everything else
+ * uses bpf_get_btf_vmlinux() or bpf_peek_btf_vmlinux().
  *
  * If the module cannot be loaded, returns NULL like a kernel without BTF;
  * the next call tries again.
  */
 struct btf *bpf_load_btf_vmlinux(void)
 {
+       /* Held from loading until the module BTF kept aside is registered */
+       static DEFINE_MUTEX(load_mutex);
        struct btf *btf;
        u32 size;
 
@@ -21956,13 +21960,14 @@ struct btf *bpf_load_btf_vmlinux(void)
 
        /* Pairs with the smp_store_release() below */
        btf = smp_load_acquire(&btf_vmlinux);
-       if (btf)
+       if (btf && !btf_deferred_modules_pending())
                return btf;
 
-       /* Outside btf_vmlinux_lock, the module's notifier must not wait for us 
*/
-       if (!btf_vmlinux_data(&size, true))
+       /* Outside the locks, the module's notifier must not wait for us */
+       if (!btf && !btf_vmlinux_data(&size, true))
                return NULL;
 
+       mutex_lock(&load_mutex);
        mutex_lock(&btf_vmlinux_lock);
        btf = btf_vmlinux;
        if (!btf) {
@@ -21974,12 +21979,22 @@ struct btf *bpf_load_btf_vmlinux(void)
                 */
                if (IS_ERR(btf) && PTR_ERR(btf) == -ENOMEM) {
                        mutex_unlock(&btf_vmlinux_lock);
+                       mutex_unlock(&load_mutex);
                        return btf;
                }
                /* As in bpf_get_btf_vmlinux(): publish after the parse */
                smp_store_release(&btf_vmlinux, btf);
        }
        mutex_unlock(&btf_vmlinux_lock);
+
+       /*
+        * Until this is done, the module BTFs may lack a module, which
+        * btf_deferred_modules_pending() tells the searches of module BTFs,
+        * and which is why concurrent callers wait for it on the mutex.
+        */
+       btf_parse_deferred_modules();
+       mutex_unlock(&load_mutex);
+
        return btf;
 }
 
@@ -21994,6 +22009,12 @@ unsigned int bpf_btf_vmlinux_misses(void)
 {
        return atomic_read(&btf_vmlinux_misses);
 }
+
+/* A lookup that needs the BTF of a module that is not registered yet */
+void bpf_btf_vmlinux_miss(void)
+{
+       atomic_inc(&btf_vmlinux_misses);
+}
 #endif
 
 /*
-- 
2.47.3


Reply via email to