With CONFIG_DEBUG_INFO_BTF=m, bpf_get_btf_vmlinux() and
bpf_find_btf_id() do not load the vmlinux BTF: loading waits for user
space, and their callers were not written for that.  Besides the bpf()
system call and /sys/kernel/btf/vmlinux, some tracefs and bpffs requests
need the BTF.  Load it at the start of those, with
bpf_load_btf_vmlinux(), and have the code that only uses the BTF if it
happens to be there peek:

 - Reading a tracepoint's btf_ids file (events/*/btf_ids): built-in
   events use the vmlinux BTF, and module BTF is only registered once
   that is loaded.  event_btf_ids_read() looks the BTF up under
   event_mutex, which the trace notifier of the btf_vmlinux module
   takes, so load it before taking the mutex, on the first read() of
   the file (not again for the one that returns EOF).
 - Probe events with BTF arguments ($argN, argument names, $retval,
   $current, typecasts): the parser loads the BTF before looking a
   function or struct up.  It holds dyn_event_ops_mutex, which loading
   a module never takes: besides event creation, only
   dyn_event_register() takes it, from built-in init code.  Loading
   here also keeps a $retval from silently losing its type.
 - The ftrace function argument printer (func-args, funcgraph-args)
   runs in the trace output path, which includes ftrace_dump() with
   interrupts disabled.  It only prints the arguments if the BTF is
   already loaded, and never loads it: that also avoids one modprobe
   per trace line when the module is not installed.
 - bpffs: parsing delegate_* mount options (fs_context) loads the BTF
   when a value names commands or types, which are looked up in it;
   "any" and numeric masks need no BTF and do not load it.  Showing the
   options in /proc/*/mountinfo runs under namespace_sem; it only uses
   the names if the BTF is already there and falls back to hex, as it
   already does without BTF.

With CONFIG_DEBUG_INFO_BTF=y bpf_load_btf_vmlinux() is
bpf_get_btf_vmlinux() and the BTF is parsed at boot, so nothing changes.

Signed-off-by: Jay Wang <[email protected]>
---
 kernel/bpf/inode.c          | 44 ++++++++++++++++++++++++-------------
 kernel/trace/trace_events.c | 10 +++++++++
 kernel/trace/trace_output.c |  7 ++++++
 kernel/trace/trace_probe.c  | 16 ++++++++++++++
 4 files changed, 62 insertions(+), 15 deletions(-)

diff --git a/kernel/bpf/inode.c b/kernel/bpf/inode.c
index 7837968c0842..d05bbb61a593 100644
--- a/kernel/bpf/inode.c
+++ b/kernel/bpf/inode.c
@@ -658,7 +658,11 @@ struct bpffs_btf_enums {
        const struct btf_type *attach_t;
 };
 
-static int find_bpffs_btf_enums(struct bpffs_btf_enums *info)
+/*
+ * @load: load the vmlinux BTF if necessary (CONFIG_DEBUG_INFO_BTF=m), see
+ * bpf_load_btf_vmlinux(); otherwise only use it if it is already parsed.
+ */
+static int find_bpffs_btf_enums(struct bpffs_btf_enums *info, bool load)
 {
        struct {
                const struct btf_type **type;
@@ -674,7 +678,7 @@ static int find_bpffs_btf_enums(struct bpffs_btf_enums 
*info)
 
        memset(info, 0, sizeof(*info));
 
-       btf = bpf_get_btf_vmlinux();
+       btf = load ? bpf_load_btf_vmlinux() : bpf_peek_btf_vmlinux();
        if (IS_ERR(btf))
                return PTR_ERR(btf);
        if (!btf)
@@ -795,8 +799,11 @@ static int bpf_show_options(struct seq_file *m, struct 
dentry *root)
            opts->delegate_progs || opts->delegate_attachs) {
                struct bpffs_btf_enums info;
 
-               /* ignore errors, fallback to hex */
-               (void)find_bpffs_btf_enums(&info);
+               /*
+                * ignore errors, fallback to hex; this runs under
+                * namespace_sem, so do not load the BTF from here
+                */
+               (void)find_bpffs_btf_enums(&info, false);
 
                mask = (1ULL << __MAX_BPF_CMD) - 1;
                seq_print_delegate_opts(m, "delegate_cmds",
@@ -1052,35 +1059,33 @@ static int bpf_parse_param(struct fs_context *fc, 
struct fs_parameter *param)
        case OPT_DELEGATE_MAPS:
        case OPT_DELEGATE_PROGS:
        case OPT_DELEGATE_ATTACHS: {
-               struct bpffs_btf_enums info;
-               const struct btf_type *enum_t;
+               struct bpffs_btf_enums info = {};
+               const struct btf_type **enum_t;
+               bool enums_tried = false;
                const char *enum_pfx;
-               u64 *delegate_msk, msk = 0;
+               u64 *delegate_msk, msk = 0, num;
                char *p, *str;
                int val;
 
-               /* ignore errors, fallback to hex */
-               (void)find_bpffs_btf_enums(&info);
-
                switch (opt) {
                case OPT_DELEGATE_CMDS:
                        delegate_msk = &opts->delegate_cmds;
-                       enum_t = info.cmd_t;
+                       enum_t = &info.cmd_t;
                        enum_pfx = "BPF_";
                        break;
                case OPT_DELEGATE_MAPS:
                        delegate_msk = &opts->delegate_maps;
-                       enum_t = info.map_t;
+                       enum_t = &info.map_t;
                        enum_pfx = "BPF_MAP_TYPE_";
                        break;
                case OPT_DELEGATE_PROGS:
                        delegate_msk = &opts->delegate_progs;
-                       enum_t = info.prog_t;
+                       enum_t = &info.prog_t;
                        enum_pfx = "BPF_PROG_TYPE_";
                        break;
                case OPT_DELEGATE_ATTACHS:
                        delegate_msk = &opts->delegate_attachs;
-                       enum_t = info.attach_t;
+                       enum_t = &info.attach_t;
                        enum_pfx = "BPF_";
                        break;
                default:
@@ -1089,9 +1094,18 @@ static int bpf_parse_param(struct fs_context *fc, struct 
fs_parameter *param)
 
                str = param->string;
                while ((p = strsep(&str, ":"))) {
+                       /*
+                        * Only names need the vmlinux BTF: "any" and numbers do
+                        * not load it.  Ignore errors, fallback to hex.
+                        */
+                       if (strcmp(p, "any") && kstrtou64(p, 0, &num) && 
!enums_tried) {
+                               (void)find_bpffs_btf_enums(&info, true);
+                               enums_tried = true;
+                       }
+
                        if (strcmp(p, "any") == 0) {
                                msk |= ~0ULL;
-                       } else if (find_btf_enum_const(info.btf, enum_t, 
enum_pfx, p, &val)) {
+                       } else if (find_btf_enum_const(info.btf, *enum_t, 
enum_pfx, p, &val)) {
                                msk |= 1ULL << val;
                        } else {
                                err = kstrtou64(p, 0, &msk);
diff --git a/kernel/trace/trace_events.c b/kernel/trace/trace_events.c
index 30c0ddf90887..c887ac6a4857 100644
--- a/kernel/trace/trace_events.c
+++ b/kernel/trace/trace_events.c
@@ -23,6 +23,7 @@
 #include <linux/sort.h>
 #include <linux/slab.h>
 #include <linux/delay.h>
+#include <linux/bpf.h>
 #include <linux/btf.h>
 
 #include <trace/events/sched.h>
@@ -2245,6 +2246,15 @@ event_btf_ids_read(struct file *filp, char __user *ubuf, 
size_t cnt, loff_t *ppo
        char buf[128];
        int len;
 
+       /*
+        * Built-in events use the vmlinux BTF, and with CONFIG_DEBUG_INFO_BTF=m
+        * module BTF is only registered once that is loaded.  Loading it loads
+        * a module, whose trace notifier takes event_mutex: load it before
+        * taking that, and only for the first read, not again for the EOF one.
+        */
+       if (!*ppos)
+               bpf_load_btf_vmlinux();
+
        /* Module unload could free call->class and ids[] mid-read. */
        scoped_guard(mutex, &event_mutex) {
                file = event_file_file(filp);
diff --git a/kernel/trace/trace_output.c b/kernel/trace/trace_output.c
index a5ad76175d10..1f346e524ee6 100644
--- a/kernel/trace/trace_output.c
+++ b/kernel/trace/trace_output.c
@@ -739,6 +739,13 @@ void print_function_args(struct trace_seq *s, unsigned 
long *args,
        if (lookup_symbol_name(func, name))
                goto out;
 
+       /*
+        * This can run with interrupts disabled (ftrace_dump()): only use
+        * the vmlinux BTF if it is parsed, never load it from here.
+        */
+       if (IS_ERR_OR_NULL(bpf_peek_btf_vmlinux()))
+               goto out;
+
        /* TODO: Pass module name here too */
        t = btf_find_func_proto(name, &btf);
        if (IS_ERR_OR_NULL(t))
diff --git a/kernel/trace/trace_probe.c b/kernel/trace/trace_probe.c
index 804442b2f7d2..d53ee1ef820c 100644
--- a/kernel/trace/trace_probe.c
+++ b/kernel/trace/trace_probe.c
@@ -531,6 +531,19 @@ static const char *fetch_type_from_btf_type(struct btf 
*btf,
        return NULL;
 }
 
+/*
+ * Arguments described by BTF need the vmlinux BTF.  With
+ * CONFIG_DEBUG_INFO_BTF=m it may not be loaded yet, so load it before looking
+ * anything up.  Parsing holds dyn_event_ops_mutex, which loading a module
+ * never takes: besides event creation, only dyn_event_register() takes it,
+ * from built-in init code.
+ */
+static void trace_probe_load_btf(void)
+{
+       lockdep_assert_held(&dyn_event_ops_mutex);
+       bpf_load_btf_vmlinux();
+}
+
 static int query_btf_context(struct traceprobe_parse_context *ctx)
 {
        const struct btf_param *param;
@@ -544,6 +557,7 @@ static int query_btf_context(struct 
traceprobe_parse_context *ctx)
        if (!ctx->funcname)
                return -EINVAL;
 
+       trace_probe_load_btf();
        type = btf_find_func_proto(ctx->funcname, &btf);
        if (!type)
                return -ENOENT;
@@ -762,6 +776,7 @@ static int parse_btf_arg(char *varname,
        if (!strcmp(varname, "$current")) {
                code->op = FETCH_OP_CURRENT;
                /* If no typecast is specified for $current, use task_struct by 
default */
+               trace_probe_load_btf();
                ret = bpf_find_btf_id("task_struct", BTF_KIND_STRUCT, 
&ctx->struct_btf);
                if (ret < 0) {
                        trace_probe_log_err(ctx->offset, NO_BTF_ENTRY);
@@ -890,6 +905,7 @@ static int query_btf_struct(const char *sname, struct 
traceprobe_parse_context *
                ctx->struct_btf = NULL;
        }
 
+       trace_probe_load_btf();
        id = bpf_find_btf_id(sname, BTF_KIND_STRUCT, &btf);
        if (id < 0)
                return id;
-- 
2.47.3


Reply via email to