On Mon, Sep 21, 2026 at 07:30:45PM +0800, Xiang Gao wrote:
> Report the memory consumed by the tracing ring buffers, rather than the
> usable data capacity exposed by buffer_size_kb. Android low-memory
> diagnostics need this to attribute the memory used by tracing when
> calculating lost RAM.
> 
> The buffers can be spread across the global trace array, dynamically
> created instances, and snapshot buffers. Userspace currently has to
> discover and sum every instance, and snapshot memory is not exposed by
> the per-instance totals.
> 
> Add a trace_stats directory with memory_usage_kb reporting:
> 
>   buffers:
>   snapshot_buffers:
> 
> covering the global trace array, all instances, and the bootstrapping
> temp_buffer across all CPUs.
> 
> The values account for the full pages backing the data sub-buffers and
> reader page, plus the cached read page and mmap metadata page when
> present. Slab-allocated ring-buffer metadata is not included, as it is
> already reported through Slab and would be double-counted when
> subtracting tracing memory from lost RAM. Remote buffers, whose pages
> are externally owned, report zero.

For the next version, it is good practice to __not__ in-reply-to with previous
version.

> 
> Signed-off-by: Xiang Gao <[email protected]>
> ---
>  Documentation/trace/ftrace.rst | 12 +++++
>  include/linux/ring_buffer.h    |  1 +
>  kernel/trace/ring_buffer.c     | 41 +++++++++++++++
>  kernel/trace/trace.c           | 95 ++++++++++++++++++++++++++++++++++
>  4 files changed, 149 insertions(+)
> 
> diff --git a/Documentation/trace/ftrace.rst b/Documentation/trace/ftrace.rst
> index 7261f25f8b4b..99ddfe26b7cd 100644
> --- a/Documentation/trace/ftrace.rst
> +++ b/Documentation/trace/ftrace.rst
> @@ -218,6 +218,18 @@ of ftrace. Here is a list of some of the key files:
>  
>       This displays the total combined size of all the trace buffers.
>  
> +  trace_stats/memory_usage_kb:
> +
> +     This reports the memory consumed by the ring buffers, as opposed to
> +     the usable data capacity shown by buffer_size_kb. The value covers the
> +     main and snapshot buffers of the global trace array and all tracing
> +     instances. It does not include slab-allocated ring-buffer metadata.
> +
> +     Output::
> +
> +         buffers: ...
> +         snapshot_buffers: ...
> +
>    buffer_subbuf_size_kb:
>  
>       This sets or displays the sub buffer size. The ring buffer is broken up
> diff --git a/include/linux/ring_buffer.h b/include/linux/ring_buffer.h
> index eac3e9080c3c..96b99e6757d4 100644
> --- a/include/linux/ring_buffer.h
> +++ b/include/linux/ring_buffer.h
> @@ -167,6 +167,7 @@ int ring_buffer_iter_empty(struct ring_buffer_iter *iter);
>  bool ring_buffer_iter_dropped(struct ring_buffer_iter *iter);
>  
>  unsigned long ring_buffer_size(struct trace_buffer *buffer, int cpu);
> +unsigned long ring_buffer_memory_size(struct trace_buffer *buffer, int cpu);
>  unsigned long ring_buffer_max_event_size(struct trace_buffer *buffer);
>  
>  void ring_buffer_reset_cpu(struct trace_buffer *buffer, int cpu);
> diff --git a/kernel/trace/ring_buffer.c b/kernel/trace/ring_buffer.c
> index 04bb94c29f58..efb88bf8970c 100644
> --- a/kernel/trace/ring_buffer.c
> +++ b/kernel/trace/ring_buffer.c
> @@ -6559,6 +6559,47 @@ unsigned long ring_buffer_size(struct trace_buffer 
> *buffer, int cpu)
>  }
>  EXPORT_SYMBOL_GPL(ring_buffer_size);
>  
> +/**
> + * ring_buffer_memory_size - return the memory used by the buffer (in bytes)
> + * @buffer: The ring buffer.
> + * @cpu: The CPU to get ring buffer memory from.
> + *
> + * Returns the page-allocator memory consumed by @cpu, including the data
> + * sub-buffers, the reader page, the cached read page, and the mmap
> + * metadata page. Unlike ring_buffer_size(), which reports the usable data
> + * capacity, this accounts for the full pages allocated to the buffer.
> + * Remote buffers do not own page-allocator memory and report zero.
> + */
> +unsigned long ring_buffer_memory_size(struct trace_buffer *buffer, int cpu)
> +{
> +     struct ring_buffer_per_cpu *cpu_buffer;
> +     unsigned long subbuf_size;
> +     unsigned long size;
> +
> +     if (!cpumask_test_cpu(cpu, buffer->cpumask))
> +             return 0;
> +
> +     /* Remote buffers use externally owned memory. */
> +     if (buffer->remote)
> +             return 0;

This is a generic interface. If you want to call this function on a remote
buffer, you should be able to.

Moreover, remote buffer in-production current use is for Android... So not only
ring_buffer_memory_size() should support them, but they should probably be
actively reported somewhere...

> +
> +     cpu_buffer = buffer->buffers[cpu];
> +     subbuf_size = PAGE_SIZE << READ_ONCE(buffer->subbuf_order);
> +
> +     /* Data sub-buffers plus the reader page. */
> +     size = (READ_ONCE(cpu_buffer->nr_pages) + 1) * subbuf_size;
> +
> +     /* The cached read page, if present, is a full sub-buffer page. */
> +     if (READ_ONCE(cpu_buffer->free_page.data))
> +             size += subbuf_size;
> +
> +     /* The mmap metadata page is a single system page. */
> +     if (READ_ONCE(cpu_buffer->meta_page))
> +             size += PAGE_SIZE;
> +
> +     return size;
> +}
> +
>  /**
>   * ring_buffer_max_event_size - return the max data size of an event
>   * @buffer: The ring buffer.
> diff --git a/kernel/trace/trace.c b/kernel/trace/trace.c
> index e4a490d3d08c..d4a913ff8a69 100644
> --- a/kernel/trace/trace.c
> +++ b/kernel/trace/trace.c
> @@ -5771,6 +5771,79 @@ tracing_total_entries_read(struct file *filp, char 
> __user *ubuf,
>       return simple_read_from_buffer(ubuf, cnt, ppos, buf, r);
>  }
>  
> +struct trace_mem_stats {
> +     unsigned long   buffers;
> +     unsigned long   snapshot;
> +};
> +
> +static void
> +trace_array_buffer_memory(struct trace_array *tr, int cpu,
> +                       unsigned long *buffers, unsigned long *snapshot)
> +{
> +     if (tr->array_buffer.buffer)
> +             *buffers += ring_buffer_memory_size(tr->array_buffer.buffer, 
> cpu);
> +
> +#ifdef CONFIG_TRACER_SNAPSHOT
> +     if (tr->snapshot_buffer.buffer)
> +             *snapshot += 
> ring_buffer_memory_size(tr->snapshot_buffer.buffer, cpu);
> +#endif
> +}
> +
> +static struct trace_mem_stats trace_buffers_memory(void)
> +{
> +     struct trace_mem_stats stats = {};
> +     struct trace_array *tr;
> +     int cpu;
> +
> +     guard(mutex)(&trace_types_lock);
> +
> +     list_for_each_entry(tr, &ftrace_trace_arrays, list) {
> +             for_each_tracing_cpu(cpu)
> +                     trace_array_buffer_memory(tr, cpu, &stats.buffers,
> +                                               &stats.snapshot);
> +     }
> +
> +     /*
> +      * temp_buffer is allocated in tracer_alloc_buffers() and is never
> +      * attached to a trace array. It temporarily holds event data for
> +      * triggers when tracing is off. Account for its pages too.
> +      */
> +     if (temp_buffer) {
> +             for_each_tracing_cpu(cpu)
> +                     stats.buffers += ring_buffer_memory_size(temp_buffer, 
> cpu);
> +     }
> +
> +     return stats;
> +}
> +
> +static int trace_mem_show(struct seq_file *m, void *v)
> +{
> +     struct trace_mem_stats stats = trace_buffers_memory();
> +
> +     seq_printf(m, "buffers: %lu\n", stats.buffers >> 10);
> +     seq_printf(m, "snapshot_buffers: %lu\n", stats.snapshot >> 10);
> +
> +     return 0;
> +}
> +
> +static int trace_mem_open(struct inode *inode, struct file *file)
> +{
> +     int ret;
> +
> +     ret = tracing_check_open_get_tr(NULL);
> +     if (ret)
> +             return ret;
> +
> +     return single_open(file, trace_mem_show, inode->i_private);
> +}
> +
> +static const struct file_operations trace_mem_fops = {
> +     .open           = trace_mem_open,
> +     .read           = seq_read,
> +     .llseek         = seq_lseek,
> +     .release        = single_release,
> +};
> +
>  #define LAST_BOOT_HEADER ((void *)1)
>  
>  static void *l_next(struct seq_file *m, void *v, loff_t *pos)
> @@ -9285,6 +9358,26 @@ static struct notifier_block trace_module_nb = {
>  };
>  #endif /* CONFIG_MODULES */
>  
> +static __init void init_trace_stats_tracefs(void)
> +{
> +     struct dentry *stats_dir;
> +
> +     /*
> +      * tracer_alloc_buffers() frees tracing_buffer_mask and temp_buffer
> +      * on failure without NULLing them, so do not iterate tracing CPUs
> +      * here when tracing failed to initialize.
> +      */
> +     if (tracing_disabled)
> +             return;
> +
> +     stats_dir = tracefs_create_dir("trace_stats", NULL);
> +     if (!stats_dir)
> +             return;
> +
> +     trace_create_file("memory_usage_kb", TRACE_MODE_READ, stats_dir,
> +                       NULL, &trace_mem_fops);
> +}
> +
>  static __init void tracer_init_tracefs_work_func(struct work_struct *work)
>  {
>  
> @@ -9293,6 +9386,8 @@ static __init void tracer_init_tracefs_work_func(struct 
> work_struct *work)
>       init_tracer_tracefs(&global_trace, NULL);
>       ftrace_init_tracefs_toplevel(&global_trace, NULL);
>  
> +     init_trace_stats_tracefs();
> +
>       trace_create_file("tracing_thresh", TRACE_MODE_WRITE, NULL,
>                       &global_trace, &tracing_thresh_fops);
>  
> -- 
> 2.34.1
> 

-- 
Vincent

Reply via email to