When a concurrency-managed per-CPU work item runs continuously without sleeping for longer than wq_cpu_intensive_thresh_us, wq_worker_tick() marks the worker as WORKER_CPU_INTENSIVE and kicks it out of concurrency management so that pending work items on the pool are not starved.
While CONFIG_WQ_CPU_INTENSIVE_REPORT logs rate-limited warnings and pwq->stats[PWQ_STAT_CPU_INTENSIVE] maintains a cumulative counter, there is currently no tracepoint emitted at the moment of this transition. Therefore, add the workqueue_cpu_intensive tracepoint, recording the work_struct pointer and callback function pointer, workqueue name, executing CPU, and the runtime duration consumed in microseconds. This enables eBPF profilers, bpftrace, and Ftrace to immediately detect and attribute CPU-hogging work items in real time. Signed-off-by: Aaron Tomlin <[email protected]> --- include/trace/events/workqueue.h | 40 ++++++++++++++++++++++++++++++++ kernel/workqueue.c | 7 ++++++ 2 files changed, 47 insertions(+) diff --git a/include/trace/events/workqueue.h b/include/trace/events/workqueue.h index b0de2bc9ed52..d2ad38f93a83 100644 --- a/include/trace/events/workqueue.h +++ b/include/trace/events/workqueue.h @@ -126,7 +126,47 @@ TRACE_EVENT(workqueue_execute_end, TP_printk("work struct %p: function %ps", __entry->work, __entry->function) ); +/** + * workqueue_cpu_intensive - called when a work item exceeds cpu_intensive threshold + * @pwq: pointer to struct pool_workqueue + * @work: pointer to struct work_struct + * @function: pointer to worker function + * @duration_us: CPU time consumed in microseconds + * + * This event occurs when a concurrency-managed work item runs for longer + * than wq_cpu_intensive_thresh_us without sleeping and is excluded from + * concurrency management to prevent stalling other work items. + */ +TRACE_EVENT(workqueue_cpu_intensive, + + TP_PROTO(struct pool_workqueue *pwq, struct work_struct *work, + work_func_t function, u64 duration_us), + + TP_ARGS(pwq, work, function, duration_us), + + TP_STRUCT__entry( + __field( void *, work ) + __field( void *, function ) + __string( workqueue, pwq->wq->name ) + __field( int, cpu ) + __field( u64, duration_us ) + ), + + TP_fast_assign( + __entry->work = work; + __entry->function = function; + __assign_str(workqueue); + __entry->cpu = pwq->pool->cpu; + __entry->duration_us = duration_us; + ), + + TP_printk("work struct=%p function=%ps workqueue=%s cpu=%d duration_us=%llu", + __entry->work, __entry->function, __get_str(workqueue), + __entry->cpu, __entry->duration_us) +); + #endif /* _TRACE_WORKQUEUE_H */ /* This part must be outside protection */ #include <trace/define_trace.h> + diff --git a/kernel/workqueue.c b/kernel/workqueue.c index 3c034cbc5bb3..bad1217cf45a 100644 --- a/kernel/workqueue.c +++ b/kernel/workqueue.c @@ -56,6 +56,7 @@ #include <linux/kvm_para.h> #include <linux/delay.h> #include <linux/irq_work.h> +#include <linux/math64.h> #include "workqueue_internal.h" @@ -1532,6 +1533,7 @@ void wq_worker_tick(struct task_struct *task) struct worker *worker = kthread_data(task); struct pool_workqueue *pwq = worker->current_pwq; struct worker_pool *pool = worker->pool; + u64 delta; if (!pwq) return; @@ -1572,6 +1574,11 @@ void wq_worker_tick(struct task_struct *task) pwq->stats[PWQ_STAT_CM_WAKEUP]++; raw_spin_unlock(&pool->lock); + + delta = READ_ONCE(worker->task->se.sum_exec_runtime) - worker->current_at; + trace_workqueue_cpu_intensive(pwq, worker->current_work, + worker->current_func, + div_u64(delta, NSEC_PER_USEC)); } /** -- 2.55.0
