[PATCH 2/3] workqueue: Add workqueue_mayday and workqueue_rescued tracepoints

From: Aaron Tomlin

Date: Sat Aug 29 2026 - 19:06:23 EST


When a worker pool fails to create a new worker thread within
MAYDAY_INTERVAL (e.g., typically under severe memory pressure where
memory reclaim paths depend on pending work items), send_mayday()
signals distress to the workqueue's rescuer thread. The rescuer then
takes over processing the pending work items via assign_rescuer_work().

While pwq->stats[PWQ_STAT_MAYDAY] and pwq->stats[PWQ_STAT_RESCUED] track
these occurrences cumulatively, there is currently no event-driven
mechanism to observe exactly when mayday distress occurs or which work
items require rescue.

Add two new tracepoints namely workqueue_mayday and workqueue_rescued to
make distress and rescuer execution easily observable (e.g., via Ftrace
or eBPF).

Signed-off-by: Aaron Tomlin <atomlin@xxxxxxxxxxx>
---
include/trace/events/workqueue.h | 68 ++++++++++++++++++++++++++++++++
kernel/workqueue.c | 2 +
2 files changed, 70 insertions(+)

diff --git a/include/trace/events/workqueue.h b/include/trace/events/workqueue.h
index ef0f3a4b73a8..013cfa472f6d 100644
--- a/include/trace/events/workqueue.h
+++ b/include/trace/events/workqueue.h
@@ -165,6 +165,74 @@ TRACE_EVENT(workqueue_cpu_intensive,
__entry->cpu, __entry->duration_us)
);

+/**
+ * workqueue_mayday - called when a pool_workqueue sends mayday to rescuer
+ * @pwq: pointer to struct pool_workqueue
+ *
+ * This event occurs when a worker pool fails to create a new worker
+ * within MAYDAY_INTERVAL and requests the workqueue's rescuer thread to
+ * process pending works.
+ */
+TRACE_EVENT(workqueue_mayday,
+
+ TP_PROTO(struct pool_workqueue *pwq),
+
+ TP_ARGS(pwq),
+
+ TP_STRUCT__entry(
+ __string( workqueue, pwq->wq->name )
+ __field( int, pool_id )
+ __field( int, cpu )
+ __field( int, nr_active )
+ ),
+
+ TP_fast_assign(
+ __assign_str(workqueue);
+ __entry->pool_id = pwq->pool->id;
+ __entry->cpu = pwq->pool->cpu;
+ __entry->nr_active = pwq->nr_active;
+ ),
+
+ TP_printk("workqueue=%s pool_id=%d cpu=%d nr_active=%d",
+ __get_str(workqueue), __entry->pool_id, __entry->cpu,
+ __entry->nr_active)
+);
+
+/**
+ * workqueue_rescued - called when a work item is assigned to a rescuer
+ * @pwq: pointer to struct pool_workqueue
+ * @work: pointer to struct work_struct
+ * @function: pointer to worker function
+ *
+ * This event occurs when a work item is claimed by a rescuer thread
+ * to guarantee forward progress.
+ */
+TRACE_EVENT(workqueue_rescued,
+
+ TP_PROTO(struct pool_workqueue *pwq, struct work_struct *work,
+ work_func_t function),
+
+ TP_ARGS(pwq, work, function),
+
+ TP_STRUCT__entry(
+ __field( void *, work )
+ __field( void *, function )
+ __string( workqueue, pwq->wq->name )
+ __field( int, cpu )
+ ),
+
+ TP_fast_assign(
+ __entry->work = work;
+ __entry->function = function;
+ __assign_str(workqueue);
+ __entry->cpu = pwq->pool->cpu;
+ ),
+
+ TP_printk("work struct=%p function=%ps workqueue=%s cpu=%d",
+ __entry->work, __entry->function, __get_str(workqueue),
+ __entry->cpu)
+);
+
#endif /* _TRACE_WORKQUEUE_H */

/* This part must be outside protection */
diff --git a/kernel/workqueue.c b/kernel/workqueue.c
index 6ee644155b82..6f6fe2068389 100644
--- a/kernel/workqueue.c
+++ b/kernel/workqueue.c
@@ -3122,6 +3122,7 @@ static void send_mayday(struct pool_workqueue *pwq)
list_add_tail(&pwq->mayday_node, &wq->maydays);
wake_up_process(wq->rescuer->task);
pwq->stats[PWQ_STAT_MAYDAY]++;
+ trace_workqueue_mayday(pwq);
}
}

@@ -3619,6 +3620,7 @@ static bool assign_rescuer_work(struct pool_workqueue *pwq, struct worker *rescu
list_for_each_entry_safe_from(work, n, &pool->worklist, entry) {
if (get_work_pwq(work) == pwq && assign_work(work, rescuer, &n)) {
pwq->stats[PWQ_STAT_RESCUED]++;
+ trace_workqueue_rescued(pwq, work, work->func);
/* put the cursor for next search */
list_move_tail(&cursor->entry, &n->entry);
return true;
--
2.55.0