Re: [PATCH v11 1/2] hung_task: Reset warning budget when problem gets resolved
From: Bradley Morgan
Date: Sun Oct 04 2026 - 12:52:57 EST
On 4 October 2026 17:36:09 BST, Aaron Tomlin <atomlin@xxxxxxxxxxx> wrote:
>The sysctl hung_task_warnings currently holds both the configured warning
>limit and the remaining budget. Each detailed report decrements the
>sysctl, so once it reaches zero, the configured limit is lost and cannot
>be restored automatically.
>
>Keep sysctl_hung_task_warnings as the configured warning limit and make
>khungtaskd the sole owner of the remaining budget. A check that finds no
>hung tasks reloads the budget directly from the configured limit. A
>successful sysctl write publishes an atomic reset request, which
>khungtaskd consumes at the start of the next check.
>
>Suggested-by: Petr Mladek <pmladek@xxxxxxxx>
>Suggested-by: Lance Yang <lance.yang@xxxxxxxxx>
>Tested-by: Lance Yang <lance.yang@xxxxxxxxx>
>Reviewed-by: Lance Yang <lance.yang@xxxxxxxxx>
LGTM:
Reviewed-by: Bradley Morgan <brads@xxxxxxxxxxxxxx>
>Signed-off-by: Aaron Tomlin <atomlin@xxxxxxxxxxx>
>---
> Documentation/admin-guide/sysctl/kernel.rst | 5 ++-
> kernel/hung_task.c | 49 +++++++++++++++++----
> 2 files changed, 44 insertions(+), 10 deletions(-)
>
>diff --git a/Documentation/admin-guide/sysctl/kernel.rst b/Documentation/admin-guide/sysctl/kernel.rst
>index ffea61d448eb..c03369e234a9 100644
>--- a/Documentation/admin-guide/sysctl/kernel.rst
>+++ b/Documentation/admin-guide/sysctl/kernel.rst
>@@ -459,8 +459,9 @@ hung_task_warnings
> ==================
>
> The maximum number of warnings to report. During a check interval
>-if a hung task is detected, this value is decreased by 1.
>-When this value reaches 0, no more warnings will be reported.
>+if a hung task is detected, the internal warning budget is decreased by 1.
>+When this budget reaches 0, no more detailed warnings will be reported. The
>+warning budget is reset to the configured limit when no hung task is found.
> This file shows up if ``CONFIG_DETECT_HUNG_TASK`` is enabled.
>
> -1: report an infinite number of warnings.
>diff --git a/kernel/hung_task.c b/kernel/hung_task.c
>index 6fcc94ce4ca9..a5043188456d 100644
>--- a/kernel/hung_task.c
>+++ b/kernel/hung_task.c
>@@ -57,8 +57,20 @@ unsigned long __read_mostly sysctl_hung_task_timeout_secs = CONFIG_DEFAULT_HUNG_
> */
> static unsigned long __read_mostly sysctl_hung_task_check_interval_secs;
>
>+/*
>+ * Limit the number of printed hung tasks to prevent printing
>+ * the same or similar backtraces repeatedly.
>+ */
> static int __read_mostly sysctl_hung_task_warnings = 10;
>
>+/*
>+ * The number of hung tasks which still can be reported.
>+ * The budget gets restored to the original limit when
>+ * the previous stall is resolved.
>+ */
>+static int hung_task_warnings_budget = 10;
>+static atomic_t reset_hung_task_warnings = ATOMIC_INIT(0);
>+
> static int __read_mostly did_panic;
> static bool hung_task_call_panic;
>
>@@ -245,11 +257,11 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout,
> /*
> * The given task did not get scheduled for more than
> * CONFIG_DEFAULT_HUNG_TASK_TIMEOUT. Therefore, complain
>- * accordingly
>+ * accordingly with full details if the budget is not exhausted.
> */
>- if (sysctl_hung_task_warnings || hung_task_call_panic) {
>- if (sysctl_hung_task_warnings > 0)
>- sysctl_hung_task_warnings--;
>+ if (hung_task_warnings_budget || hung_task_call_panic) {
>+ if (hung_task_warnings_budget > 0)
>+ hung_task_warnings_budget--;
> pr_err("INFO: task %s:%d blocked%s for more than %ld seconds.\n",
> t->comm, t->pid, t->in_iowait ? " in I/O wait" : "",
> (jiffies - t->last_switch_time) / HZ);
>@@ -264,7 +276,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout,
> sched_show_task(t);
> debug_show_blocker(t, timeout);
>
>- if (!sysctl_hung_task_warnings)
>+ if (!hung_task_warnings_budget)
> pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n");
> }
>
>@@ -304,7 +316,7 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout)
> unsigned long last_break = jiffies;
> struct task_struct *g, *t;
> unsigned long this_round_count;
>- int need_warning = sysctl_hung_task_warnings;
>+ int need_warning;
> unsigned long si_mask = hung_task_si_mask;
>
> /*
>@@ -314,6 +326,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout)
> if (test_taint(TAINT_DIE) || did_panic)
> return;
>
>+ if (atomic_xchg_acquire(&reset_hung_task_warnings, 0))
>+ hung_task_warnings_budget =
>+ READ_ONCE(sysctl_hung_task_warnings);
>+ need_warning = hung_task_warnings_budget;
>+
> this_round_count = 0;
> rcu_read_lock();
> for_each_process_thread(g, t) {
>@@ -340,8 +357,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout)
> unlock:
> rcu_read_unlock();
>
>- if (!this_round_count)
>+ if (!this_round_count) {
>+ hung_task_warnings_budget =
>+ READ_ONCE(sysctl_hung_task_warnings);
> return;
>+ }
>
> if (need_warning || hung_task_call_panic) {
> si_mask |= SYS_INFO_LOCKS;
>@@ -425,6 +445,19 @@ static int proc_dohung_task_timeout_secs(const struct ctl_table *table, int writ
> return ret;
> }
>
>+static int proc_dohung_task_warnings(const struct ctl_table *table, int write,
>+ void *buffer,
>+ size_t *lenp, loff_t *ppos)
>+{
>+ int ret;
>+
>+ ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos);
>+ if (!ret && write)
>+ atomic_set_release(&reset_hung_task_warnings, 1);
>+
>+ return ret;
>+}
>+
> /*
> * This is needed for proc_doulongvec_minmax of sysctl_hung_task_timeout_secs
> * and hung_task_check_interval_secs
>@@ -480,7 +513,7 @@ static const struct ctl_table hung_task_sysctls[] = {
> .data = &sysctl_hung_task_warnings,
> .maxlen = sizeof(int),
> .mode = 0644,
>- .proc_handler = proc_dointvec_minmax,
>+ .proc_handler = proc_dohung_task_warnings,
> .extra1 = SYSCTL_NEG_ONE,
> },
> {
>
--- Thanks!
"I'm not a very positive person" - Linus torvalds