[PATCH 4/5] timekeeping: Reinstate proportional correction of ntp_error
From: David Woodhouse
Date: Thu Oct 01 2026 - 16:22:20 EST
From: David Woodhouse <dwmw@xxxxxxxxxxxx>
The ±1 mult dithering is sufficient to keep ntp_error at zero once
it's already there, but it doesn't do much to reduce it once any
significant ntp_error has accumulated (e.g. from frequency or phase
adjustments) — it can typically only drain single-digit nanoseconds
per second. Commit dc491596f639 ("timekeeping: Rework frequency
adjustments to work better w/ nohz") removed a larger skew because in
a tickless kernel, it would remain in effect for a full idle period
and overshoot. Now that timekeeping_max_deferment() ensures that the
system wakes at the top of the second when skew is active, the
correction can be reinstated. If ntp_error exceeds an amount that a
single ±1 change to mult can drain within a minute, add an
additional bias to mult to close the gap.
Signed-off-by: David Woodhouse <dwmw@xxxxxxxxxxxx>
Assisted-by: LLM
---
include/linux/timekeeper_internal.h | 5 ++-
kernel/time/timekeeping.c | 55 ++++++++++++++++++++++++++---
2 files changed, 54 insertions(+), 6 deletions(-)
diff --git a/include/linux/timekeeper_internal.h b/include/linux/timekeeper_internal.h
index fe077d97b5f8..85d123bcb270 100644
--- a/include/linux/timekeeper_internal.h
+++ b/include/linux/timekeeper_internal.h
@@ -97,6 +97,8 @@ struct tk_read_base {
* @ntp_error_shift: Shift conversion between clock shifted nano seconds and
* ntp shifted nano seconds.
* @ntp_err_mult: Multiplication factor for scaled math conversion
+ * @err_drain: Extra mult bias repaying ntp_error beyond the
+ * dither's reach; sized at second boundaries
* @cs_tick_adj: Per-second adjustment handed to NTP via ntp_clear()
* accounting for the difference between the nominal
* NTP interval and the real time taken by the
@@ -186,7 +188,8 @@ struct timekeeper {
u64 ntp_tick;
s64 ntp_error;
u32 ntp_error_shift;
- u32 ntp_err_mult;
+ s32 ntp_err_mult;
+ s32 err_drain;
s64 cs_tick_adj;
u32 skip_second_overflow;
s64 skew_delta;
diff --git a/kernel/time/timekeeping.c b/kernel/time/timekeeping.c
index 986649cae46d..48d916a6c433 100644
--- a/kernel/time/timekeeping.c
+++ b/kernel/time/timekeeping.c
@@ -325,6 +325,15 @@ static inline void clocksource_disable_inline_read(void) { }
static inline void clocksource_enable_inline_read(void) { }
#endif
+/*
+ * The proportional ntp_error drain engages when the error exceeds
+ * TK_NTP_ERR_THRESH seconds' worth of the ±1 dither's delivery (so
+ * noise the dither handles alone never wakes the CPU), and then
+ * repays over ~TK_NTP_ERR_HORIZON seconds.
+ */
+#define TK_NTP_ERR_THRESH 60
+#define TK_NTP_ERR_HORIZON 10
+
/**
* tk_setup_internals - Set up internals to use clocksource clock.
*
@@ -422,6 +431,7 @@ static void tk_setup_internals(struct timekeeper *tk, struct clocksource *clock)
tk->tkr_mono.mult = clock->mult;
tk->tkr_raw.mult = clock->mult;
tk->ntp_err_mult = 0;
+ tk->err_drain = 0;
tk->skip_second_overflow = 0;
tk->skew_delta = 0;
@@ -830,6 +840,7 @@ static void timekeeping_update_from_shadow(struct tk_data *tkd, unsigned int act
if (action & TK_CLEAR_NTP) {
tk->ntp_error = 0;
+ tk->err_drain = 0;
ntp_clear(tk->id, tk->cs_tick_adj);
}
@@ -1977,11 +1988,11 @@ u64 timekeeping_max_deferment(void)
/*
* Skew introduced for phase offset is intended to be
- * recalculated each second. Ensure a tickless kernel
- * wakes up for that and does not overshoot, while skew
- * is being applied.
+ * recalculated each second, as is the proportional
+ * ntp_error drain. Ensure a tickless kernel wakes up
+ * for that and does not overshoot.
*/
- if (abs(tk->skew_delta) > 1) {
+ if (abs(tk->skew_delta) > 1 || (u32)tk->ntp_err_mult > 1) {
u64 nsec_in_sec = tk->tkr_mono.xtime_nsec >>
tk->tkr_mono.shift;
u64 remaining = NSEC_PER_SEC - min_t(u64, nsec_in_sec,
@@ -2251,6 +2262,7 @@ void timekeeping_resume(void)
tks->tkr_raw.cycle_last = cycle_now;
tks->ntp_error = 0;
+ tks->err_drain = 0;
timekeeping_suspended = 0;
timekeeping_update_from_shadow(&tk_core, TK_CLOCK_WAS_SET);
raw_spin_unlock_irqrestore(&tk_core.lock, flags);
@@ -2493,7 +2505,16 @@ static void timekeeping_adjust(struct timekeeper *tk, s64 offset)
* tick division, the clock will slow down. Otherwise it will stay
* ahead until the tick length changes to a non-divisible value.
*/
- tk->ntp_err_mult = tk->ntp_error > 0 ? 1 : 0;
+ /* Drop the ntp_error drain the moment it overshoots */
+ if (tk->err_drain &&
+ (((s64)tk->err_drain ^ tk->ntp_error) < 0 || !tk->ntp_error))
+ tk->err_drain = 0;
+
+ if (tk->err_drain)
+ tk->ntp_err_mult = tk->err_drain;
+ else
+ tk->ntp_err_mult = tk->ntp_error > 0 ? 1 : 0;
+
mult += tk->ntp_err_mult;
timekeeping_apply_adjustment(tk, offset, mult - tk->tkr_mono.mult);
@@ -2536,6 +2557,7 @@ static inline unsigned int accumulate_nsecs_to_secs(struct timekeeper *tk)
{
u64 nsecps = (u64)NSEC_PER_SEC << tk->tkr_mono.shift;
unsigned int clock_set = 0;
+ bool crossed = false;
while (tk->tkr_mono.xtime_nsec >= nsecps) {
int leap;
@@ -2568,6 +2590,29 @@ static inline unsigned int accumulate_nsecs_to_secs(struct timekeeper *tk)
clock_set = TK_CLOCK_WAS_SET;
}
+
+ crossed = true;
+ }
+
+ /*
+ * Core timekeeper only; aux timekeepers are not covered by the
+ * timekeeping_max_deferment() wakeup which re-sizes the drain
+ * each second before it can overshoot.
+ */
+ if (crossed && tk->id == TIMEKEEPER_CORE) {
+ /* Re-size the proportional ntp_error drain for this second */
+ s64 lsb_sec = ((s64)tk->cycle_interval <<
+ tk->ntp_error_shift) * NTP_INTERVAL_FREQ;
+
+ tk->err_drain = 0;
+ if (abs(tk->ntp_error) > lsb_sec * TK_NTP_ERR_THRESH) {
+ s32 headroom = tk->tkr_mono.clock->maxadj / 4;
+
+ tk->err_drain = clamp_t(s64,
+ div64_s64(tk->ntp_error,
+ lsb_sec * TK_NTP_ERR_HORIZON),
+ -headroom, headroom);
+ }
}
return clock_set;
}
--
2.43.0