[PATCH] rcu: Advance callbacks for expedited GP completion in note_gp_changes()

From: Puranjay Mohan

Date: Fri Jul 24 2026 - 10:26:25 EST


When rcu_pending() triggers rcu_core(), the callback advancement path
through note_gp_changes() -> __note_gp_changes() bails out when
rdp->gp_seq =3D=3D rnp->gp_seq (no normal GP change). Since expedited GPs d=
o
not update rnp->gp_seq, rcu_advance_cbs() is never reached from there and
callbacks satisfied by an expedited GP would otherwise remain stuck in
RCU_WAIT_TAIL until the next normal GP.

This is currently handled by a dedicated advancement block in rcu_core()
that polls the callback list and advances under a trylock. But callback
advancement no longer depends on the leaf-node grace-period delta; it is
driven by the grace-period state stored in the callback list, which
tracks both normal and expedited GPs. __note_gp_changes() is the natural
home for it, and hosting it there lets every note_gp_changes() caller
benefit from expedited completions rather than just rcu_core().

Move the advancement into __note_gp_changes(): trigger rcu_advance_cbs()
whenever rcu_segcblist_nextgp() confirmed with
poll_state_synchronize_rcu_full_unordered() reports a completed grace
period, and add need_note_gp_changes() so the lockless preamble takes the
lock for an expedited-only completion instead of short-circuiting on
rdp->gp_seq =3D=3D rnp->gp_seq. Remove the now-redundant rcu_core() block.

The quiescent-state bookkeeping stays keyed to an actual rnp->gp_seq
change, so an expedited completion never clears a still-pending
core_needs_qs and stalls the normal grace period.

Signed-off-by: Puranjay Mohan <puranjay@xxxxxxxxxx>
---
kernel/rcu/tree.c | 56 ++++++++++++++++++++++++-----------------------
kernel/rcu/tree.h | 1 +
2 files changed, 30 insertions(+), 27 deletions(-)

diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c
index 21b6ce1dffb63..0e700d0ecf27e 100644
--- a/kernel/rcu/tree.c
+++ b/kernel/rcu/tree.c
@@ -1270,27 +1270,33 @@ static bool __note_gp_changes(struct rcu_node
*rnp, struct rcu_data *rdp)
{
bool ret =3D false;
bool need_qs;
+ struct rcu_gp_seq gp_state;
const bool offloaded =3D rcu_rdp_is_offloaded(rdp);
+ bool completed =3D rcu_seq_completed_gp(rdp->gp_seq, rnp->gp_seq) |=
|
+ unlikely(rdp->gpwrap);

raw_lockdep_assert_held_rcu_node(rnp);

- if (rdp->gp_seq =3D=3D rnp->gp_seq)
- return false; /* Nothing to do. */
-
/* Handle the ends of any preceding grace periods first. */
- if (rcu_seq_completed_gp(rdp->gp_seq, rnp->gp_seq) ||
- unlikely(rdp->gpwrap)) {
+ if (completed ||
+ (rcu_segcblist_nextgp(&rdp->cblist, &gp_state) &&
+ poll_state_synchronize_rcu_full_unordered(&gp_state))) {
if (!offloaded)
ret =3D rcu_advance_cbs(rnp, rdp); /* Advance CBs. =
*/
- rdp->core_needs_qs =3D false;
- trace_rcu_grace_period(rcu_state.name, rdp->gp_seq,
TPS("cpuend"));
- } else {
+ if (completed) {
+ rdp->core_needs_qs =3D false;
+ trace_rcu_grace_period(rcu_state.name,
rdp->gp_seq, TPS("cpuend"));
+ }
+ } else if (rdp->gp_seq !=3D rnp->gp_seq) {
if (!offloaded)
ret =3D rcu_accelerate_cbs(rnp, rdp); /* Recent CBs=
. */
if (rdp->core_needs_qs)
rdp->core_needs_qs =3D !!(rnp->qsmask & rdp->grpmas=
k);
}

+ if (rdp->gp_seq =3D=3D rnp->gp_seq)
+ return ret; /* Nothing else to do. */
+
/* Now handle the beginnings of any new-to-this-CPU grace periods. =
*/
if (rcu_seq_new_gp(rdp->gp_seq, rnp->gp_seq) ||
unlikely(rdp->gpwrap)) {
@@ -1315,6 +1321,20 @@ static bool __note_gp_changes(struct rcu_node
*rnp, struct rcu_data *rdp)
return ret;
}

+static bool need_note_gp_changes(struct rcu_data *rdp)
+{
+ struct rcu_gp_seq gp_state;
+ struct rcu_node *rnp =3D rdp->mynode;
+
+ if (rdp->gp_seq !=3D rcu_seq_current(&rnp->gp_seq) ||
+ unlikely(READ_ONCE(rdp->gpwrap)))
+ return true;
+
+ /* Has a grace period a callback is waiting on completed? */
+ return rcu_segcblist_nextgp(&rdp->cblist, &gp_state) &&
+ poll_state_synchronize_rcu_full_unordered(&gp_state);
+}
+
static void note_gp_changes(struct rcu_data *rdp)
{
unsigned long flags;
@@ -1323,8 +1343,7 @@ static void note_gp_changes(struct rcu_data *rdp)

local_irq_save(flags);
rnp =3D rdp->mynode;
- if ((rdp->gp_seq =3D=3D rcu_seq_current(&rnp->gp_seq) &&
- !unlikely(READ_ONCE(rdp->gpwrap))) || /* w/out lock. */
+ if (!need_note_gp_changes(rdp) || /* w/out lock. */
!raw_spin_trylock_rcu_node(rnp)) { /* irqs already off, so late=
r. */
local_irq_restore(flags);
return;
@@ -2886,23 +2905,6 @@ static __latent_entropy void rcu_core(void)
/* Update RCU state based on any recent quiescent states. */
rcu_check_quiescent_state(rdp);

- /* Advance callbacks if an expedited GP has completed. */
- if (!rcu_rdp_is_offloaded(rdp) &&
rcu_segcblist_is_enabled(&rdp->cblist)) {
- struct rcu_gp_seq gp_state;
-
- if (rcu_segcblist_nextgp(&rdp->cblist, &gp_state) &&
- poll_state_synchronize_rcu_full(&gp_state)) {
- guard(irqsave)();
- if (raw_spin_trylock_rcu_node(rnp)) {
- bool needwake =3D rcu_advance_cbs(rnp, rdp)=
;
-
- raw_spin_unlock_rcu_node(rnp);
- if (needwake)
- rcu_gp_kthread_wake();
- }
- }
- }
-
/* No grace period and unregistered callbacks? */
if (!rcu_gp_in_progress() &&
rcu_segcblist_is_enabled(&rdp->cblist) &&
!rcu_rdp_is_offloaded(rdp)) {
diff --git a/kernel/rcu/tree.h b/kernel/rcu/tree.h
index eedfa43059e80..962f86afc3c9a 100644
--- a/kernel/rcu/tree.h
+++ b/kernel/rcu/tree.h
@@ -521,6 +521,7 @@ static void rcu_nocb_unlock(struct rcu_data *rdp);
static void rcu_nocb_unlock_irqrestore(struct rcu_data *rdp,
unsigned long flags);
static void rcu_lockdep_assert_cblist_protected(struct rcu_data *rdp);
+static bool poll_state_synchronize_rcu_full_unordered(struct rcu_gp_seq *g=
sp);
#ifdef CONFIG_RCU_NOCB_CPU
static void __init rcu_organize_nocb_kthreads(void);


base-commit: dc597bcabf31a58f8c1aca01677a6af0b3307398
--
2.53.0-Meta

-- 8< --

Thanks,
Puranjay