[PATCH v4 1/2] ext4: fix shrinker scan budget accounting in ext4_es_scan()
From: Qiliang Yuan
Date: Thu Oct 01 2026 - 03:48:38 EST
do_shrink_slab() reads count_objects() once per invocation to derive
a one-shot scan budget, then calls scan_objects() repeatedly until
that budget is exhausted or scan_objects() returns SHRINK_STOP.
include/linux/shrinker.h documents that scan_objects() "should track
its actual progress" in sc->nr_scanned, so do_shrink_slab() can tell
when there is nothing left to examine and stop early.
ext4_es_scan() never updates sc->nr_scanned, so it defaults to the
full sc->nr_to_scan on every call. do_shrink_slab() therefore always
believes a full batch was examined, regardless of what __es_shrink()
actually did, and keeps calling scan_objects() until the budget
derived from the (possibly stale) percpu extent_status count is
drained, even after sbi->s_es_list has nothing left to examine.
Make __es_shrink() report the number of extent_status objects it
actually examined through a new nr_scanned output parameter, derived
from the existing per-extent nr_to_scan counter that
es_reclaim_extents() already decrements as it walks the tree. Have
ext4_es_scan() copy this into sc->nr_scanned, and return SHRINK_STOP
once it comes back zero. Key this off nr_scanned rather than
nr_shrunk: a batch that finds every extent still referenced
correctly ages them without freeing any, and that is real progress
nr_shrunk would miss.
nr_scanned can come back zero for more than a genuinely empty list:
es_stats_shk_cnt is a stale percpu count, an inode can be
momentarily skipped (precached, lock contended), or durably have
nothing shrinkable (i_es_shk_nr == 0) without es_reclaim_extents()
ever touching nr_to_scan. Treat all of these alike: not stopping
doesn't help the transient cases either, since do_shrink_slab()'s
budget only decrements by nr_scanned, so retrying forever at
nr_scanned == 0 never terminates.
Tested by fallocate(2)-ing 10000 4K files (to populate the shrinker
with reclaimable unwritten extents without also exercising the
extent_status "referenced" second-chance path, which needs a
separate two-pass accounting of its own) and triggering
"echo 2 > /proc/sys/vm/drop_caches", while tracing the
ext4_es_shrink* tracepoints:
total scan_objects() calls with
calls nr_shrunk == 0
before this patch 429 189 (44%)
after this patch 239 1 (0.4%)
Fixes: 1ab6c4997e04 ("fs: convert fs shrinkers to new scan/count API")
Cc: stable@xxxxxxxxxxxxxxx
Signed-off-by: Qiliang Yuan <odys.yuan@xxxxxxxxx>
Reviewed-by: Jan Kara <jack@xxxxxxx>
---
fs/ext4/extents_status.c | 32 ++++++++++++++++++++++++++++----
1 file changed, 28 insertions(+), 4 deletions(-)
diff --git a/fs/ext4/extents_status.c b/fs/ext4/extents_status.c
index 6e4a191e82191..b8821b93693a9 100644
--- a/fs/ext4/extents_status.c
+++ b/fs/ext4/extents_status.c
@@ -184,7 +184,7 @@ static int __es_remove_extent(struct inode *inode, ext4_lblk_t lblk,
struct extent_status *prealloc);
static int es_reclaim_extents(struct ext4_inode_info *ei, int *nr_to_scan);
static int __es_shrink(struct ext4_sb_info *sbi, int nr_to_scan,
- struct ext4_inode_info *locked_ei);
+ struct ext4_inode_info *locked_ei, int *nr_scanned);
static int __revise_pending(struct inode *inode, ext4_lblk_t lblk,
ext4_lblk_t len,
struct pending_reservation **prealloc);
@@ -1670,7 +1670,7 @@ void ext4_es_remove_extent(struct inode *inode, ext4_lblk_t lblk,
}
static int __es_shrink(struct ext4_sb_info *sbi, int nr_to_scan,
- struct ext4_inode_info *locked_ei)
+ struct ext4_inode_info *locked_ei, int *nr_scanned)
{
struct ext4_inode_info *ei;
struct ext4_es_stats *es_stats;
@@ -1679,6 +1679,7 @@ static int __es_shrink(struct ext4_sb_info *sbi, int nr_to_scan,
int nr_to_walk;
int nr_shrunk = 0;
int retried = 0, nr_skipped = 0;
+ int orig_nr_to_scan = nr_to_scan;
es_stats = &sbi->s_es_stats;
start_time = ktime_get();
@@ -1738,6 +1739,8 @@ static int __es_shrink(struct ext4_sb_info *sbi, int nr_to_scan,
nr_shrunk = es_reclaim_extents(locked_ei, &nr_to_scan);
out:
+ *nr_scanned = orig_nr_to_scan - nr_to_scan;
+
scan_time = ktime_to_ns(ktime_sub(ktime_get(), start_time));
if (likely(es_stats->es_stats_scan_time))
es_stats->es_stats_scan_time = (scan_time +
@@ -1774,15 +1777,36 @@ static unsigned long ext4_es_scan(struct shrinker *shrink,
{
struct ext4_sb_info *sbi = shrink->private_data;
int nr_to_scan = sc->nr_to_scan;
- int ret, nr_shrunk;
+ int ret, nr_shrunk, nr_scanned;
ret = percpu_counter_read_positive(&sbi->s_es_stats.es_stats_shk_cnt);
trace_ext4_es_shrink_scan_enter(sbi->s_sb, nr_to_scan, ret);
- nr_shrunk = __es_shrink(sbi, nr_to_scan, NULL);
+ nr_shrunk = __es_shrink(sbi, nr_to_scan, NULL, &nr_scanned);
+ sc->nr_scanned = nr_scanned;
ret = percpu_counter_read_positive(&sbi->s_es_stats.es_stats_shk_cnt);
trace_ext4_es_shrink_scan_exit(sbi->s_sb, nr_shrunk, ret);
+
+ /*
+ * Key SHRINK_STOP off nr_scanned (extents actually examined), not
+ * nr_shrunk (extents actually freed): a batch that finds every
+ * extent still referenced correctly ages them without freeing
+ * any, and that is real progress nr_shrunk would miss.
+ *
+ * nr_scanned can come back 0 for more than a genuinely empty
+ * sbi->s_es_list: es_stats_shk_cnt is a stale percpu count, an
+ * inode can be momentarily skipped (precached, lock contended),
+ * or durably have nothing shrinkable (i_es_shk_nr == 0) without
+ * es_reclaim_extents() ever touching nr_to_scan. Treat all of
+ * these alike: not stopping doesn't help the transient cases
+ * either, since do_shrink_slab()'s budget only decrements by
+ * nr_scanned, so retrying forever at nr_scanned == 0 never
+ * terminates.
+ */
+ if (nr_scanned == 0)
+ return SHRINK_STOP;
+
return nr_shrunk;
}
--
2.43.0