[PATCH v5 3/7] mm/page_counter: introduce per-page_counter stock
From: Joshua Hahn
Date: Mon Aug 31 2026 - 15:03:44 EST
In order to avoid expensive hierarchy walks on every memcg charge and
limit check, memcontrol uses per-cpu stocks (memcg_stock_pcp) to cache
pre-charged pages and introduce a fast path to try_charge_memcg.
However, there are a few quirks with the current implementation that
can be improved upon.
First, each memcg_stock_pcp can only cache the charges of 7 memcgs
(NR_MEMCG_STOCK). When an 8th memcg wants to cache its charge on a CPU,
a victim memcg is chosen among the 7 cached memcgs and is evicted,
losing all cached charges.
Second, stock draining is per-CPU rather than per-memcg. That is,
when a memcg is under pressure and must retrieve all cached charges,
it iterates through every CPU and drains the stock charges of all
present memcgs. This means that one under-pressure memcg evicts the
caches of all co-cpu-resident memcg stock caches.
Finally, stock is tightly coupled with memcg, so adding new
page_counters to memcg is an unscalable operation where only one counter
gets to use the fastpath.
We can address all of these concerns by pushing stock caches down to the
page_counter level, and making each counter responsible for its own
charge.
Introduce struct page_counter_stock along with its allocation, free, and
per-CPU drain helpers.
No functional change intended.
Suggested-by: Johannes Weiner <hannes@xxxxxxxxxxx>
Signed-off-by: Joshua Hahn <joshua.hahnjy@xxxxxxxxx>
---
include/linux/page_counter.h | 16 +++++++
mm/page_counter.c | 90 ++++++++++++++++++++++++++++++++++++
2 files changed, 106 insertions(+)
diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h
index 89a083f16fbf7..c1fe331f34e7e 100644
--- a/include/linux/page_counter.h
+++ b/include/linux/page_counter.h
@@ -5,8 +5,11 @@
#include <linux/atomic.h>
#include <linux/cache.h>
#include <linux/limits.h>
+#include <linux/workqueue_types.h>
#include <asm/page.h>
+struct page_counter_stock;
+
struct page_counter {
/*
* Make sure 'usage' does not share cacheline with any other field in
@@ -41,6 +44,13 @@ struct page_counter {
unsigned long high;
unsigned long max;
struct page_counter *parent;
+ struct page_counter_stock __percpu *stock;
+ unsigned long batch;
+
+ /* make sure the work_struct is separate from the read most fields */
+ CACHELINE_PADDING(_pad3_);
+
+ struct work_struct drain_work;
} ____cacheline_internodealigned_in_smp;
#if BITS_PER_LONG == 32
@@ -61,6 +71,8 @@ static inline void page_counter_init(struct page_counter *counter,
counter->parent = parent;
counter->protection_support = protection_support;
counter->track_failcnt = false;
+ counter->stock = NULL;
+ counter->batch = 0;
}
static inline unsigned long page_counter_read(struct page_counter *counter)
@@ -99,6 +111,10 @@ static inline void page_counter_reset_watermark(struct page_counter *counter)
counter->watermark = usage;
}
+void page_counter_drain_cpu_stock(struct page_counter *counter, int cpu);
+void page_counter_alloc_stock(struct page_counter *counter, unsigned long batch);
+void page_counter_free_stock(struct page_counter *counter);
+
#if IS_ENABLED(CONFIG_MEMCG) || IS_ENABLED(CONFIG_CGROUP_DMEM)
void page_counter_calculate_protection(struct page_counter *root,
struct page_counter *counter,
diff --git a/mm/page_counter.c b/mm/page_counter.c
index a934619cc7bf7..3f61eba695518 100644
--- a/mm/page_counter.c
+++ b/mm/page_counter.c
@@ -8,11 +8,18 @@
#include <linux/page_counter.h>
#include <linux/atomic.h>
#include <linux/kernel.h>
+#include <linux/percpu.h>
#include <linux/string.h>
#include <linux/sched.h>
+#include <linux/spinlock.h>
#include <linux/bug.h>
#include <asm/page.h>
+struct page_counter_stock {
+ raw_spinlock_t lock;
+ unsigned long nr_pages;
+};
+
static bool track_protection(struct page_counter *c)
{
return c->protection_support;
@@ -295,6 +302,89 @@ int page_counter_memparse(const char *buf, const char *max,
return 0;
}
+/**
+ * page_counter_drain_cpu_stock - release @cpu's cached charges
+ * @counter: counter whose stock to drain
+ * @cpu: CPU whose stock is drained
+ */
+void page_counter_drain_cpu_stock(struct page_counter *counter, int cpu)
+{
+ struct page_counter_stock __percpu *stock = READ_ONCE(counter->stock);
+ struct page_counter_stock *pcp_stock;
+ unsigned long nr_pages;
+ unsigned long flags;
+
+ if (!stock)
+ return;
+
+ pcp_stock = per_cpu_ptr(stock, cpu);
+ raw_spin_lock_irqsave(&pcp_stock->lock, flags);
+ nr_pages = pcp_stock->nr_pages;
+ pcp_stock->nr_pages = 0;
+ raw_spin_unlock_irqrestore(&pcp_stock->lock, flags);
+
+ if (nr_pages)
+ page_counter_uncharge(counter, nr_pages);
+}
+
+/**
+ * page_counter_alloc_stock - allocate the percpu stock for a page_counter
+ * @counter: counter to allocate percpu stock for
+ * @batch: maximum number of pages a CPU may cache
+ *
+ * Failure to allocate is not fatal; @counter falls back to hierarchy charges.
+ * The caller must not (un)charge @counter concurrently with this call, and this
+ * must not be called twice on the same counter. A concurrent drain is fine
+ * since the stock is published with a release store the drain paths pair with.
+ *
+ * Context: Process context. May sleep, the percpu alloc uses GFP_KERNEL.
+ */
+void page_counter_alloc_stock(struct page_counter *counter, unsigned long batch)
+{
+ struct page_counter_stock __percpu *stock;
+ int cpu;
+
+ if (WARN_ON_ONCE(counter->stock))
+ return;
+
+ stock = alloc_percpu_gfp(struct page_counter_stock, GFP_KERNEL_ACCOUNT);
+ if (!stock)
+ return;
+
+ for_each_possible_cpu(cpu) {
+ struct page_counter_stock *pcp_stock = per_cpu_ptr(stock, cpu);
+
+ raw_spin_lock_init(&pcp_stock->lock);
+ }
+
+ counter->batch = batch;
+ /* Publish stock only after percpu allocs / inits are finished */
+ smp_store_release(&counter->stock, stock);
+}
+
+/**
+ * page_counter_free_stock - free @counter's percpu cached charge
+ * @counter: page_counter whose stock to free
+ *
+ * Caller must guarantee no (un)charge or drain of @counter is in flight or can
+ * start. memcg only calls this once the cgroup is dead and unreachable.
+ */
+void page_counter_free_stock(struct page_counter *counter)
+{
+ struct page_counter_stock __percpu *stock = counter->stock;
+ int cpu;
+
+ if (!stock)
+ return;
+
+ /* Stop greedy over-charging before the stock goes away */
+ counter->batch = 0;
+ for_each_possible_cpu(cpu)
+ page_counter_drain_cpu_stock(counter, cpu);
+
+ WRITE_ONCE(counter->stock, NULL);
+ free_percpu(stock);
+}
#if IS_ENABLED(CONFIG_MEMCG) || IS_ENABLED(CONFIG_CGROUP_DMEM)
/*
--
2.53.0-Meta