[PATCH 6/6] memcg: group the fields of struct mem_cgroup_per_node
From: Shakeel Butt
Date: Fri Sep 04 2026 - 23:08:12 EST
Replace the two ad-hoc CACHELINE_PADDING members with named cache line
groups:
memcg_pn_read_mostly memcg, lruvec_stats_percpu, lruvec_stats,
shrinker_info, objcg
memcg_pn_lruvec lruvec
memcg_pn_write_hot lru_zone_size, iter, nmi slab stats
memcg_pn_cold orig_objcg, objcg_list
The group markers give the same isolation the padding did, but they
are named and the build now checks them.
lruvec still gets its own lines. Commit f59adcf59332 ("mm: memcg: add
cacheline padding after lruvec in mem_cgroup_per_node") showed why
that matters: lru_zone_size[] is written under lru_lock but read
without it by lruvec_lru_size(), so it must not share a line with
lruvec.
Splitting the cold fields out costs one extra cache line per node per
memcg.
No functional change.
Signed-off-by: Shakeel Butt <shakeel.butt@xxxxxxxxx>
---
include/linux/memcontrol.h | 32 ++++++++++++++++++++++----------
mm/memcontrol.c | 30 ++++++++++++++++++++++++++++++
2 files changed, 52 insertions(+), 10 deletions(-)
diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
index d0f3458f9250..e10a3eaae890 100644
--- a/include/linux/memcontrol.h
+++ b/include/linux/memcontrol.h
@@ -82,7 +82,8 @@ struct mem_cgroup_reclaim_iter {
* per-node information in memory controller.
*/
struct mem_cgroup_per_node {
- /* Keep the read-only fields at the start */
+ /* Set when the memcg is created, then only read. */
+ __cacheline_group_begin_aligned(memcg_pn_read_mostly);
struct mem_cgroup *memcg; /* Back pointer, we cannot */
/* use container_of */
@@ -91,14 +92,30 @@ struct mem_cgroup_per_node {
struct shrinker_info __rcu *shrinker_info;
struct obj_cgroup __rcu *objcg;
- CACHELINE_PADDING(_pad1_);
+ __cacheline_group_end_aligned(memcg_pn_read_mostly);
- /* Fields which get updated often at the end. */
+ /*
+ * Keep lruvec on its own lines. Sharing them with lru_zone_size[]
+ * regressed, see commit f59adcf59332 ("mm: memcg: add cacheline
+ * padding after lruvec in mem_cgroup_per_node").
+ */
+ __cacheline_group_begin_aligned(memcg_pn_lruvec);
struct lruvec lruvec;
- CACHELINE_PADDING(_pad2_);
+ __cacheline_group_end_aligned(memcg_pn_lruvec);
+
+ /* Written on every LRU update and on every reclaim iteration. */
+ __cacheline_group_begin_aligned(memcg_pn_write_hot);
unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS];
struct mem_cgroup_reclaim_iter iter;
+#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC
+ /* slab stats for nmi context */
+ atomic_t slab_reclaimable;
+ atomic_t slab_unreclaimable;
+#endif
+ __cacheline_group_end_aligned(memcg_pn_write_hot);
+ /* Touched only when the memcg is reparented or freed. */
+ __cacheline_group_begin_aligned(memcg_pn_cold);
/*
* orig_objcg preserves a pointer (and a reference) to the original
* objcg until the end of life of memcg.
@@ -106,12 +123,7 @@ struct mem_cgroup_per_node {
struct obj_cgroup *orig_objcg;
/* list of inherited objcgs, protected by objcg_lock */
struct list_head objcg_list;
-
-#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC
- /* slab stats for nmi context */
- atomic_t slab_reclaimable;
- atomic_t slab_unreclaimable;
-#endif
+ __cacheline_group_end_aligned(memcg_pn_cold);
};
struct mem_cgroup_threshold {
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index 4a5a30439a03..6976a60c911f 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -5796,6 +5796,36 @@ static void __init memcg_struct_check(void)
kmemcg_id);
CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly,
oom_group);
+
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_read_mostly, memcg);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_read_mostly, lruvec_stats_percpu);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_read_mostly, lruvec_stats);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_read_mostly, shrinker_info);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_read_mostly, objcg);
+
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_lruvec, lruvec);
+
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_write_hot, lru_zone_size);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_write_hot, iter);
+#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_write_hot, slab_reclaimable);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_write_hot, slab_unreclaimable);
+#endif
+
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_cold, orig_objcg);
+ CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node,
+ memcg_pn_cold, objcg_list);
}
int __init mem_cgroup_init(void)
--
2.53.0-Meta