[PATCH bpf-next v4] selftests/bpf: Add bpf_proactive_reclaim test

From: Hui Zhu

Date: Fri Oct 09 2026 - 02:33:19 EST


From: Hui Zhu <zhuhui@xxxxxxxxxx>

bpf_proactive_reclaim() performs one bounded reclaim pass per call and,
unlike a write to memory.reclaim, does not retry until the goal is
reached.

Charge 32 MiB of page cache to a cgroup, ask for all of it in one call,
and check that the result is positive but well below the request: one
pass is capped at MEMCG_CHARGE_BATCH pages, far short of 32 MiB on both
4K and 64K page kernels.

The test deliberately covers only the kfunc itself. A full example of
the intended asynchronous use, where a BPF program watches one cgroup's
workingset refaults and reclaims another one from bpf_wq callbacks, is
maintained out of tree at [1], released under GPLv2.

Add CONFIG_MEMCG to the config fragment, without which mm/bpf_memcontrol.c
is not built at all.

[1] https://github.com/teawater/memcg-async-reclaim

Signed-off-by: Hui Zhu <zhuhui@xxxxxxxxxx>
---
Changelog:
v4:
According to the comment of Barry, test skips instead of failing on a
read-only mount.
v3:
According to the comment of Barry, use write() + fsync() to charge the
page cache instead of ftruncate() + read().
According to the comment of AI, skip instead of fail when no regular
filesystem is available for the workload file.
v2:
According to the comment of Alexei, remove the samples/bpf sample patch.
Ship it out-of-tree as a standalone project.

tools/testing/selftests/bpf/config | 1 +
.../bpf/prog_tests/memcg_proactive_reclaim.c | 130 ++++++++++++++++++
.../bpf/progs/memcg_proactive_reclaim.c | 34 +++++
3 files changed, 165 insertions(+)
create mode 100644 tools/testing/selftests/bpf/prog_tests/memcg_proactive_reclaim.c
create mode 100644 tools/testing/selftests/bpf/progs/memcg_proactive_reclaim.c

diff --git a/tools/testing/selftests/bpf/config b/tools/testing/selftests/bpf/config
index 2b883b388f90..5621ef94ad7e 100644
--- a/tools/testing/selftests/bpf/config
+++ b/tools/testing/selftests/bpf/config
@@ -57,6 +57,7 @@ CONFIG_LIRC=y
CONFIG_LIVEPATCH=y
CONFIG_LWTUNNEL=y
CONFIG_LWTUNNEL_BPF=y
+CONFIG_MEMCG=y
CONFIG_MODULE_SIG=y
CONFIG_MODULE_SRCVERSION_ALL=y
CONFIG_MODULE_UNLOAD=y
diff --git a/tools/testing/selftests/bpf/prog_tests/memcg_proactive_reclaim.c b/tools/testing/selftests/bpf/prog_tests/memcg_proactive_reclaim.c
new file mode 100644
index 000000000000..c22dcfabe856
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/memcg_proactive_reclaim.c
@@ -0,0 +1,130 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <test_progs.h>
+#include <limits.h>
+#include <linux/magic.h>
+#include <sys/statfs.h>
+#include <unistd.h>
+
+#include "cgroup_helpers.h"
+#include "memcg_proactive_reclaim.skel.h"
+
+#define CG_PATH "/memcg_proactive_reclaim"
+
+/*
+ * Large enough that a single reclaim pass cannot come close to it, so that
+ * "reclaimed less than was asked for" is not page-size noise: one pass is
+ * capped at MEMCG_CHARGE_BATCH pages, which is 256 KiB on 4K pages but 4 MiB
+ * on 64K pages.
+ */
+#define FILE_SIZE (32 * 1024 * 1024UL)
+#define BUF_SIZE (64 * 1024)
+
+struct reclaim_args {
+ __u64 cgroup_id;
+ __u64 size;
+};
+
+/*
+ * The data file has to sit on a writable regular filesystem: tmpfs pages are
+ * charged as shmem, so whether they can be reclaimed at all depends on swap
+ * being available, and a read-only mount cannot host the file at all (EROFS
+ * applies to root too). /tmp is tmpfs on many systems, and test_progs is
+ * routinely run from a tmpfs working directory, so both candidates need the
+ * check.
+ */
+static const char *workload_dir(void)
+{
+ static const char * const dirs[] = { "/tmp", "." };
+ struct statfs st;
+ int i;
+
+ for (i = 0; i < ARRAY_SIZE(dirs); i++)
+ if (!statfs(dirs[i], &st) && st.f_type != TMPFS_MAGIC &&
+ st.f_type != RAMFS_MAGIC && !access(dirs[i], W_OK))
+ return dirs[i];
+
+ return NULL;
+}
+
+void test_memcg_proactive_reclaim(void)
+{
+ struct memcg_proactive_reclaim *skel = NULL;
+ struct reclaim_args args = {};
+
+ LIBBPF_OPTS(bpf_test_run_opts, opts,
+ .ctx_in = &args,
+ .ctx_size_in = sizeof(args));
+
+ char data_file[PATH_MAX];
+ static char buf[BUF_SIZE];
+ __u64 cgroup_id;
+ const char *dir;
+ off_t off;
+ int cg_fd = -1, data_fd = -1, err;
+
+ /*
+ * Both candidate directories can be unusable: tmpfs (initramfs-based
+ * CI, for instance) or a read-only mount. That is an environment
+ * limitation, not a test failure.
+ */
+ dir = workload_dir();
+ if (!dir) {
+ test__skip();
+ return;
+ }
+
+ snprintf(data_file, sizeof(data_file),
+ "%s/memcg_proactive_reclaim_XXXXXX", dir);
+ data_fd = mkstemp(data_file);
+ if (!ASSERT_GE(data_fd, 0, "mkstemp"))
+ return;
+
+ cg_fd = cgroup_setup_and_join(CG_PATH);
+ if (!ASSERT_OK_FD(cg_fd, "cgroup_setup_and_join"))
+ goto out;
+
+ cgroup_id = get_cgroup_id(CG_PATH);
+ if (!ASSERT_GT(cgroup_id, 0, "get_cgroup_id"))
+ goto out;
+
+ skel = memcg_proactive_reclaim__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "open_and_load"))
+ goto out;
+
+ /* Charge FILE_SIZE of clean page cache to the cgroup. */
+ for (off = 0; off < (off_t)FILE_SIZE; off += sizeof(buf))
+ if (!ASSERT_EQ(write(data_fd, buf, sizeof(buf)),
+ (ssize_t)sizeof(buf), "write"))
+ goto out;
+ if (!ASSERT_OK(fsync(data_fd), "fsync"))
+ goto out;
+
+ args.cgroup_id = cgroup_id;
+ args.size = FILE_SIZE;
+ skel->bss->reclaimed = 0;
+ err = bpf_prog_test_run_opts(bpf_program__fd(skel->progs.memcg_proactive_reclaim),
+ &opts);
+ if (!ASSERT_OK(err, "test_run"))
+ goto out;
+ if (!ASSERT_EQ(opts.retval, 0, "retval"))
+ goto out;
+
+ /*
+ * A single call is a single bounded pass: it reclaims something, but
+ * stops well short of the requested size instead of retrying until the
+ * goal is reached the way a write to memory.reclaim does.
+ */
+ ASSERT_GT(skel->bss->reclaimed, 0, "reclaimed");
+ ASSERT_LT(skel->bss->reclaimed, (__s64)FILE_SIZE, "single pass");
+
+out:
+ if (skel)
+ memcg_proactive_reclaim__destroy(skel);
+ if (cg_fd >= 0)
+ close(cg_fd);
+ if (data_fd >= 0) {
+ close(data_fd);
+ unlink(data_file);
+ }
+ cleanup_cgroup_environment();
+}
diff --git a/tools/testing/selftests/bpf/progs/memcg_proactive_reclaim.c b/tools/testing/selftests/bpf/progs/memcg_proactive_reclaim.c
new file mode 100644
index 000000000000..b551192526fc
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/memcg_proactive_reclaim.c
@@ -0,0 +1,34 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+
+struct reclaim_args {
+ __u64 cgroup_id;
+ __u64 size;
+};
+
+/* Signed, because bpf_test_run_opts.retval is a __u32. */
+__s64 reclaimed;
+
+SEC("syscall")
+int memcg_proactive_reclaim(struct reclaim_args *ctx)
+{
+ struct mem_cgroup *memcg;
+ struct cgroup *cgrp;
+
+ cgrp = bpf_cgroup_from_id(ctx->cgroup_id);
+ if (!cgrp)
+ return 0;
+
+ memcg = bpf_get_mem_cgroup(&cgrp->self);
+ if (memcg) {
+ reclaimed = bpf_proactive_reclaim(memcg, ctx->size, -1);
+ bpf_put_mem_cgroup(memcg);
+ }
+ bpf_cgroup_release(cgrp);
+
+ return 0;
+}
+
+char _license[] SEC("license") = "GPL";
--
2.43.0