[RFC PATCH bpf-next 12/12] selftests/bpf: exercise the lockless path ancestor iterator

Justin Suess utilityemal77 at gmail.com
Tue Oct 6 00:20:19 UTC 2026


Walk the same ancestry four ways and require the position counts to
agree: lockless from a non-sleepable program, where the RCU critical
section is implicit; lockless under an explicit bpf_rcu_read_lock();
referenced; and the hybrid that walks lockless to the second position,
hands it over to a referenced iteration, and resumes there - the
escalated position therefore being walked twice, once per mode.

The sleepable work the escalation exists for (d_path, an xattr read
through the position's dentry) runs on the resumed iteration's first
position, after bpf_rcu_read_unlock(), which is the only place a
sleepable kfunc can run at all.

Signed-off-by: Justin Suess <utilityemal77 at gmail.com>
---
 .../selftests/bpf/prog_tests/path_ancestors.c | 33 ++++++-
 .../selftests/bpf/progs/path_ancestors.c      | 89 ++++++++++++++++++-
 2 files changed, 118 insertions(+), 4 deletions(-)

diff --git a/tools/testing/selftests/bpf/prog_tests/path_ancestors.c b/tools/testing/selftests/bpf/prog_tests/path_ancestors.c
index 2de79673a13b..ce1ded844c3a 100644
--- a/tools/testing/selftests/bpf/prog_tests/path_ancestors.c
+++ b/tools/testing/selftests/bpf/prog_tests/path_ancestors.c
@@ -2,6 +2,7 @@
 /* Copyright (c) 2026 Justin Suess */
 
 #include <sys/stat.h>
+#include <sys/xattr.h>
 #include <stdlib.h>
 #include <unistd.h>
 #include <test_progs.h>
@@ -12,6 +13,8 @@ void test_path_ancestors(void)
 	char base[] = "/tmp/path_ancestors_XXXXXX";
 	struct path_ancestors *skel = NULL;
 	char suba[280], subb[280];
+	bool xattr_works;
+	int err;
 
 	if (!ASSERT_OK_PTR(mkdtemp(base), "mkdtemp"))
 		return;
@@ -20,6 +23,12 @@ void test_path_ancestors(void)
 	if (!ASSERT_OK(mkdir(suba, 0755), "mkdir_a"))
 		goto out_rm;
 
+	/* Read back by the program at the escalated position (== base). */
+	err = setxattr(base, "user.walk", "hello", 6, 0);
+	xattr_works = !err;
+	if (err && errno != EOPNOTSUPP && !ASSERT_OK(err, "setxattr"))
+		goto out_rm;
+
 	skel = path_ancestors__open_and_load();
 	if (!ASSERT_OK_PTR(skel, "open_and_load"))
 		goto out_rm;
@@ -31,14 +40,34 @@ void test_path_ancestors(void)
 	if (!ASSERT_OK(mkdir(subb, 0755), "mkdir_b"))
 		goto out;
 
-	/* suba, base, /tmp, / at least. */
-	ASSERT_GE(skel->bss->ref_count, 3, "ref_count");
+	ASSERT_EQ(skel->bss->test_err, 0, "test_err");
+	ASSERT_EQ(skel->bss->escalate_err, 0, "escalate_err");
+	/* suba, base, /tmp, / at least; equality across modes is the point. */
+	ASSERT_GE(skel->bss->rcu_count, 3, "rcu_count");
+	ASSERT_EQ(skel->bss->ref_count, skel->bss->rcu_count, "ref_vs_rcu");
+	ASSERT_EQ(skel->bss->rcu_ns_count, skel->bss->rcu_count,
+		  "nonsleepable_vs_rcu");
+	/*
+	 * The escalated position is walked twice: once lockless, then again
+	 * as the resumed referenced iteration's first position.
+	 */
+	ASSERT_EQ(skel->bss->hybrid_count, skel->bss->rcu_count + 1,
+		  "hybrid_vs_rcu");
+	ASSERT_EQ(skel->bss->retry_flags, 0, "no_retry");
 	ASSERT_EQ(skel->bss->ref_flags, 0, "ref_flags");
 
 	/* The acquired second position, used after its step was taken. */
 	ASSERT_STREQ(skel->bss->second_path, base, "second_path");
 	ASSERT_EQ(skel->bss->second_len, strlen(base) + 1, "second_len");
 
+	/* The escalated position is the walk's second one: base. */
+	ASSERT_STREQ(skel->bss->escalated_path, base, "escalated_path");
+	ASSERT_EQ(skel->bss->escalated_len, strlen(base) + 1, "escalated_len");
+	if (xattr_works) {
+		ASSERT_EQ(skel->bss->xattr_ret, 6, "xattr_len");
+		ASSERT_STREQ(skel->bss->xattr_value, "hello", "xattr_value");
+	}
+
 out:
 	path_ancestors__destroy(skel);
 out_rm:
diff --git a/tools/testing/selftests/bpf/progs/path_ancestors.c b/tools/testing/selftests/bpf/progs/path_ancestors.c
index af6b777e8bec..50ce0ce163dd 100644
--- a/tools/testing/selftests/bpf/progs/path_ancestors.c
+++ b/tools/testing/selftests/bpf/progs/path_ancestors.c
@@ -11,29 +11,74 @@ char _license[] SEC("license") = "GPL";
 
 __u32 monitored_pid;
 
+int rcu_count;		/* positions seen by the pure lockless walk */
+int rcu_ns_count;	/* ditto, from the non-sleepable program */
 int ref_count;		/* positions seen by the pure referenced walk */
+int hybrid_count;	/* positions seen by the lockless+escalate walk */
+int retry_flags;	/* BPF_PATH_ANCESTORS_RETRY observations */
 int ref_flags;		/* pos flags seen by the referenced walk */
 int second_len;		/* d_path length of the walk's second position */
+int xattr_ret;		/* xattr read at the escalated position */
+int escalated_len;	/* d_path length of the escalated position */
+int escalate_err;	/* bpf_path_ancestors_legitimize() result */
+int test_err;
 char second_path[256];
+char escalated_path[256];
+char xattr_value[16];
 
 static bool monitored(void)
 {
 	return (bpf_get_current_pid_tgid() >> 32) == monitored_pid;
 }
 
+/*
+ * Lockless walk from a non-sleepable program: the RCU critical section is
+ * implicit, no bpf_rcu_read_lock() needed.
+ */
+SEC("lsm/path_mkdir")
+int BPF_PROG(rcu_nonsleepable, const struct path *dir, struct dentry *dentry,
+	     umode_t mode)
+{
+	struct bpf_iter_path_ancestors_rcu rit;
+
+	if (!monitored())
+		return 0;
+
+	bpf_iter_path_ancestors_rcu_new(&rit, (struct path *)dir, 0);
+	while (bpf_iter_path_ancestors_rcu_next(&rit))
+		rcu_ns_count++;
+	retry_flags |= bpf_path_ancestors_rcu_pos_flags(&rit);
+	bpf_iter_path_ancestors_rcu_destroy(&rit);
+	return 0;
+}
+
 SEC("lsm.s/path_mkdir")
 int BPF_PROG(walk_modes, const struct path *dir, struct dentry *dentry,
 	     umode_t mode)
 {
+	struct bpf_iter_path_ancestors_rcu rit;
 	struct bpf_iter_path_ancestors it;
+	struct bpf_dynptr value_ptr;
 	struct path *pos;
 
 	if (!monitored())
 		return 0;
 
 	/*
-	 * Referenced walk: every position comes acquired, so it stays valid
-	 * for sleepable work and past the step that yielded it.
+	 * Mode 1: pure lockless, under an explicit RCU critical section.
+	 * Positions are borrowed, so nothing is released here.
+	 */
+	bpf_rcu_read_lock();
+	bpf_iter_path_ancestors_rcu_new(&rit, (struct path *)dir, 0);
+	while (bpf_iter_path_ancestors_rcu_next(&rit))
+		rcu_count++;
+	retry_flags |= bpf_path_ancestors_rcu_pos_flags(&rit);
+	bpf_iter_path_ancestors_rcu_destroy(&rit);
+	bpf_rcu_read_unlock();
+
+	/*
+	 * Mode 2: pure referenced.  Every position comes acquired, so it
+	 * stays valid for sleepable work and past the step that yielded it.
 	 */
 	bpf_iter_path_ancestors_new(&it, (struct path *)dir, 0);
 	while ((pos = bpf_iter_path_ancestors_next(&it))) {
@@ -45,5 +90,45 @@ int BPF_PROG(walk_modes, const struct path *dir, struct dentry *dentry,
 		bpf_path_put(pos);
 	}
 	bpf_iter_path_ancestors_destroy(&it);
+
+	/*
+	 * Mode 3: hybrid.  Walk lockless to the second position, then hand
+	 * that position over to a referenced iteration which resumes there.
+	 */
+	bpf_rcu_read_lock();
+	bpf_iter_path_ancestors_rcu_new(&rit, (struct path *)dir, 0);
+	while (bpf_iter_path_ancestors_rcu_next(&rit)) {
+		hybrid_count++;
+		if (hybrid_count == 2)
+			break;
+	}
+	escalate_err = bpf_path_ancestors_legitimize(&it, &rit);
+	retry_flags |= bpf_path_ancestors_rcu_pos_flags(&rit);
+	bpf_iter_path_ancestors_rcu_destroy(&rit);
+	bpf_rcu_read_unlock();
+
+	if (escalate_err)
+		test_err = 1;
+
+	/*
+	 * Out of the RCU critical section.  The resumed iteration's first
+	 * position is the escalated one, kept alive by the reference the
+	 * iteration hands out, so sleepable work can run on it.
+	 */
+	while ((pos = bpf_iter_path_ancestors_next(&it))) {
+		hybrid_count++;
+		if (hybrid_count == 3) {
+			escalated_len = bpf_path_d_path(pos, escalated_path,
+							sizeof(escalated_path));
+			bpf_dynptr_from_mem(xattr_value, sizeof(xattr_value),
+					    0, &value_ptr);
+			/* A trusted path's dentry is trusted, never NULL. */
+			xattr_ret = bpf_get_dentry_xattr(pos->dentry,
+							 "user.walk",
+							 &value_ptr);
+		}
+		bpf_path_put(pos);
+	}
+	bpf_iter_path_ancestors_destroy(&it);
 	return 0;
 }
-- 
2.55.0




More information about the Linux-security-module-archive mailing list