[RFC PATCH bpf-next 12/12] selftests/bpf: exercise the lockless path ancestor iterator
Justin Suess
utilityemal77 at gmail.com
Tue Oct 6 00:20:19 UTC 2026
Walk the same ancestry four ways and require the position counts to
agree: lockless from a non-sleepable program, where the RCU critical
section is implicit; lockless under an explicit bpf_rcu_read_lock();
referenced; and the hybrid that walks lockless to the second position,
hands it over to a referenced iteration, and resumes there - the
escalated position therefore being walked twice, once per mode.
The sleepable work the escalation exists for (d_path, an xattr read
through the position's dentry) runs on the resumed iteration's first
position, after bpf_rcu_read_unlock(), which is the only place a
sleepable kfunc can run at all.
Signed-off-by: Justin Suess <utilityemal77 at gmail.com>
---
.../selftests/bpf/prog_tests/path_ancestors.c | 33 ++++++-
.../selftests/bpf/progs/path_ancestors.c | 89 ++++++++++++++++++-
2 files changed, 118 insertions(+), 4 deletions(-)
diff --git a/tools/testing/selftests/bpf/prog_tests/path_ancestors.c b/tools/testing/selftests/bpf/prog_tests/path_ancestors.c
index 2de79673a13b..ce1ded844c3a 100644
--- a/tools/testing/selftests/bpf/prog_tests/path_ancestors.c
+++ b/tools/testing/selftests/bpf/prog_tests/path_ancestors.c
@@ -2,6 +2,7 @@
/* Copyright (c) 2026 Justin Suess */
#include <sys/stat.h>
+#include <sys/xattr.h>
#include <stdlib.h>
#include <unistd.h>
#include <test_progs.h>
@@ -12,6 +13,8 @@ void test_path_ancestors(void)
char base[] = "/tmp/path_ancestors_XXXXXX";
struct path_ancestors *skel = NULL;
char suba[280], subb[280];
+ bool xattr_works;
+ int err;
if (!ASSERT_OK_PTR(mkdtemp(base), "mkdtemp"))
return;
@@ -20,6 +23,12 @@ void test_path_ancestors(void)
if (!ASSERT_OK(mkdir(suba, 0755), "mkdir_a"))
goto out_rm;
+ /* Read back by the program at the escalated position (== base). */
+ err = setxattr(base, "user.walk", "hello", 6, 0);
+ xattr_works = !err;
+ if (err && errno != EOPNOTSUPP && !ASSERT_OK(err, "setxattr"))
+ goto out_rm;
+
skel = path_ancestors__open_and_load();
if (!ASSERT_OK_PTR(skel, "open_and_load"))
goto out_rm;
@@ -31,14 +40,34 @@ void test_path_ancestors(void)
if (!ASSERT_OK(mkdir(subb, 0755), "mkdir_b"))
goto out;
- /* suba, base, /tmp, / at least. */
- ASSERT_GE(skel->bss->ref_count, 3, "ref_count");
+ ASSERT_EQ(skel->bss->test_err, 0, "test_err");
+ ASSERT_EQ(skel->bss->escalate_err, 0, "escalate_err");
+ /* suba, base, /tmp, / at least; equality across modes is the point. */
+ ASSERT_GE(skel->bss->rcu_count, 3, "rcu_count");
+ ASSERT_EQ(skel->bss->ref_count, skel->bss->rcu_count, "ref_vs_rcu");
+ ASSERT_EQ(skel->bss->rcu_ns_count, skel->bss->rcu_count,
+ "nonsleepable_vs_rcu");
+ /*
+ * The escalated position is walked twice: once lockless, then again
+ * as the resumed referenced iteration's first position.
+ */
+ ASSERT_EQ(skel->bss->hybrid_count, skel->bss->rcu_count + 1,
+ "hybrid_vs_rcu");
+ ASSERT_EQ(skel->bss->retry_flags, 0, "no_retry");
ASSERT_EQ(skel->bss->ref_flags, 0, "ref_flags");
/* The acquired second position, used after its step was taken. */
ASSERT_STREQ(skel->bss->second_path, base, "second_path");
ASSERT_EQ(skel->bss->second_len, strlen(base) + 1, "second_len");
+ /* The escalated position is the walk's second one: base. */
+ ASSERT_STREQ(skel->bss->escalated_path, base, "escalated_path");
+ ASSERT_EQ(skel->bss->escalated_len, strlen(base) + 1, "escalated_len");
+ if (xattr_works) {
+ ASSERT_EQ(skel->bss->xattr_ret, 6, "xattr_len");
+ ASSERT_STREQ(skel->bss->xattr_value, "hello", "xattr_value");
+ }
+
out:
path_ancestors__destroy(skel);
out_rm:
diff --git a/tools/testing/selftests/bpf/progs/path_ancestors.c b/tools/testing/selftests/bpf/progs/path_ancestors.c
index af6b777e8bec..50ce0ce163dd 100644
--- a/tools/testing/selftests/bpf/progs/path_ancestors.c
+++ b/tools/testing/selftests/bpf/progs/path_ancestors.c
@@ -11,29 +11,74 @@ char _license[] SEC("license") = "GPL";
__u32 monitored_pid;
+int rcu_count; /* positions seen by the pure lockless walk */
+int rcu_ns_count; /* ditto, from the non-sleepable program */
int ref_count; /* positions seen by the pure referenced walk */
+int hybrid_count; /* positions seen by the lockless+escalate walk */
+int retry_flags; /* BPF_PATH_ANCESTORS_RETRY observations */
int ref_flags; /* pos flags seen by the referenced walk */
int second_len; /* d_path length of the walk's second position */
+int xattr_ret; /* xattr read at the escalated position */
+int escalated_len; /* d_path length of the escalated position */
+int escalate_err; /* bpf_path_ancestors_legitimize() result */
+int test_err;
char second_path[256];
+char escalated_path[256];
+char xattr_value[16];
static bool monitored(void)
{
return (bpf_get_current_pid_tgid() >> 32) == monitored_pid;
}
+/*
+ * Lockless walk from a non-sleepable program: the RCU critical section is
+ * implicit, no bpf_rcu_read_lock() needed.
+ */
+SEC("lsm/path_mkdir")
+int BPF_PROG(rcu_nonsleepable, const struct path *dir, struct dentry *dentry,
+ umode_t mode)
+{
+ struct bpf_iter_path_ancestors_rcu rit;
+
+ if (!monitored())
+ return 0;
+
+ bpf_iter_path_ancestors_rcu_new(&rit, (struct path *)dir, 0);
+ while (bpf_iter_path_ancestors_rcu_next(&rit))
+ rcu_ns_count++;
+ retry_flags |= bpf_path_ancestors_rcu_pos_flags(&rit);
+ bpf_iter_path_ancestors_rcu_destroy(&rit);
+ return 0;
+}
+
SEC("lsm.s/path_mkdir")
int BPF_PROG(walk_modes, const struct path *dir, struct dentry *dentry,
umode_t mode)
{
+ struct bpf_iter_path_ancestors_rcu rit;
struct bpf_iter_path_ancestors it;
+ struct bpf_dynptr value_ptr;
struct path *pos;
if (!monitored())
return 0;
/*
- * Referenced walk: every position comes acquired, so it stays valid
- * for sleepable work and past the step that yielded it.
+ * Mode 1: pure lockless, under an explicit RCU critical section.
+ * Positions are borrowed, so nothing is released here.
+ */
+ bpf_rcu_read_lock();
+ bpf_iter_path_ancestors_rcu_new(&rit, (struct path *)dir, 0);
+ while (bpf_iter_path_ancestors_rcu_next(&rit))
+ rcu_count++;
+ retry_flags |= bpf_path_ancestors_rcu_pos_flags(&rit);
+ bpf_iter_path_ancestors_rcu_destroy(&rit);
+ bpf_rcu_read_unlock();
+
+ /*
+ * Mode 2: pure referenced. Every position comes acquired, so it
+ * stays valid for sleepable work and past the step that yielded it.
*/
bpf_iter_path_ancestors_new(&it, (struct path *)dir, 0);
while ((pos = bpf_iter_path_ancestors_next(&it))) {
@@ -45,5 +90,45 @@ int BPF_PROG(walk_modes, const struct path *dir, struct dentry *dentry,
bpf_path_put(pos);
}
bpf_iter_path_ancestors_destroy(&it);
+
+ /*
+ * Mode 3: hybrid. Walk lockless to the second position, then hand
+ * that position over to a referenced iteration which resumes there.
+ */
+ bpf_rcu_read_lock();
+ bpf_iter_path_ancestors_rcu_new(&rit, (struct path *)dir, 0);
+ while (bpf_iter_path_ancestors_rcu_next(&rit)) {
+ hybrid_count++;
+ if (hybrid_count == 2)
+ break;
+ }
+ escalate_err = bpf_path_ancestors_legitimize(&it, &rit);
+ retry_flags |= bpf_path_ancestors_rcu_pos_flags(&rit);
+ bpf_iter_path_ancestors_rcu_destroy(&rit);
+ bpf_rcu_read_unlock();
+
+ if (escalate_err)
+ test_err = 1;
+
+ /*
+ * Out of the RCU critical section. The resumed iteration's first
+ * position is the escalated one, kept alive by the reference the
+ * iteration hands out, so sleepable work can run on it.
+ */
+ while ((pos = bpf_iter_path_ancestors_next(&it))) {
+ hybrid_count++;
+ if (hybrid_count == 3) {
+ escalated_len = bpf_path_d_path(pos, escalated_path,
+ sizeof(escalated_path));
+ bpf_dynptr_from_mem(xattr_value, sizeof(xattr_value),
+ 0, &value_ptr);
+ /* A trusted path's dentry is trusted, never NULL. */
+ xattr_ret = bpf_get_dentry_xattr(pos->dentry,
+ "user.walk",
+ &value_ptr);
+ }
+ bpf_path_put(pos);
+ }
+ bpf_iter_path_ancestors_destroy(&it);
return 0;
}
--
2.55.0
More information about the Linux-security-module-archive
mailing list