[RFC PATCH bpf-next 06/12] bpf: add a path ancestor iterator
Justin Suess
utilityemal77 at gmail.com
Tue Oct 6 00:20:13 UTC 2026
Let BPF programs evaluate a path's ancestry with an open-coded iterator
over the stepwise vfs_walk_ancestors() engine. The iteration holds a
reference on its current position, which is what lets a program sleep
between positions - a dput() of the last reference may - so the kfuncs
are KF_SLEEPABLE and the iterator is available to sleepable programs
only. That covers the LSM hooks a path-based policy attaches to:
file_open, file_permission, the path_* hooks, mmap_file and bprm_* are
all in the sleepable allowlist.
bpf_iter_path_ancestors_next() hands each position to the program as an
acquired reference of its own, to release with bpf_path_put(). The
walk's own reference moves on with the walk and is dropped at the next
step, so it cannot be what keeps a position alive: a program that saves
a position's dentry, or hands one to a sleepable kfunc, needs it to
outlive the step it came from. struct path is a value type with nothing
a BPF reference could be taken on, so an acquired position is a copy of
its own; the allocation is a sleepable GFP_KERNEL one, and a failure
ends the iteration with %BPF_PATH_ANCESTORS_NOMEM rather than silently
truncating the ancestry.
An acquiring KF_ITER_NEXT needs one thing from the verifier: the state
that assumes the drained, NULL-returning branch must not keep the
reference the acquire bookkeeping created for the assumed non-NULL
return. Open-coded iterators reach that branch through
process_iter_next_call() rather than through mark_ptr_or_null_regs(),
which is where a plain KF_ACQUIRE | KF_RET_NULL kfunc releases it.
The per-position VFS flags are not derivable from the position alone -
whether a disconnected root is a mountpoint a crossing landed on is
walk state - so they are read with bpf_path_ancestors_pos_flags(),
which a program needs to reproduce Landlock's evaluation of
disconnected positions.
The iterator state is deliberately larger than this walk mode needs:
its size is part of the contract with programs, which size their stack
slot from it, so growing it later would reject programs built against
the smaller one. The walk mode is a parameter of the shared engine for
the same reason - so that a mode added later is not an ABI change.
Signed-off-by: Justin Suess <utilityemal77 at gmail.com>
---
fs/bpf_fs_kfuncs.c | 147 ++++++++++++++++++++++++++++++++++++++++++
kernel/bpf/verifier.c | 7 ++
2 files changed, 154 insertions(+)
diff --git a/fs/bpf_fs_kfuncs.c b/fs/bpf_fs_kfuncs.c
index 08f0847c4970..265cb414a08a 100644
--- a/fs/bpf_fs_kfuncs.c
+++ b/fs/bpf_fs_kfuncs.c
@@ -13,7 +13,11 @@
#include <linux/kernfs.h>
#include <linux/lsm_hooks.h>
#include <linux/mm.h>
+#include <linux/namei.h>
#include <linux/net.h>
+#include <linux/slab.h>
+
+#include "internal.h"
#include <linux/xattr.h>
__bpf_kfunc_start_defs();
@@ -500,6 +504,143 @@ __bpf_kfunc struct inode *bpf_real_data_inode(struct file *file)
__bpf_kfunc_end_defs();
+enum bpf_path_ancestors_flag {
+ /* bpf_path_ancestors_pos_flags() bits */
+ BPF_PATH_ANCESTORS_DISCONNECTED = (1 << 0),
+ /* the position is a mountpoint a mount crossing landed on */
+ BPF_PATH_ANCESTORS_MOUNTPOINT = (1 << 1),
+ /* the iteration ended on a failed allocation, not at the root */
+ BPF_PATH_ANCESTORS_NOMEM = (1 << 2),
+};
+
+/*
+ * Walks over a path's ancestors.
+ *
+ * bpf_iter_path_ancestors runs with references. Its kfuncs are
+ * sleepable, so the iteration may sleep between positions.
+ *
+ * Each position is handed to the program as an acquired reference of its
+ * own, to release with bpf_path_put(). The walk's own reference moves on
+ * with the walk, so a position that outlives the step it came from - a
+ * dentry saved for later, or the path a sleepable kfunc is still working
+ * on - has to be kept alive by the program's reference rather than by the
+ * iterator's.
+ */
+struct bpf_iter_path_ancestors {
+ __u64 __opaque[5];
+} __aligned(8);
+
+struct bpf_path_ancestors_kern {
+ struct vfs_ancestor_walk aw;
+ int step; /* last vfs_walk_next() result, or -ENOMEM */
+} __aligned(8);
+
+static int bpf_path_ancestors_new(struct bpf_path_ancestors_kern *kit,
+ struct path *path, u64 flags,
+ unsigned int walk_flags)
+{
+ BUILD_BUG_ON(sizeof(struct bpf_path_ancestors_kern) >
+ sizeof(struct bpf_iter_path_ancestors));
+ BUILD_BUG_ON(__alignof__(struct bpf_path_ancestors_kern) !=
+ __alignof__(struct bpf_iter_path_ancestors));
+
+ if (flags) {
+ /* A zeroed walk makes destroying the iterator a no-op. */
+ memset(kit, 0, sizeof(*kit));
+ kit->step = 1;
+ return -EINVAL;
+ }
+ kit->step = 0;
+ vfs_walk_start(&kit->aw, path, walk_flags);
+ return 0;
+}
+
+/* The walk's own view of the next position, which the step after it ends. */
+static struct path *bpf_path_ancestors_step(struct bpf_path_ancestors_kern *kit)
+{
+ if (kit->step)
+ return NULL;
+ kit->step = vfs_walk_next(&kit->aw);
+ return kit->step ? NULL : &kit->aw.pos;
+}
+
+static u32 bpf_path_ancestors_flags(const struct bpf_path_ancestors_kern *kit)
+{
+ u32 flags = 0;
+
+ if (kit->step == -ENOMEM)
+ return BPF_PATH_ANCESTORS_NOMEM;
+ if (!kit->step) {
+ if (kit->aw.pos_flags & VFS_WALK_POS_DISCONNECTED)
+ flags |= BPF_PATH_ANCESTORS_DISCONNECTED;
+ if (kit->aw.pos_flags & VFS_WALK_POS_MOUNTPOINT)
+ flags |= BPF_PATH_ANCESTORS_MOUNTPOINT;
+ }
+ return flags;
+}
+
+__bpf_kfunc_start_defs();
+
+__bpf_kfunc int bpf_iter_path_ancestors_new(struct bpf_iter_path_ancestors *it,
+ struct path *path, u64 flags)
+{
+ return bpf_path_ancestors_new((void *)it, path, flags, 0);
+}
+
+/**
+ * bpf_iter_path_ancestors_next - acquire the walk's next position
+ * @it: the iterator
+ *
+ * Return: the next position with a reference held, to release with
+ * bpf_path_put(), or NULL once the walk has passed the real root - or on
+ * an allocation failure, which ends the iteration and is reported as
+ * %BPF_PATH_ANCESTORS_NOMEM by bpf_path_ancestors_pos_flags().
+ */
+__bpf_kfunc struct path *
+bpf_iter_path_ancestors_next(struct bpf_iter_path_ancestors *it)
+{
+ struct bpf_path_ancestors_kern *kit = (void *)it;
+ struct path *pos = bpf_path_ancestors_step(kit);
+ struct path *held;
+
+ if (!pos)
+ return NULL;
+ /*
+ * The position must outlive the walk's own view of it, so it gets a
+ * reference and a struct path of its own to live in: struct path is
+ * a value type, with nothing a BPF reference could be taken on
+ * otherwise. Sleepable, so no atomic allocation.
+ */
+ held = kmalloc_obj(*held);
+ if (!held) {
+ kit->step = -ENOMEM;
+ return NULL;
+ }
+ *held = *pos;
+ path_get(held);
+ return held;
+}
+
+__bpf_kfunc void
+bpf_iter_path_ancestors_destroy(struct bpf_iter_path_ancestors *it)
+{
+ vfs_walk_end(&((struct bpf_path_ancestors_kern *)it)->aw);
+}
+
+__bpf_kfunc u32
+bpf_path_ancestors_pos_flags(struct bpf_iter_path_ancestors *it__iter)
+{
+ return bpf_path_ancestors_flags((void *)it__iter);
+}
+
+__bpf_kfunc void bpf_path_put(struct path *path)
+{
+ path_put(path);
+ kfree(path);
+}
+
+__bpf_kfunc_end_defs();
+
BTF_KFUNCS_START(bpf_fs_kfunc_set_ids)
BTF_ID_FLAGS(func, bpf_get_task_exe_file, KF_ACQUIRE | KF_RET_NULL)
BTF_ID_FLAGS(func, bpf_put_file, KF_RELEASE)
@@ -513,6 +654,12 @@ BTF_ID_FLAGS(func, bpf_inode_init_xattr)
#ifdef CONFIG_NET
BTF_ID_FLAGS(func, bpf_sock_read_xattr, KF_RCU)
#endif
+BTF_ID_FLAGS(func, bpf_iter_path_ancestors_new, KF_ITER_NEW | KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_iter_path_ancestors_next,
+ KF_ITER_NEXT | KF_ACQUIRE | KF_RET_NULL | KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_iter_path_ancestors_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_path_ancestors_pos_flags)
+BTF_ID_FLAGS(func, bpf_path_put, KF_RELEASE | KF_SLEEPABLE)
BTF_KFUNCS_END(bpf_fs_kfunc_set_ids)
/* Side-effecting kfuncs that stay exclusive to LSM programs. */
diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c
index 24ec4b037de7..066c4b838b85 100644
--- a/kernel/bpf/verifier.c
+++ b/kernel/bpf/verifier.c
@@ -8648,6 +8648,13 @@ static int process_iter_next_call(struct bpf_verifier_env *env, int insn_idx,
/* switch to DRAINED state, but keep the depth unchanged */
/* mark current iter state as drained and assume returned NULL */
cur_iter->iter.state = BPF_ITER_STATE_DRAINED;
+ /*
+ * An acquiring iter_next() hands out nothing once drained: the
+ * acquired reference exists only in the forked active state, not on
+ * this NULL-returning branch.
+ */
+ if (meta->kfunc_flags & KF_ACQUIRE)
+ WARN_ON_ONCE(release_reference_nomark(env, cur_fr->regs[BPF_REG_0].id));
__mark_reg_const_zero(env, &cur_fr->regs[BPF_REG_0]);
return 0;
--
2.55.0
More information about the Linux-security-module-archive
mailing list