summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--Documentation/filesystems/failfs.rst73
-rw-r--r--Documentation/filesystems/index.rst1
-rw-r--r--arch/alpha/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/arm/tools/syscall.tbl1
-rw-r--r--arch/arm64/tools/syscall_32.tbl1
-rw-r--r--arch/m68k/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/microblaze/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/mips/kernel/syscalls/syscall_n32.tbl1
-rw-r--r--arch/mips/kernel/syscalls/syscall_n64.tbl1
-rw-r--r--arch/mips/kernel/syscalls/syscall_o32.tbl1
-rw-r--r--arch/parisc/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/powerpc/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/s390/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/sh/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/sparc/kernel/syscalls/syscall.tbl1
-rw-r--r--arch/x86/entry/syscalls/syscall_32.tbl1
-rw-r--r--arch/x86/entry/syscalls/syscall_64.tbl1
-rw-r--r--arch/xtensa/kernel/syscalls/syscall.tbl1
-rw-r--r--fs/Makefile2
-rw-r--r--fs/d_path.c3
-rw-r--r--fs/failfs.c166
-rw-r--r--fs/internal.h4
-rw-r--r--fs/namespace.c1
-rw-r--r--fs/open.c51
-rw-r--r--include/linux/syscalls.h1
-rw-r--r--include/uapi/asm-generic/unistd.h6
-rw-r--r--include/uapi/linux/fcntl.h1
-rw-r--r--include/uapi/linux/magic.h1
-rw-r--r--scripts/syscall.tbl1
-rw-r--r--tools/include/uapi/asm-generic/unistd.h6
-rw-r--r--tools/perf/arch/arm/entry/syscalls/syscall.tbl1
-rw-r--r--tools/perf/arch/mips/entry/syscalls/syscall_n64.tbl1
-rw-r--r--tools/perf/arch/powerpc/entry/syscalls/syscall.tbl1
-rw-r--r--tools/perf/arch/s390/entry/syscalls/syscall.tbl1
-rw-r--r--tools/perf/arch/sh/entry/syscalls/syscall.tbl1
-rw-r--r--tools/perf/arch/sparc/entry/syscalls/syscall.tbl1
-rw-r--r--tools/perf/arch/x86/entry/syscalls/syscall_32.tbl1
-rw-r--r--tools/perf/arch/x86/entry/syscalls/syscall_64.tbl1
-rw-r--r--tools/perf/arch/xtensa/entry/syscalls/syscall.tbl1
-rw-r--r--tools/scripts/syscall.tbl1
-rw-r--r--tools/testing/selftests/Makefile1
-rw-r--r--tools/testing/selftests/filesystems/failfs/.gitignore2
-rw-r--r--tools/testing/selftests/filesystems/failfs/Makefile5
-rw-r--r--tools/testing/selftests/filesystems/failfs/failfs_test.c585
44 files changed, 931 insertions, 5 deletions
diff --git a/Documentation/filesystems/failfs.rst b/Documentation/filesystems/failfs.rst
new file mode 100644
index 000000000000..21ff2db7941d
--- /dev/null
+++ b/Documentation/filesystems/failfs.rst
@@ -0,0 +1,73 @@
+.. SPDX-License-Identifier: GPL-2.0
+
+======
+failfs
+======
+
+failfs is a kernel-internal filesystem that fails every operation
+reaching it with ``EOPNOTSUPP``. It is the counterpart to nullfs. Where
+nullfs is permanently empty, failfs means "nothing is supported here".
+It cannot be mounted from userspace, nothing can be mounted on top of
+it. It cannot be cloned.
+
+The only way into it is the ``FD_FAILFS_ROOT`` file descriptor sentinel which
+is understood by ``fchdir(2)`` and ``fchroot(2)``.
+
+Semantics
+=========
+
+Every path walk of a component through failfs fails with
+``EOPNOTSUPP`` before that component is parsed, including ``.``.
+
+No path lookup can open the root, not even with ``O_PATH``.
+
+A process with its working directory in failfs fails every
+``AT_FDCWD``-relative lookup. As with any working directory that is
+unreachable from the process root, the ``getcwd(2)`` system call returns
+a path prefixed with ``(unreachable)``.
+
+A process with its root directory in failfs fails every absolute path
+lookup including absolute symlinks and the interpreter of dynamically
+linked binaries. In other words, this fails exec.
+
+Lookups anchored at explicit directory file descriptors keep working. It
+is the ``fs_struct`` equivalent of ``RESOLVE_BENEATH``. The process must
+anchor every lookup at a file descriptor it explicitly holds.
+
+Entering
+========
+
+``fchroot(FD_FAILFS_ROOT, 0)`` requires ``CAP_SYS_CHROOT`` in the
+caller's user namespace, mirroring ``chroot(2)``. Unprivileged callers
+may enter if all of the following hold:
+
+* ``no_new_privs`` is set: setuid binaries on regular mounts remain
+ reachable via inherited directory file descriptors and executing them
+ with an unusable root directory is the classic confused deputy.
+
+* The caller is not already chrooted: the root directory is what
+ confines ``..`` resolution and the failfs root can never be reached by
+ walking up a real mount tree, so moving the root of a chrooted task to
+ failfs would allow it to escape its chroot via ``openat(fd, "..")``.
+
+* The caller does not share its ``fs_struct``: ``no_new_privs`` is
+ checked on the calling thread, but the root lives in the ``fs_struct``.
+ A ``CLONE_FS`` sibling without ``no_new_privs`` could otherwise execute
+ a setuid binary with the failfs root, so entry requires ``fs->users ==
+ 1``, the same restriction ``setns(2)`` applies for the mount and user
+ namespaces.
+
+Leaving
+=======
+
+Backing out is currently hard, but this is a property of the current
+implementation, not a guaranteed interface, and may be loosened later.
+For now a process that entered failfs counts as chrooted, so it cannot
+create user namespaces to regain ``CAP_SYS_CHROOT``, and ``chroot(2)``
+or ``fchroot(2)`` back out require ``CAP_SYS_CHROOT``. The remaining way
+out today is ``setns(2)`` with a mount namespace file descriptor, which
+requires ``CAP_SYS_ADMIN`` over the target mount namespace as well as
+``CAP_SYS_CHROOT`` and ``CAP_SYS_ADMIN`` in the caller's user namespace
+and resets both root and working directory. A process that holds no such
+file descriptor and restricts ``*chdir()``/``*chroot()``/``setns()`` via
+seccomp cannot currently get back out.
diff --git a/Documentation/filesystems/index.rst b/Documentation/filesystems/index.rst
index 1f71cf159547..734a45e51667 100644
--- a/Documentation/filesystems/index.rst
+++ b/Documentation/filesystems/index.rst
@@ -91,6 +91,7 @@ Documentation for filesystem implementations.
ext3
ext4/index
f2fs
+ failfs
gfs2/index
hfs
hfsplus
diff --git a/arch/alpha/kernel/syscalls/syscall.tbl b/arch/alpha/kernel/syscalls/syscall.tbl
index f31b7afffc34..52e3538cc7df 100644
--- a/arch/alpha/kernel/syscalls/syscall.tbl
+++ b/arch/alpha/kernel/syscalls/syscall.tbl
@@ -511,3 +511,4 @@
579 common file_setattr sys_file_setattr
580 common listns sys_listns
581 common rseq_slice_yield sys_rseq_slice_yield
+582 common fchroot sys_fchroot
diff --git a/arch/arm/tools/syscall.tbl b/arch/arm/tools/syscall.tbl
index 94351e22bfcf..55717ed32c27 100644
--- a/arch/arm/tools/syscall.tbl
+++ b/arch/arm/tools/syscall.tbl
@@ -486,3 +486,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/arm64/tools/syscall_32.tbl b/arch/arm64/tools/syscall_32.tbl
index 62d93d88e0fe..df2d1d82fb3c 100644
--- a/arch/arm64/tools/syscall_32.tbl
+++ b/arch/arm64/tools/syscall_32.tbl
@@ -483,3 +483,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/m68k/kernel/syscalls/syscall.tbl b/arch/m68k/kernel/syscalls/syscall.tbl
index 248934257101..ba7a4d8903d0 100644
--- a/arch/m68k/kernel/syscalls/syscall.tbl
+++ b/arch/m68k/kernel/syscalls/syscall.tbl
@@ -471,3 +471,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/microblaze/kernel/syscalls/syscall.tbl b/arch/microblaze/kernel/syscalls/syscall.tbl
index 223d26303627..c55a5c96f49b 100644
--- a/arch/microblaze/kernel/syscalls/syscall.tbl
+++ b/arch/microblaze/kernel/syscalls/syscall.tbl
@@ -477,3 +477,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/mips/kernel/syscalls/syscall_n32.tbl b/arch/mips/kernel/syscalls/syscall_n32.tbl
index 7430714e2b8f..9ae88e4eac61 100644
--- a/arch/mips/kernel/syscalls/syscall_n32.tbl
+++ b/arch/mips/kernel/syscalls/syscall_n32.tbl
@@ -410,3 +410,4 @@
469 n32 file_setattr sys_file_setattr
470 n32 listns sys_listns
471 n32 rseq_slice_yield sys_rseq_slice_yield
+472 n32 fchroot sys_fchroot
diff --git a/arch/mips/kernel/syscalls/syscall_n64.tbl b/arch/mips/kernel/syscalls/syscall_n64.tbl
index 630aab9e5425..83dc93a0712f 100644
--- a/arch/mips/kernel/syscalls/syscall_n64.tbl
+++ b/arch/mips/kernel/syscalls/syscall_n64.tbl
@@ -386,3 +386,4 @@
469 n64 file_setattr sys_file_setattr
470 n64 listns sys_listns
471 n64 rseq_slice_yield sys_rseq_slice_yield
+472 n64 fchroot sys_fchroot
diff --git a/arch/mips/kernel/syscalls/syscall_o32.tbl b/arch/mips/kernel/syscalls/syscall_o32.tbl
index 128653112284..9c62429c9b7b 100644
--- a/arch/mips/kernel/syscalls/syscall_o32.tbl
+++ b/arch/mips/kernel/syscalls/syscall_o32.tbl
@@ -459,3 +459,4 @@
469 o32 file_setattr sys_file_setattr
470 o32 listns sys_listns
471 o32 rseq_slice_yield sys_rseq_slice_yield
+472 o32 fchroot sys_fchroot
diff --git a/arch/parisc/kernel/syscalls/syscall.tbl b/arch/parisc/kernel/syscalls/syscall.tbl
index c6331dad9461..88adc4016cce 100644
--- a/arch/parisc/kernel/syscalls/syscall.tbl
+++ b/arch/parisc/kernel/syscalls/syscall.tbl
@@ -470,3 +470,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/powerpc/kernel/syscalls/syscall.tbl b/arch/powerpc/kernel/syscalls/syscall.tbl
index 4fcc7c58a105..cfbb70039ff0 100644
--- a/arch/powerpc/kernel/syscalls/syscall.tbl
+++ b/arch/powerpc/kernel/syscalls/syscall.tbl
@@ -562,3 +562,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 nospu rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/s390/kernel/syscalls/syscall.tbl b/arch/s390/kernel/syscalls/syscall.tbl
index 09a7ef04d979..1b45e68a217b 100644
--- a/arch/s390/kernel/syscalls/syscall.tbl
+++ b/arch/s390/kernel/syscalls/syscall.tbl
@@ -398,3 +398,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/sh/kernel/syscalls/syscall.tbl b/arch/sh/kernel/syscalls/syscall.tbl
index 70b315cbe710..ace068dff0de 100644
--- a/arch/sh/kernel/syscalls/syscall.tbl
+++ b/arch/sh/kernel/syscalls/syscall.tbl
@@ -475,3 +475,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/sparc/kernel/syscalls/syscall.tbl b/arch/sparc/kernel/syscalls/syscall.tbl
index 7e71bf7fcd14..5b9fe0e8140f 100644
--- a/arch/sparc/kernel/syscalls/syscall.tbl
+++ b/arch/sparc/kernel/syscalls/syscall.tbl
@@ -517,3 +517,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/arch/x86/entry/syscalls/syscall_32.tbl b/arch/x86/entry/syscalls/syscall_32.tbl
index f832ebd2d79b..2c172ef48dfd 100644
--- a/arch/x86/entry/syscalls/syscall_32.tbl
+++ b/arch/x86/entry/syscalls/syscall_32.tbl
@@ -477,3 +477,4 @@
469 i386 file_setattr sys_file_setattr
470 i386 listns sys_listns
471 i386 rseq_slice_yield sys_rseq_slice_yield
+472 i386 fchroot sys_fchroot
diff --git a/arch/x86/entry/syscalls/syscall_64.tbl b/arch/x86/entry/syscalls/syscall_64.tbl
index 524155d655da..d5b6045b0090 100644
--- a/arch/x86/entry/syscalls/syscall_64.tbl
+++ b/arch/x86/entry/syscalls/syscall_64.tbl
@@ -396,6 +396,7 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
#
# Due to a historical design error, certain syscalls are numbered differently
diff --git a/arch/xtensa/kernel/syscalls/syscall.tbl b/arch/xtensa/kernel/syscalls/syscall.tbl
index a9bca4e484de..d354bb231796 100644
--- a/arch/xtensa/kernel/syscalls/syscall.tbl
+++ b/arch/xtensa/kernel/syscalls/syscall.tbl
@@ -442,3 +442,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/fs/Makefile b/fs/Makefile
index d4704321c3d8..055dfc23d82b 100644
--- a/fs/Makefile
+++ b/fs/Makefile
@@ -16,7 +16,7 @@ obj-y := open.o read_write.o file_table.o super.o \
stack.o fs_struct.o statfs.o fs_pin.o nsfs.o \
fs_dirent.o fs_context.o fs_parser.o fsopen.o init.o \
kernel_read_file.o mnt_idmapping.o remap_range.o pidfs.o \
- file_attr.o fserror.o nullfs.o
+ file_attr.o fserror.o nullfs.o failfs.o
obj-$(CONFIG_BUFFER_HEAD) += buffer.o mpage.o
obj-$(CONFIG_PROC_FS) += proc_namespace.o
diff --git a/fs/d_path.c b/fs/d_path.c
index a48957c0971e..c25309006d5d 100644
--- a/fs/d_path.c
+++ b/fs/d_path.c
@@ -279,7 +279,8 @@ char *d_path(const struct path *path, char *buf, int buflen)
* and instead have d_path return the mounted path.
*/
if (path->dentry->d_op && path->dentry->d_op->d_dname &&
- (!IS_ROOT(path->dentry) || path->dentry != path->mnt->mnt_root))
+ (!IS_ROOT(path->dentry) || path->dentry != path->mnt->mnt_root ||
+ failfs_mnt(path->mnt)))
return path->dentry->d_op->d_dname(path->dentry, buf, buflen);
rcu_read_lock();
diff --git a/fs/failfs.c b/fs/failfs.c
new file mode 100644
index 000000000000..66a36da3d236
--- /dev/null
+++ b/fs/failfs.c
@@ -0,0 +1,166 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/* Copyright (c) 2026 Christian Brauner <brauner@kernel.org> */
+#include <linux/fs.h>
+#include <linux/fs/super_types.h>
+#include <linux/fs_context.h>
+#include <linux/fs_struct.h>
+#include <linux/magic.h>
+#include <linux/mount.h>
+
+#include "internal.h"
+
+static struct path failfs_root_path = {};
+
+void failfs_get_root(struct path *path)
+{
+ *path = failfs_root_path;
+ path_get(path);
+}
+
+bool failfs_mnt(const struct vfsmount *mnt)
+{
+ return mnt->mnt_sb == failfs_root_path.mnt->mnt_sb;
+}
+
+static int failfs_permission(struct mnt_idmap *idmap, struct inode *inode,
+ int mask)
+{
+ return -EOPNOTSUPP;
+}
+
+static struct dentry *failfs_lookup(struct inode *dir, struct dentry *dentry,
+ unsigned int flags)
+{
+ /* Unreachable: ->permission() already failed the walk. */
+ return ERR_PTR(-EOPNOTSUPP);
+}
+
+static int failfs_getattr(struct mnt_idmap *idmap, const struct path *path,
+ struct kstat *stat, u32 request_mask,
+ unsigned int query_flags)
+{
+ return -EOPNOTSUPP;
+}
+
+static const struct inode_operations failfs_dir_inode_operations = {
+ .permission = failfs_permission,
+ .lookup = failfs_lookup,
+ .getattr = failfs_getattr,
+};
+
+static const struct file_operations failfs_dir_operations = {};
+
+static int failfs_d_weak_revalidate(struct dentry *dentry, unsigned int flags)
+{
+ /*
+ * The root is only ever reached as a path-walk terminal by jumping
+ * to it: as "/" when it is the caller's root, or through a
+ * /proc/<pid>/{root,cwd} magic link. ->permission() already fails
+ * every walk of a component, but a jump lands on the root without
+ * one. Refuse here too so the root cannot be pinned by an O_PATH
+ * open or encoded into a file handle.
+ */
+ return -EOPNOTSUPP;
+}
+
+static char *failfs_dname(struct dentry *dentry, char *buffer, int buflen)
+{
+ return dynamic_dname(buffer, buflen, "failfs:/");
+}
+
+static const struct dentry_operations failfs_dentry_operations = {
+ .d_dname = failfs_dname,
+ .d_weak_revalidate = failfs_d_weak_revalidate,
+};
+
+static int failfs_statfs(struct dentry *dentry, struct kstatfs *buf)
+{
+ return -EOPNOTSUPP;
+}
+
+static const struct super_operations failfs_super_operations = {
+ .statfs = failfs_statfs,
+};
+
+static int failfs_fill_super(struct super_block *s, struct fs_context *fc)
+{
+ struct inode *inode;
+
+ s->s_maxbytes = MAX_LFS_FILESIZE;
+ s->s_blocksize = PAGE_SIZE;
+ s->s_blocksize_bits = PAGE_SHIFT;
+ s->s_magic = FAIL_FS_MAGIC;
+ s->s_op = &failfs_super_operations;
+ s->s_export_op = NULL;
+ s->s_xattr = NULL;
+ s->s_time_gran = 1;
+ s->s_d_flags = 0;
+
+ inode = new_inode(s);
+ if (!inode)
+ return -ENOMEM;
+
+ /* failfs supports no operations... */
+ inode->i_mode = S_IFDIR;
+ set_nlink(inode, 2);
+ inode->i_op = &failfs_dir_inode_operations;
+ inode->i_fop = &failfs_dir_operations;
+ simple_inode_init_ts(inode);
+ inode->i_ino = 1;
+ /* ... and is immutable. */
+ inode->i_flags |= S_IMMUTABLE;
+
+ set_default_d_op(s, &failfs_dentry_operations);
+ s->s_root = d_make_root(inode);
+ if (!s->s_root)
+ return -ENOMEM;
+
+ return 0;
+}
+
+static int failfs_get_tree(struct fs_context *fc)
+{
+ return get_tree_single(fc, failfs_fill_super);
+}
+
+static const struct fs_context_operations failfs_context_ops = {
+ .get_tree = failfs_get_tree,
+};
+
+static int failfs_init_fs_context(struct fs_context *fc)
+{
+ fc->ops = &failfs_context_ops;
+ fc->global = true;
+ fc->sb_flags |= SB_NOUSER;
+ fc->s_iflags |= SB_I_NOEXEC | SB_I_NODEV;
+ return 0;
+}
+
+int failfs_current_chdir(void)
+{
+ struct path path;
+
+ failfs_get_root(&path);
+ set_fs_pwd(current->fs, &path);
+ path_put(&path);
+ return 0;
+}
+
+static struct file_system_type failfs_fs_type = {
+ .name = "failfs",
+ .init_fs_context = failfs_init_fs_context,
+ .kill_sb = kill_anon_super,
+};
+
+void __init failfs_init(void)
+{
+ struct vfsmount *mnt;
+
+ /* A single instance that is member of no mount namespace. */
+ mnt = kern_mount(&failfs_fs_type);
+ if (IS_ERR(mnt))
+ panic("VFS: Failed to create failfs");
+
+ failfs_root_path.mnt = mnt;
+ failfs_root_path.dentry = mnt->mnt_root;
+}
diff --git a/fs/internal.h b/fs/internal.h
index 355d93f92208..67aa0444351b 100644
--- a/fs/internal.h
+++ b/fs/internal.h
@@ -362,3 +362,7 @@ int anon_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
struct iattr *attr);
void pidfs_get_root(struct path *path);
void nsfs_get_root(struct path *path);
+void failfs_get_root(struct path *path);
+void __init failfs_init(void);
+bool failfs_mnt(const struct vfsmount *mnt);
+int failfs_current_chdir(void);
diff --git a/fs/namespace.c b/fs/namespace.c
index 3d5cd5bf3b05..87c365f2f82b 100644
--- a/fs/namespace.c
+++ b/fs/namespace.c
@@ -6274,6 +6274,7 @@ void __init mnt_init(void)
shmem_init();
init_rootfs();
init_mount_tree();
+ failfs_init();
}
void put_mnt_ns(struct mnt_namespace *ns)
diff --git a/fs/open.c b/fs/open.c
index 408925d7bd0b..6b1c14e684a9 100644
--- a/fs/open.c
+++ b/fs/open.c
@@ -570,9 +570,12 @@ retry:
SYSCALL_DEFINE1(fchdir, unsigned int, fd)
{
- CLASS(fd_raw, f)(fd);
int error;
+ if ((int)fd == FD_FAILFS_ROOT)
+ return failfs_current_chdir();
+
+ CLASS(fd_raw, f)(fd);
if (fd_empty(f))
return -EBADF;
@@ -615,6 +618,52 @@ dput_and_out:
return error;
}
+SYSCALL_DEFINE2(fchroot, int, fd, unsigned int, flags)
+{
+ struct path path;
+ int error;
+
+ if (flags)
+ return -EINVAL;
+
+ if (fd == FD_FAILFS_ROOT) {
+ if (!ns_capable(current_user_ns(), CAP_SYS_CHROOT)) {
+ if (!task_no_new_privs(current))
+ return -EPERM;
+ /* A shared fs_struct lets a sibling exec setuid past the check above. */
+ if (current->fs->users != 1)
+ return -EINVAL;
+ /* Moving the root to failfs lifts the old root's ".." barrier. */
+ if (current_chrooted())
+ return -EPERM;
+ }
+ failfs_get_root(&path);
+ } else {
+ CLASS(fd_raw, f)(fd);
+ if (fd_empty(f))
+ return -EBADF;
+
+ if (!d_can_lookup(fd_file(f)->f_path.dentry))
+ return -ENOTDIR;
+
+ error = file_permission(fd_file(f), MAY_EXEC | MAY_CHDIR);
+ if (error)
+ return error;
+
+ if (!ns_capable(current_user_ns(), CAP_SYS_CHROOT))
+ return -EPERM;
+
+ path = fd_file(f)->f_path;
+ path_get(&path);
+ }
+
+ error = security_path_chroot(&path);
+ if (!error)
+ set_fs_root(current->fs, &path);
+ path_put(&path);
+ return error;
+}
+
int chmod_common(const struct path *path, umode_t mode)
{
struct inode *inode = path->dentry->d_inode;
diff --git a/include/linux/syscalls.h b/include/linux/syscalls.h
index 874d9067a43b..8413b624ad47 100644
--- a/include/linux/syscalls.h
+++ b/include/linux/syscalls.h
@@ -457,6 +457,7 @@ asmlinkage long sys_faccessat2(int dfd, const char __user *filename, int mode,
asmlinkage long sys_chdir(const char __user *filename);
asmlinkage long sys_fchdir(unsigned int fd);
asmlinkage long sys_chroot(const char __user *filename);
+asmlinkage long sys_fchroot(int fd, unsigned int flags);
asmlinkage long sys_fchmod(unsigned int fd, umode_t mode);
asmlinkage long sys_fchmodat(int dfd, const char __user *filename,
umode_t mode);
diff --git a/include/uapi/asm-generic/unistd.h b/include/uapi/asm-generic/unistd.h
index a627acc8fb5f..5b7e77a7c736 100644
--- a/include/uapi/asm-generic/unistd.h
+++ b/include/uapi/asm-generic/unistd.h
@@ -863,8 +863,12 @@ __SYSCALL(__NR_listns, sys_listns)
#define __NR_rseq_slice_yield 471
__SYSCALL(__NR_rseq_slice_yield, sys_rseq_slice_yield)
+/* fs/open.c */
+#define __NR_fchroot 472
+__SYSCALL(__NR_fchroot, sys_fchroot)
+
#undef __NR_syscalls
-#define __NR_syscalls 472
+#define __NR_syscalls 473
/*
* 32 bit systems traditionally used different
diff --git a/include/uapi/linux/fcntl.h b/include/uapi/linux/fcntl.h
index aadfbf6e0cb3..e43e3de3e9ee 100644
--- a/include/uapi/linux/fcntl.h
+++ b/include/uapi/linux/fcntl.h
@@ -124,6 +124,7 @@ struct delegation {
#define FD_PIDFS_ROOT -10002 /* Root of the pidfs filesystem */
#define FD_NSFS_ROOT -10003 /* Root of the nsfs filesystem */
+#define FD_FAILFS_ROOT -10004 /* Root of the failfs filesystem */
#define FD_INVALID -10009 /* Invalid file descriptor: -10000 - EBADF = -10009 */
/* Generic flags for the *at(2) family of syscalls. */
diff --git a/include/uapi/linux/magic.h b/include/uapi/linux/magic.h
index 4f2da935a76c..fd5f0e95648e 100644
--- a/include/uapi/linux/magic.h
+++ b/include/uapi/linux/magic.h
@@ -105,5 +105,6 @@
#define PID_FS_MAGIC 0x50494446 /* "PIDF" */
#define GUEST_MEMFD_MAGIC 0x474d454d /* "GMEM" */
#define NULL_FS_MAGIC 0x4E554C4C /* "NULL" */
+#define FAIL_FS_MAGIC 0x4641494C /* "FAIL" */
#endif /* __LINUX_MAGIC_H__ */
diff --git a/scripts/syscall.tbl b/scripts/syscall.tbl
index 7a42b32b6577..0ab531605120 100644
--- a/scripts/syscall.tbl
+++ b/scripts/syscall.tbl
@@ -412,3 +412,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/include/uapi/asm-generic/unistd.h b/tools/include/uapi/asm-generic/unistd.h
index a627acc8fb5f..5b7e77a7c736 100644
--- a/tools/include/uapi/asm-generic/unistd.h
+++ b/tools/include/uapi/asm-generic/unistd.h
@@ -863,8 +863,12 @@ __SYSCALL(__NR_listns, sys_listns)
#define __NR_rseq_slice_yield 471
__SYSCALL(__NR_rseq_slice_yield, sys_rseq_slice_yield)
+/* fs/open.c */
+#define __NR_fchroot 472
+__SYSCALL(__NR_fchroot, sys_fchroot)
+
#undef __NR_syscalls
-#define __NR_syscalls 472
+#define __NR_syscalls 473
/*
* 32 bit systems traditionally used different
diff --git a/tools/perf/arch/arm/entry/syscalls/syscall.tbl b/tools/perf/arch/arm/entry/syscalls/syscall.tbl
index 94351e22bfcf..55717ed32c27 100644
--- a/tools/perf/arch/arm/entry/syscalls/syscall.tbl
+++ b/tools/perf/arch/arm/entry/syscalls/syscall.tbl
@@ -486,3 +486,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/perf/arch/mips/entry/syscalls/syscall_n64.tbl b/tools/perf/arch/mips/entry/syscalls/syscall_n64.tbl
index 630aab9e5425..83dc93a0712f 100644
--- a/tools/perf/arch/mips/entry/syscalls/syscall_n64.tbl
+++ b/tools/perf/arch/mips/entry/syscalls/syscall_n64.tbl
@@ -386,3 +386,4 @@
469 n64 file_setattr sys_file_setattr
470 n64 listns sys_listns
471 n64 rseq_slice_yield sys_rseq_slice_yield
+472 n64 fchroot sys_fchroot
diff --git a/tools/perf/arch/powerpc/entry/syscalls/syscall.tbl b/tools/perf/arch/powerpc/entry/syscalls/syscall.tbl
index 4fcc7c58a105..cfbb70039ff0 100644
--- a/tools/perf/arch/powerpc/entry/syscalls/syscall.tbl
+++ b/tools/perf/arch/powerpc/entry/syscalls/syscall.tbl
@@ -562,3 +562,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 nospu rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/perf/arch/s390/entry/syscalls/syscall.tbl b/tools/perf/arch/s390/entry/syscalls/syscall.tbl
index 09a7ef04d979..1b45e68a217b 100644
--- a/tools/perf/arch/s390/entry/syscalls/syscall.tbl
+++ b/tools/perf/arch/s390/entry/syscalls/syscall.tbl
@@ -398,3 +398,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/perf/arch/sh/entry/syscalls/syscall.tbl b/tools/perf/arch/sh/entry/syscalls/syscall.tbl
index 70b315cbe710..ace068dff0de 100644
--- a/tools/perf/arch/sh/entry/syscalls/syscall.tbl
+++ b/tools/perf/arch/sh/entry/syscalls/syscall.tbl
@@ -475,3 +475,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/perf/arch/sparc/entry/syscalls/syscall.tbl b/tools/perf/arch/sparc/entry/syscalls/syscall.tbl
index 7e71bf7fcd14..5b9fe0e8140f 100644
--- a/tools/perf/arch/sparc/entry/syscalls/syscall.tbl
+++ b/tools/perf/arch/sparc/entry/syscalls/syscall.tbl
@@ -517,3 +517,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/perf/arch/x86/entry/syscalls/syscall_32.tbl b/tools/perf/arch/x86/entry/syscalls/syscall_32.tbl
index f832ebd2d79b..2c172ef48dfd 100644
--- a/tools/perf/arch/x86/entry/syscalls/syscall_32.tbl
+++ b/tools/perf/arch/x86/entry/syscalls/syscall_32.tbl
@@ -477,3 +477,4 @@
469 i386 file_setattr sys_file_setattr
470 i386 listns sys_listns
471 i386 rseq_slice_yield sys_rseq_slice_yield
+472 i386 fchroot sys_fchroot
diff --git a/tools/perf/arch/x86/entry/syscalls/syscall_64.tbl b/tools/perf/arch/x86/entry/syscalls/syscall_64.tbl
index 524155d655da..d5b6045b0090 100644
--- a/tools/perf/arch/x86/entry/syscalls/syscall_64.tbl
+++ b/tools/perf/arch/x86/entry/syscalls/syscall_64.tbl
@@ -396,6 +396,7 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
#
# Due to a historical design error, certain syscalls are numbered differently
diff --git a/tools/perf/arch/xtensa/entry/syscalls/syscall.tbl b/tools/perf/arch/xtensa/entry/syscalls/syscall.tbl
index a9bca4e484de..d354bb231796 100644
--- a/tools/perf/arch/xtensa/entry/syscalls/syscall.tbl
+++ b/tools/perf/arch/xtensa/entry/syscalls/syscall.tbl
@@ -442,3 +442,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/scripts/syscall.tbl b/tools/scripts/syscall.tbl
index 7a42b32b6577..0ab531605120 100644
--- a/tools/scripts/syscall.tbl
+++ b/tools/scripts/syscall.tbl
@@ -412,3 +412,4 @@
469 common file_setattr sys_file_setattr
470 common listns sys_listns
471 common rseq_slice_yield sys_rseq_slice_yield
+472 common fchroot sys_fchroot
diff --git a/tools/testing/selftests/Makefile b/tools/testing/selftests/Makefile
index b622052ec3e9..4854a8eda449 100644
--- a/tools/testing/selftests/Makefile
+++ b/tools/testing/selftests/Makefile
@@ -33,6 +33,7 @@ TARGETS += fchmodat2
TARGETS += filesystems
TARGETS += filesystems/binderfs
TARGETS += filesystems/epoll
+TARGETS += filesystems/failfs
TARGETS += filesystems/fat
TARGETS += filesystems/overlayfs
TARGETS += filesystems/statmount
diff --git a/tools/testing/selftests/filesystems/failfs/.gitignore b/tools/testing/selftests/filesystems/failfs/.gitignore
new file mode 100644
index 000000000000..cd3b5d884d7e
--- /dev/null
+++ b/tools/testing/selftests/filesystems/failfs/.gitignore
@@ -0,0 +1,2 @@
+# SPDX-License-Identifier: GPL-2.0-only
+failfs_test
diff --git a/tools/testing/selftests/filesystems/failfs/Makefile b/tools/testing/selftests/filesystems/failfs/Makefile
new file mode 100644
index 000000000000..3c5d98b4fe72
--- /dev/null
+++ b/tools/testing/selftests/filesystems/failfs/Makefile
@@ -0,0 +1,5 @@
+# SPDX-License-Identifier: GPL-2.0
+CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
+TEST_GEN_PROGS := failfs_test
+
+include ../../lib.mk
diff --git a/tools/testing/selftests/filesystems/failfs/failfs_test.c b/tools/testing/selftests/filesystems/failfs/failfs_test.c
new file mode 100644
index 000000000000..29a3c294127e
--- /dev/null
+++ b/tools/testing/selftests/filesystems/failfs/failfs_test.c
@@ -0,0 +1,585 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <link.h>
+#include <sched.h>
+#include <signal.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/types.h>
+#include <sys/vfs.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef __NR_fchroot
+#define __NR_fchroot 472
+#endif
+
+#ifndef FD_PIDFS_ROOT
+#define FD_PIDFS_ROOT -10002
+#endif
+
+#ifndef FD_NSFS_ROOT
+#define FD_NSFS_ROOT -10003
+#endif
+
+#ifndef FD_FAILFS_ROOT
+#define FD_FAILFS_ROOT -10004
+#endif
+
+#define NOBODY_UID 65534
+
+/* Child sentinel exit code: the exec was blocked as expected. */
+#define FAILFS_EXEC_BLOCKED 99
+
+/* Stack for the CLONE_FS helper in fchroot_sentinel_shared_fs_struct. */
+#define FAILFS_CLONE_STACK (64 * 1024)
+
+static int sys_fchroot(int fd, unsigned int flags)
+{
+ return syscall(__NR_fchroot, fd, flags);
+}
+
+/*
+ * Raw syscall: glibc's getcwd() rejects the kernel's "(unreachable)"
+ * result and falls back to a generic implementation.
+ */
+static long sys_getcwd(char *buf, size_t size)
+{
+ return syscall(__NR_getcwd, buf, size);
+}
+
+static int drop_to_nobody(void)
+{
+ return setresuid(NOBODY_UID, NOBODY_UID, NOBODY_UID);
+}
+
+/* Parked CLONE_FS child; dies with its parent so it never leaks. */
+static int failfs_park(void *arg)
+{
+ pid_t parent = (pid_t)(long)arg;
+
+ prctl(PR_SET_PDEATHSIG, SIGKILL);
+ /* The parent may have died before the death signal was armed. */
+ if (getppid() != parent)
+ _exit(0);
+ pause();
+ return 0;
+}
+
+/* Is fd a dynamically linked ELF with an absolute PT_INTERP interpreter? */
+static int elf_has_absolute_interp(int fd)
+{
+ ElfW(Ehdr) ehdr;
+ ElfW(Phdr) phdr;
+ char interp;
+ int i;
+
+ if (pread(fd, &ehdr, sizeof(ehdr), 0) != sizeof(ehdr))
+ return 0;
+ if (memcmp(ehdr.e_ident, ELFMAG, SELFMAG) != 0)
+ return 0;
+
+ for (i = 0; i < ehdr.e_phnum; i++) {
+ if (pread(fd, &phdr, sizeof(phdr),
+ ehdr.e_phoff + i * sizeof(phdr)) != sizeof(phdr))
+ return 0;
+ if (phdr.p_type != PT_INTERP)
+ continue;
+ if (pread(fd, &interp, 1, phdr.p_offset) != 1)
+ return 0;
+ return interp == '/';
+ }
+
+ return 0;
+}
+
+TEST(fchdir_sentinel)
+{
+ char buf[PATH_MAX];
+ int fd;
+
+ ASSERT_EQ(fchdir(FD_FAILFS_ROOT), 0);
+
+ /* The working directory is unreachable from the process root. */
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strncmp(buf, "(unreachable)", 13), 0);
+
+ /* Every AT_FDCWD-relative lookup fails. */
+ ASSERT_EQ(openat(AT_FDCWD, "foo", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(openat(AT_FDCWD, ".", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(openat(AT_FDCWD, "..", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(openat(AT_FDCWD, "foo", O_WRONLY | O_CREAT, 0600), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The cwd cannot be pinned by following /proc/self/cwd into it. */
+ ASSERT_EQ(open("/proc/self/cwd", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The root is untouched so absolute lookups keep working... */
+ fd = open("/", O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ /* ... and the working directory can be recovered. */
+ ASSERT_EQ(chdir("/"), 0);
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strcmp(buf, "/"), 0);
+}
+
+TEST(fchdir_rejects_other_sentinels)
+{
+ ASSERT_EQ(fchdir(FD_PIDFS_ROOT), -1);
+ ASSERT_EQ(errno, EBADF);
+ ASSERT_EQ(fchdir(FD_NSFS_ROOT), -1);
+ ASSERT_EQ(errno, EBADF);
+ ASSERT_EQ(fchdir(-10009), -1);
+ ASSERT_EQ(errno, EBADF);
+}
+
+TEST(fchroot_flags)
+{
+ int fd;
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 1), -1);
+ ASSERT_EQ(errno, EINVAL);
+
+ fd = open("/", O_PATH | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(sys_fchroot(fd, 1), -1);
+ ASSERT_EQ(errno, EINVAL);
+ ASSERT_EQ(close(fd), 0);
+}
+
+TEST(fchroot_bad_fd)
+{
+ ASSERT_EQ(sys_fchroot(-1, 0), -1);
+ ASSERT_EQ(errno, EBADF);
+
+ /* Only FD_FAILFS_ROOT is a valid sentinel. */
+ ASSERT_EQ(sys_fchroot(FD_PIDFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EBADF);
+ ASSERT_EQ(sys_fchroot(FD_NSFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EBADF);
+}
+
+TEST(fchroot_notdir)
+{
+ int fd;
+
+ fd = open("/proc/self/status", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(sys_fchroot(fd, 0), -1);
+ ASSERT_EQ(errno, ENOTDIR);
+ ASSERT_EQ(close(fd), 0);
+}
+
+TEST(fchroot_realfd_requires_cap)
+{
+ int fd;
+
+ if (geteuid() == 0)
+ ASSERT_EQ(drop_to_nobody(), 0);
+
+ fd = open("/", O_PATH | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(sys_fchroot(fd, 0), -1);
+ ASSERT_EQ(errno, EPERM);
+ ASSERT_EQ(close(fd), 0);
+}
+
+TEST(fchroot_realfd)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ char path[PATH_MAX];
+ struct stat st;
+ int tmpfd, dfd, fd;
+
+ if (geteuid() != 0)
+ SKIP(return, "fchroot() with a regular fd requires CAP_SYS_CHROOT");
+
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+
+ ASSERT_NE(mkdtemp(template), NULL);
+ snprintf(path, sizeof(path), "%s/canary", template);
+ fd = open(path, O_WRONLY | O_CREAT, 0600);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ dfd = open(template, O_PATH | O_DIRECTORY);
+ ASSERT_GE(dfd, 0);
+ ASSERT_EQ(sys_fchroot(dfd, 0), 0);
+ ASSERT_EQ(close(dfd), 0);
+
+ ASSERT_EQ(stat("/canary", &st), 0);
+
+ /* Best-effort cleanup: dirfd-anchored I/O works with the new root. */
+ snprintf(path, sizeof(path), "%s/canary", template + strlen("/tmp/"));
+ unlinkat(tmpfd, path, 0);
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+}
+
+TEST(fchroot_sentinel)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ struct stat realroot, st;
+ struct statfs sfs;
+ char buf[PATH_MAX];
+ int procfd, tmpfd, dfd, fd;
+ struct {
+ struct file_handle handle;
+ unsigned char f_handle[MAX_HANDLE_SZ];
+ } fh;
+ int mntid;
+ ssize_t ret;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ ASSERT_EQ(stat("/", &realroot), 0);
+ procfd = open("/proc", O_PATH | O_DIRECTORY);
+ ASSERT_GE(procfd, 0);
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+ ASSERT_NE(mkdtemp(template), NULL);
+ dfd = open(template, O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(dfd, 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /* Absolute lookups fail. */
+ ASSERT_EQ(open("/etc/passwd", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(mkdir("/foo", 0700), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /*
+ * The root cannot be referenced at all - not even an O_PATH open,
+ * which skips ->permission(), because it lands on the root as a
+ * jumped walk terminal that ->d_weak_revalidate() refuses.
+ */
+ ASSERT_EQ(open("/", O_RDONLY | O_DIRECTORY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(open("/", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+ ASSERT_EQ(statfs("/", &sfs), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /*
+ * It cannot be pinned by following /proc/self/root into it either
+ * (only the root is in failfs here, so self/cwd is still real).
+ */
+ ASSERT_EQ(openat(procfd, "self/root", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* Nor encoded into a file handle. */
+ fh.handle.handle_bytes = MAX_HANDLE_SZ;
+ ASSERT_EQ(name_to_handle_at(AT_FDCWD, "/", &fh.handle, &mntid, 0), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The working directory is now unreachable from the root. */
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strncmp(buf, "(unreachable)", 13), 0);
+
+ /* Lookups anchored at real directories keep working. */
+ fd = openat(AT_FDCWD, ".", O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+ fd = openat(dfd, "canary", O_WRONLY | O_CREAT, 0600);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(write(fd, "x", 1), 1);
+ ASSERT_EQ(close(fd), 0);
+ fd = openat(dfd, "canary", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ /* ".." walks clamp at the top of the mount tree, not at failfs. */
+ fd = openat(AT_FDCWD, "../../../../../../../../../..", O_PATH);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstat(fd, &st), 0);
+ ASSERT_EQ(st.st_dev, realroot.st_dev);
+ ASSERT_EQ(st.st_ino, realroot.st_ino);
+ ASSERT_EQ(close(fd), 0);
+
+ /* readlink of the magic link still works: it does not follow. */
+ ret = readlinkat(procfd, "self/root", buf, sizeof(buf) - 1);
+ ASSERT_GT(ret, 0);
+ buf[ret] = '\0';
+ TH_LOG("/proc/self/root points to '%s'", buf);
+ /* d_path() names the failfs root synthetically, never as a real path. */
+ ASSERT_EQ(strcmp(buf, "failfs:/"), 0);
+
+ /* But following it into failfs is refused. */
+ ASSERT_EQ(fstatat(procfd, "self/root", &st, 0), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* Best-effort cleanup via the pre-opened dirfds. */
+ unlinkat(dfd, "canary", 0);
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+}
+
+TEST(fchroot_sentinel_absolute_symlink)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ int tmpfd, dfd, fd;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+ ASSERT_NE(mkdtemp(template), NULL);
+ dfd = open(template, O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(dfd, 0);
+
+ fd = openat(dfd, "target", O_WRONLY | O_CREAT, 0600);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+ ASSERT_EQ(symlinkat("target", dfd, "rel"), 0);
+ ASSERT_EQ(symlinkat("/etc", dfd, "abs"), 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /* Relative symlinks keep resolving within the dirfd-anchored walk... */
+ fd = openat(dfd, "rel", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(close(fd), 0);
+
+ /* ... absolute symlinks restart the walk at the failfs root. */
+ ASSERT_EQ(openat(dfd, "abs", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* Best-effort cleanup via the pre-opened dirfds. */
+ unlinkat(dfd, "abs", 0);
+ unlinkat(dfd, "rel", 0);
+ unlinkat(dfd, "target", 0);
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+}
+
+TEST(fchroot_sentinel_unprivileged)
+{
+ char buf[PATH_MAX];
+
+ if (geteuid() == 0)
+ ASSERT_EQ(drop_to_nobody(), 0);
+
+ /* Without no_new_privs entering failfs is not allowed... */
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EPERM);
+
+ /* ... with no_new_privs set it is allowed. */
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ ASSERT_EQ(open("/etc/passwd", O_RDONLY), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* The task counts as chrooted: no user namespaces anymore. */
+ ASSERT_EQ(unshare(CLONE_NEWUSER), -1);
+ ASSERT_EQ(errno, EPERM);
+
+ /* With both root and cwd in failfs getcwd() reports "/". */
+ ASSERT_EQ(fchdir(FD_FAILFS_ROOT), 0);
+ ASSERT_GT(sys_getcwd(buf, sizeof(buf)), 0);
+ ASSERT_EQ(strcmp(buf, "/"), 0);
+}
+
+TEST(fchroot_sentinel_rejected_when_chrooted)
+{
+ char template[] = "/tmp/failfs_test.XXXXXX";
+ int tmpfd;
+
+ if (geteuid() != 0)
+ SKIP(return, "chroot() requires CAP_SYS_CHROOT");
+
+ tmpfd = open("/tmp", O_PATH | O_DIRECTORY);
+ ASSERT_GE(tmpfd, 0);
+ ASSERT_NE(mkdtemp(template), NULL);
+ ASSERT_EQ(chroot(template), 0);
+ ASSERT_EQ(chdir("/"), 0);
+
+ /* Remove the jail while still privileged; sticky /tmp blocks nobody. */
+ unlinkat(tmpfd, template + strlen("/tmp/"), AT_REMOVEDIR);
+
+ ASSERT_EQ(drop_to_nobody(), 0);
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+
+ /* An unprivileged chrooted task must not lift its ".." barrier. */
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EPERM);
+}
+
+TEST(fchroot_sentinel_shared_fs_struct)
+{
+ char stack[FAILFS_CLONE_STACK];
+ pid_t pid;
+
+ if (geteuid() == 0)
+ ASSERT_EQ(drop_to_nobody(), 0);
+
+ /* A CLONE_FS sibling shares the fs_struct: bump fs->users to 2. */
+ pid = clone(failfs_park, stack + sizeof(stack), CLONE_FS | SIGCHLD,
+ (void *)(long)getpid());
+ ASSERT_GE(pid, 0);
+
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+
+ /*
+ * A sibling without no_new_privs could exec a setuid binary with
+ * the failfs root, so a shared fs_struct is refused even with
+ * no_new_privs set.
+ */
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), -1);
+ ASSERT_EQ(errno, EINVAL);
+
+ ASSERT_EQ(kill(pid, SIGKILL), 0);
+ ASSERT_EQ(waitpid(pid, NULL, 0), pid);
+}
+
+TEST(fchroot_sentinel_no_overmount)
+{
+ if (geteuid() != 0)
+ SKIP(return, "mounting requires privileges");
+
+ /*
+ * Contain the blast radius: if failfs ever regressed and "/"
+ * resolved to the real root, the tmpfs mount below must not touch
+ * the host. A private mount namespace keeps it local to this child.
+ */
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /*
+ * Nothing can be mounted on top of the failfs root. It cannot even
+ * be named as a mount target: resolving "/" is refused before the
+ * mount machinery (which, failfs being in no mount namespace, would
+ * reject it anyway) is ever reached. open_tree(OPEN_TREE_CLONE) is
+ * likewise moot since no fd to the root can be obtained.
+ */
+ ASSERT_EQ(mount("none", "/", "tmpfs", 0, NULL), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+}
+
+TEST(fchroot_sentinel_setns_escape)
+{
+ struct stat realroot, st;
+ int nsfd;
+
+ if (geteuid() != 0)
+ SKIP(return, "setns() to a mount namespace requires privileges");
+
+ ASSERT_EQ(stat("/", &realroot), 0);
+ nsfd = open("/proc/self/ns/mnt", O_RDONLY);
+ ASSERT_GE(nsfd, 0);
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+ ASSERT_EQ(open("/etc", O_PATH), -1);
+ ASSERT_EQ(errno, EOPNOTSUPP);
+
+ /* A mount namespace fd is the key out: it resets root and cwd. */
+ ASSERT_EQ(setns(nsfd, CLONE_NEWNS), 0);
+ ASSERT_EQ(close(nsfd), 0);
+
+ ASSERT_EQ(stat("/", &st), 0);
+ ASSERT_EQ(st.st_dev, realroot.st_dev);
+ ASSERT_EQ(st.st_ino, realroot.st_ino);
+}
+
+TEST(fchroot_sentinel_exec)
+{
+ pid_t pid;
+ int status;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /*
+ * Exec in a child: a wrongly successful exec would replace the test
+ * image and its exit code would not match the sentinel below.
+ */
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ execl("/bin/true", "true", NULL);
+ _exit(errno == EOPNOTSUPP ? FAILFS_EXEC_BLOCKED : 1);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), FAILFS_EXEC_BLOCKED);
+}
+
+TEST(fchroot_sentinel_exec_interpreter)
+{
+ static const char * const argv[] = { "failfs_test", NULL };
+ static const char * const envp[] = { NULL };
+ pid_t pid;
+ int status, exefd;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ /* Exec ourselves: the one binary guaranteed to be around. */
+ exefd = open("/proc/self/exe", O_RDONLY);
+ ASSERT_GE(exefd, 0);
+ if (!elf_has_absolute_interp(exefd))
+ SKIP(return, "test binary has no absolute PT_INTERP interpreter");
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ /*
+ * The binary itself needs no path lookup - it is executed by fd -
+ * but loading it fails on opening the absolute PT_INTERP
+ * interpreter. Run it in a child so a wrongly successful exec does
+ * not replace the test image and masquerade as a pass.
+ */
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ syscall(__NR_execveat, exefd, "", argv, envp, AT_EMPTY_PATH);
+ _exit(errno == EOPNOTSUPP ? FAILFS_EXEC_BLOCKED : 1);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), FAILFS_EXEC_BLOCKED);
+}
+
+TEST(fchroot_sentinel_inherited)
+{
+ pid_t pid;
+ int status;
+
+ if (geteuid() != 0)
+ SKIP(return, "privileged fchroot(FD_FAILFS_ROOT) requires CAP_SYS_CHROOT");
+
+ ASSERT_EQ(sys_fchroot(FD_FAILFS_ROOT, 0), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ if (open("/etc", O_PATH) != -1 || errno != EOPNOTSUPP)
+ _exit(1);
+ _exit(0);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), 0);
+}
+
+TEST_HARNESS_MAIN