summaryrefslogtreecommitdiff
path: root/tools/testing
diff options
context:
space:
mode:
Diffstat (limited to 'tools/testing')
-rw-r--r--tools/testing/selftests/exec/.gitignore11
-rw-r--r--tools/testing/selftests/exec/Makefile93
-rw-r--r--tools/testing/selftests/exec/binfmt_bind_interp.c14
-rw-r--r--tools/testing/selftests/exec/binfmt_bpf_app.c12
-rw-r--r--tools/testing/selftests/exec/binfmt_bpf_interp.c15
-rw-r--r--tools/testing/selftests/exec/binfmt_loader_payload.c146
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_bpf.c638
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_common.h315
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_disabled.c172
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_interplimit.c232
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_loader.c372
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_selfpin.c158
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_transparent.c95
-rw-r--r--tools/testing/selftests/exec/binfmt_transparent_interp.c112
-rw-r--r--tools/testing/selftests/exec/bpf_interp.bpf.c61
-rw-r--r--tools/testing/selftests/exec/config10
-rw-r--r--tools/testing/selftests/exec/interp_bind.bpf.c76
-rw-r--r--tools/testing/selftests/exec/loader.bpf.c56
-rw-r--r--tools/testing/selftests/exec/nix_origin.bpf.c224
-rw-r--r--tools/testing/selftests/exec/transparent.bpf.c57
20 files changed, 2869 insertions, 0 deletions
diff --git a/tools/testing/selftests/exec/.gitignore b/tools/testing/selftests/exec/.gitignore
index 7f3d1ae762ec..e42ecd4c908d 100644
--- a/tools/testing/selftests/exec/.gitignore
+++ b/tools/testing/selftests/exec/.gitignore
@@ -19,3 +19,14 @@ null-argv
xxxxxxxx*
pipe
S_I*.test
+binfmt_misc_bpf
+binfmt_misc_interplimit
+binfmt_bpf_interp
+binfmt_bpf_app
+binfmt_misc_transparent
+binfmt_transparent_interp
+binfmt_misc_loader
+binfmt_loader_payload
+binfmt_loader_payload_static
+*.bpf.o
+vmlinux.h
diff --git a/tools/testing/selftests/exec/Makefile b/tools/testing/selftests/exec/Makefile
index 45a3cfc435cf..b640af8f02b5 100644
--- a/tools/testing/selftests/exec/Makefile
+++ b/tools/testing/selftests/exec/Makefile
@@ -21,9 +21,56 @@ TEST_GEN_PROGS += recursion-depth
TEST_GEN_PROGS += null-argv
TEST_GEN_PROGS += check-exec
+# binfmt_misc must not be reachable as an exec source or as a stacking layer,
+# or an 'F' entry can pin the instance that owns it. Unprivileged, no bpf.
+TEST_GEN_PROGS += binfmt_misc_selfpin
+
+# The interpreters an 'F' or 'B' entry pre-opens are charged against
+# UCOUNT_BINFMT_MISC_INTERPRETERS. Unprivileged, no bpf.
+TEST_GEN_PROGS += binfmt_misc_interplimit
+
+# 'D' (register disabled) binfmt_misc test: an entry that exists but does
+# not dispatch until it is enabled. Static magic entry, no bpf toolchain.
+TEST_GEN_PROGS += binfmt_misc_disabled
+
+# Static ('T' flag) transparent binfmt_misc test; the asserting interpreter
+# is shared with the bpf harness's transparent case. No bpf toolchain needed.
+TEST_GEN_PROGS += binfmt_misc_transparent
+TEST_GEN_FILES += binfmt_transparent_interp
+
+# 'L' (loader substitution) binfmt_misc test: the payload runs as the main
+# image with a copy of the system loader substituted for its PT_INTERP and
+# asserts the native identity from inside; the static build proves the
+# override is dropped for a binary without PT_INTERP.
+TEST_GEN_PROGS += binfmt_misc_loader
+TEST_GEN_FILES += binfmt_loader_payload binfmt_loader_payload_static
+
+# binfmt_misc bpf-backed ('B') handler test: a libbpf harness plus its
+# struct_ops objects and the test interpreter/app it routes between. Only
+# built when clang, bpftool, the vmlinux BTF and libbpf are all present
+# (HAVE_BPF_TOOLCHAIN=y forces it) so the other exec selftests don't grow
+# a bpf toolchain dependency.
+CLANG ?= clang
+BPFTOOL ?= bpftool
+VMLINUX_BTF ?= /sys/kernel/btf/vmlinux
+HAVE_BPF_TOOLCHAIN ?= $(shell command -v $(CLANG) >/dev/null 2>&1 && \
+ command -v $(BPFTOOL) >/dev/null 2>&1 && \
+ test -r $(VMLINUX_BTF) && \
+ pkg-config --exists libbpf 2>/dev/null && echo y)
+ifeq ($(HAVE_BPF_TOOLCHAIN),y)
+TEST_GEN_PROGS += binfmt_misc_bpf
+TEST_GEN_FILES += bpf_interp.bpf.o nix_origin.bpf.o transparent.bpf.o
+TEST_GEN_FILES += loader.bpf.o interp_bind.bpf.o
+TEST_GEN_FILES += binfmt_bpf_interp binfmt_bpf_app binfmt_bind_interp
+else
+$(info exec selftests: skipping binfmt_misc_bpf, needs clang, bpftool, vmlinux BTF and libbpf)
+endif
+
EXTRA_CLEAN := $(OUTPUT)/subdir.moved $(OUTPUT)/execveat.moved $(OUTPUT)/xxxxx* \
$(OUTPUT)/S_I*.test
+LOCAL_HDRS += binfmt_misc_common.h
+
include ../lib.mk
CHECK_EXEC_SAMPLES := $(top_srcdir)/samples/check-exec
@@ -55,3 +102,49 @@ $(OUTPUT)/script-exec.inc: $(CHECK_EXEC_SAMPLES)/script-exec.inc
cp $< $@
$(OUTPUT)/script-noexec.inc: $(CHECK_EXEC_SAMPLES)/script-noexec.inc
cp $< $@
+
+# Reuses setup_userns()/write_file() from the filesystems selftests. Their
+# wrappers.h wants the uapi headers, so ask for them here rather than widening
+# CFLAGS for every program in this directory.
+$(OUTPUT)/binfmt_misc_selfpin: CFLAGS += $(TOOLS_INCLUDES)
+$(OUTPUT)/binfmt_misc_selfpin: ../filesystems/utils.c
+$(OUTPUT)/binfmt_misc_interplimit: CFLAGS += $(TOOLS_INCLUDES)
+$(OUTPUT)/binfmt_misc_interplimit: ../filesystems/utils.c
+
+# --- binfmt_misc bpf ('B') handler test ---------------------------------
+# The struct_ops bpf objects are compiled against the running kernel's BTF.
+# CLANG/BPFTOOL/VMLINUX_BTF are set above next to the toolchain check;
+# override LIBBPF_CFLAGS/LDLIBS to point at a libbpf install.
+BPF_CFLAGS ?= -I$(OUTPUT)
+LIBBPF_CFLAGS ?=
+LIBBPF_LDLIBS ?= -lbpf -lelf -lz
+
+$(OUTPUT)/vmlinux.h:
+ $(BPFTOOL) btf dump file $(VMLINUX_BTF) format c > $@
+
+# BPF_NO_KFUNC_PROTOTYPES: the programs declare the kfuncs they use themselves.
+$(OUTPUT)/%.bpf.o: %.bpf.c $(OUTPUT)/vmlinux.h
+ $(CLANG) -g -O2 -target bpf -mcpu=v3 -DBPF_NO_KFUNC_PROTOTYPES \
+ $(BPF_CFLAGS) $(LIBBPF_CFLAGS) -c $< -o $@
+
+$(OUTPUT)/binfmt_misc_bpf: binfmt_misc_bpf.c binfmt_misc_common.h
+ $(CC) $(CFLAGS) $(LIBBPF_CFLAGS) $(LDFLAGS) $< $(LIBBPF_LDLIBS) -o $@
+
+$(OUTPUT)/binfmt_bpf_interp: binfmt_bpf_interp.c
+ $(CC) $(CFLAGS) $(LDFLAGS) $< -o $@
+
+$(OUTPUT)/binfmt_bind_interp: binfmt_bind_interp.c
+ $(CC) $(CFLAGS) $(LDFLAGS) $< -o $@
+
+$(OUTPUT)/binfmt_loader_payload: binfmt_loader_payload.c binfmt_misc_common.h
+ $(CC) $(CFLAGS) $(LDFLAGS) -fPIE -pie $< -o $@
+
+$(OUTPUT)/binfmt_loader_payload_static: binfmt_loader_payload.c binfmt_misc_common.h
+ $(CC) $(CFLAGS) $(LDFLAGS) -static $< -o $@
+
+# PT_INTERP is set to the literal "$ORIGIN/binfmt_bpf_interp"; the nix_origin
+# handler resolves it relative to the binary at run time.
+$(OUTPUT)/binfmt_bpf_app: binfmt_bpf_app.c
+ $(CC) $(CFLAGS) $(LDFLAGS) -Wl,--dynamic-linker,'$$ORIGIN/binfmt_bpf_interp' $< -o $@
+
+EXTRA_CLEAN += $(OUTPUT)/vmlinux.h $(OUTPUT)/*.bpf.o
diff --git a/tools/testing/selftests/exec/binfmt_bind_interp.c b/tools/testing/selftests/exec/binfmt_bind_interp.c
new file mode 100644
index 000000000000..06d65062856b
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bind_interp.c
@@ -0,0 +1,14 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test interpreter for the bound-interpreter case of the binfmt_misc_bpf
+ * selftest. Two copies are installed at different paths and bound to one
+ * entry under different names; printing argv[0] - the path the kernel ran
+ * this copy under - tells the harness which of them the load program picked.
+ */
+#include <stdio.h>
+
+int main(int argc, char **argv)
+{
+ printf("BIND_RAN %s\n", argc > 0 ? argv[0] : "");
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_bpf_app.c b/tools/testing/selftests/exec/binfmt_bpf_app.c
new file mode 100644
index 000000000000..472270f148bc
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bpf_app.c
@@ -0,0 +1,12 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A relocatable binary for the binfmt_misc_bpf $ORIGIN case. The Makefile
+ * links it with PT_INTERP set to the literal "$ORIGIN/binfmt_bpf_interp"
+ * (-Wl,--dynamic-linker), which the kernel ELF loader cannot resolve. The
+ * nix_origin bpf handler resolves it relative to this binary's directory and
+ * routes execution to the co-located interpreter.
+ */
+int main(void)
+{
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_bpf_interp.c b/tools/testing/selftests/exec/binfmt_bpf_interp.c
new file mode 100644
index 000000000000..2db205f095b2
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bpf_interp.c
@@ -0,0 +1,15 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test interpreter for the binfmt_misc_bpf selftest. A bpf-backed 'B' handler
+ * routes a matched binary here; printing this marker proves the program's
+ * chosen interpreter actually ran.
+ */
+#include <unistd.h>
+
+int main(int argc, char **argv)
+{
+ (void)argc;
+ (void)argv;
+ write(1, "BPF_INTERP_RAN\n", 15);
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_loader_payload.c b/tools/testing/selftests/exec/binfmt_loader_payload.c
new file mode 100644
index 000000000000..272db8efb4b5
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_loader_payload.c
@@ -0,0 +1,146 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Payload for the binfmt_misc 'L' (loader substitution) selftest. It is
+ * executed as the MAIN image - a fully native exec - with the registered
+ * interpreter substituted for its PT_INTERP, and asserts the native
+ * identity from the inside. Exits 0 when every surface checks out.
+ *
+ * Modes, selected by the orchestrator via the environment:
+ * - default: full assertions, path-based ones included
+ * - BINFMT_TEST_MEMFD=1: executed from an inaccessible memfd, skip
+ * the path-based assertions
+ * - BINFMT_TEST_STATIC=1: static build; the override was dropped, so
+ * expect no interpreter at all
+ */
+#define _GNU_SOURCE
+#include <elf.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/auxv.h>
+#include <unistd.h>
+
+#include "binfmt_misc_common.h"
+
+/* Start of our own mapped image, courtesy of the linker. */
+extern const char __ehdr_start[];
+
+/* An image is never this large; used to bracket "within our image". */
+#define IMAGE_SPAN (16UL << 20)
+
+static int failed;
+
+static void check(int cond, const char *what)
+{
+ if (cond)
+ return;
+ fprintf(stderr, "[payload] FAILED: %s (errno %d)\n", what, errno);
+ failed = 1;
+}
+
+/* Return whether /proc/self/maps names a path starting with @prefix. */
+static int maps_has_prefix(const char *prefix)
+{
+ char *line = NULL;
+ size_t len = 0;
+ int found = 0;
+ FILE *f;
+
+ f = fopen("/proc/self/maps", "r");
+ if (!f)
+ return -1;
+ while (getline(&line, &len, f) > 0) {
+ char *path = strchr(line, '/');
+
+ if (path && !strncmp(path, prefix, strlen(prefix))) {
+ found = 1;
+ break;
+ }
+ }
+ free(line);
+ fclose(f);
+ return found;
+}
+
+int main(int argc, char *argv[])
+{
+ const char *binary = getenv("BINFMT_TEST_BINARY");
+ const char *interp = getenv("BINFMT_TEST_INTERP");
+ int memfd_mode = getenv("BINFMT_TEST_MEMFD") != NULL;
+ int static_mode = getenv("BINFMT_TEST_STATIC") != NULL;
+ unsigned long self = (unsigned long)__ehdr_start;
+ unsigned long base = getauxval(AT_BASE);
+ unsigned long phdr = getauxval(AT_PHDR);
+ unsigned long entry = getauxval(AT_ENTRY);
+ unsigned long start_code, end_code;
+
+ /* The argument vector is exactly what the caller built. */
+ check(argc == 3 && !strcmp(argv[0], PAYLOAD_ARGV0) &&
+ !strcmp(argv[1], PAYLOAD_ARG1) && !strcmp(argv[2], PAYLOAD_ARG2),
+ "argv was rewritten");
+
+ /* Native from birth: no execfd, no dispatch marker. */
+ check(getauxval(AT_EXECFD) == 0, "AT_EXECFD present");
+ check(getauxval(AT_FLAGS) == 0, "AT_FLAGS not native");
+
+ if (static_mode) {
+ /* The override was dropped: no interpreter was loaded. */
+ check(base == 0, "AT_BASE set for a static payload");
+ } else {
+ /* A loader is mapped in the interpreter slot, not our image. */
+ check(base != 0, "AT_BASE missing");
+ check(base < self || base >= self + IMAGE_SPAN,
+ "AT_BASE inside our own image");
+ }
+
+ /* We occupy the main-image slot. */
+ check(phdr >= self && phdr < self + IMAGE_SPAN,
+ "AT_PHDR outside our image");
+ check(entry >= self && entry < self + IMAGE_SPAN,
+ "AT_ENTRY outside our image");
+
+ /* The code statistics markers describe our image, natively placed. */
+ if (stat_codes(getpid(), &start_code, &end_code) == 0) {
+ check(start_code >= self && start_code < end_code &&
+ end_code < self + IMAGE_SPAN,
+ "stat start_code/end_code not our image");
+ check(entry >= start_code && entry < end_code,
+ "AT_ENTRY outside [start_code, end_code)");
+ } else {
+ check(0, "cannot parse /proc/self/stat");
+ }
+
+ if (!memfd_mode && binary) {
+ const char *execfn = (const char *)getauxval(AT_EXECFN);
+ const char *base_name = strrchr(binary, '/');
+
+ base_name = base_name ? base_name + 1 : binary;
+
+ /* exe link, AT_EXECFN and comm all follow the binary. */
+ check(exe_is(binary), "/proc/self/exe");
+ check(execfn && !strcmp(execfn, binary), "AT_EXECFN");
+ check(comm_is(base_name), "comm");
+
+ /* The running binary is write-denied, natively. */
+ check(write_denied(binary), "no ETXTBSY on the binary");
+ }
+
+ if (interp) {
+ int found = maps_has_prefix(interp);
+
+ if (static_mode)
+ /* Nothing was substituted, nothing may be mapped. */
+ check(found == 0, "loader mapped for a static payload");
+ else
+ /* The substituted loader shows under its real path. */
+ check(found == 1, "loader path not in /proc/self/maps");
+ }
+
+ if (failed)
+ return 1;
+ printf("[payload] native identity checks out\n");
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_misc_bpf.c b/tools/testing/selftests/exec/binfmt_misc_bpf.c
new file mode 100644
index 000000000000..b2a4518901b0
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_bpf.c
@@ -0,0 +1,638 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Selftest for binfmt_misc bpf-backed ('B') handlers.
+ *
+ * A handler is a struct binfmt_misc_ops struct_ops map with a sleepable match
+ * and a sleepable load program. Attaching it publishes it by name in the
+ * caller's user namespace; a 'B' entry referencing it by name in the
+ * interpreter field activates it:
+ *
+ * echo ':name:B::::<handler>:' > /proc/sys/fs/binfmt_misc/register
+ *
+ * Five self-contained cases are exercised:
+ *
+ * 1. bpf_interp: the match program matches a synthetic aarch64 ELF header
+ * from the prefetched bprm->buf and the load program routes it to a
+ * fixed interpreter of its choosing.
+ * 2. nix_origin: the match program reads the binary's program headers to
+ * commit only to a "$ORIGIN/..."-relative PT_INTERP and the load program
+ * resolves it to an interpreter co-located with the binary (the
+ * relocatable-loader case the kernel ELF loader cannot express).
+ * 3. transparent: the load program sets BPF_BINPRM_TRANSPARENT; the
+ * asserting interpreter (binfmt_transparent_interp) verifies the
+ * identity the kernel constructed (exe link, argv, cmdline, comm,
+ * AT_EXECFD, write denial) from inside the process.
+ * 4. loader: the load program sets BPF_BINPRM_LOADER; the payload
+ * (binfmt_loader_payload) runs as the main image with the selected
+ * interpreter substituted for its PT_INTERP and asserts the native
+ * identity from inside.
+ * 5. interp_bind: an entry registered disabled with 'D' is given its
+ * interpreters one write at a time, and the load program picks one by
+ * name per exec. Replacing what the path holds afterwards changes
+ * nothing, which is the point of binding a file rather than resolving
+ * a name at exec time. Enabling the entry seals it.
+ *
+ * The first two route to a test interpreter that prints BPF_INTERP_RAN,
+ * proving the program's chosen interpreter actually ran.
+ */
+#define _GNU_SOURCE
+#include <elf.h>
+#include <limits.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+
+#include <bpf/btf.h>
+#include <bpf/libbpf.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define INTERP_PATH "/tmp/binfmt_bpf_interp"
+#define AARCH64_PATH "/tmp/binfmt_bpf_aarch64"
+#define RELOC_TEMPLATE "/tmp/binfmt_relocXXXXXX"
+#define TRANS_INTERP "/tmp/binfmt_transparent_interp"
+#define TRANS_PATH "/tmp/binfmt_bpf_riscv"
+#define EXPECT "BPF_INTERP_RAN"
+#define TRANS_EXPECT "TRANSPARENT_OK"
+#define LOADER_INTERP "/tmp/binfmt_loader_interp"
+#define LOADER_PATH "/tmp/binfmt_bpf_loader.ldrtest"
+#define BIND_FIRST "/tmp/binfmt_bind_first"
+#define BIND_SECOND "/tmp/binfmt_bind_second"
+#define BIND_ARM_PATH "/tmp/binfmt_bind_arm"
+#define BIND_RISCV_PATH "/tmp/binfmt_bind_riscv"
+#define BIND_EXPECT "BIND_RAN "
+#define BIND_MAX 100
+#define INTERP_LIMIT "/proc/sys/user/max_binfmt_misc_interpreters"
+/* Exit status of the binding child when it cannot set up a budget of its own. */
+#define BIND_NO_BUDGET 200
+
+/* A minimal 64-bit little-endian ELF header, padded to the read size. */
+static int create_fake_elf(const char *path, unsigned short machine)
+{
+ unsigned char hdr[256] = {0};
+ int fd;
+
+ hdr[0] = 0x7f; hdr[1] = 'E'; hdr[2] = 'L'; hdr[3] = 'F';
+ hdr[4] = ELFCLASS64;
+ hdr[5] = ELFDATA2LSB;
+ hdr[6] = EV_CURRENT;
+ hdr[16] = ET_EXEC;
+ hdr[18] = machine & 0xff; /* e_machine, little-endian */
+ hdr[19] = machine >> 8;
+ hdr[20] = EV_CURRENT;
+
+ unlink(path);
+ fd = open(path, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, hdr, sizeof(hdr)) != (ssize_t)sizeof(hdr)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+/*
+ * Register a 'B' entry for @handler. With @flags "D" the entry is created
+ * disabled, which is what leaves it open to being given interpreters.
+ */
+static int register_entry(const char *name, const char *handler,
+ const char *flags)
+{
+ char rule[PATH_MAX];
+
+ snprintf(rule, sizeof(rule), ":%s:B::::%s:%s", name, handler,
+ flags ? flags : "");
+ return write_reg(rule);
+}
+
+static int check_output(const char *cmd, const char *expected)
+{
+ char buf[128];
+ FILE *fp;
+
+ fp = popen(cmd, "r");
+ if (!fp)
+ return -1;
+ if (!fgets(buf, sizeof(buf), fp)) {
+ pclose(fp);
+ return -1;
+ }
+ pclose(fp);
+ return strncmp(buf, expected, strlen(expected)) ? -1 : 0;
+}
+
+/* Does the kernel BTF know struct binfmt_misc_ops (CONFIG_BINFMT_MISC_BPF)? */
+static bool have_binfmt_misc_ops(void)
+{
+ struct btf *btf = btf__load_vmlinux_btf();
+ bool have;
+
+ have = btf && btf__find_by_name_kind(btf, "binfmt_misc_ops",
+ BTF_KIND_STRUCT) >= 0;
+ btf__free(btf);
+ return have;
+}
+
+/* The reason bpf handler cases cannot run here, NULL if they can. */
+static const char *bpf_handler_unsupported(void)
+{
+ if (getuid() != 0)
+ return "test must be run as root";
+ if (!have_binfmt_misc_ops())
+ return "no struct binfmt_misc_ops in the kernel BTF (CONFIG_BINFMT_MISC_BPF)";
+ if (!binfmt_misc_available())
+ return "no binfmt_misc";
+ return NULL;
+}
+
+/* An attached handler with its 'B' entry activated. */
+struct bpf_case {
+ struct bpf_object *obj;
+ struct bpf_link *link;
+ const char *entry;
+};
+
+/*
+ * Load @objfile, attach its struct_ops map @handler (which publishes the
+ * handler) and register a 'B' entry named @entry that references it, with
+ * @flags as the entry's register-string flags.
+ */
+static int bpf_case_start_flags(struct bpf_case *c, const char *objfile,
+ const char *handler, const char *entry,
+ const char *flags)
+{
+ struct bpf_map *map;
+
+ c->obj = NULL;
+ c->link = NULL;
+ c->entry = entry;
+
+ c->obj = bpf_object__open_file(objfile, NULL);
+ if (!c->obj || libbpf_get_error(c->obj)) {
+ fprintf(stderr, "open %s failed\n", objfile);
+ c->obj = NULL;
+ return -1;
+ }
+ if (bpf_object__load(c->obj)) {
+ fprintf(stderr, "load %s failed (check dmesg for the verifier log)\n",
+ objfile);
+ goto fail;
+ }
+ map = bpf_object__find_map_by_name(c->obj, handler);
+ if (!map) {
+ fprintf(stderr, "no struct_ops map '%s' in %s\n", handler, objfile);
+ goto fail;
+ }
+ c->link = bpf_map__attach_struct_ops(map);
+ if (!c->link || libbpf_get_error(c->link)) {
+ fprintf(stderr, "attach struct_ops '%s' failed\n", handler);
+ c->link = NULL;
+ goto fail;
+ }
+ if (register_entry(entry, handler, flags)) {
+ fprintf(stderr, "register 'B' entry '%s' failed\n", entry);
+ goto fail;
+ }
+ return 0;
+
+fail:
+ bpf_link__destroy(c->link);
+ bpf_object__close(c->obj);
+ c->obj = NULL;
+ c->link = NULL;
+ return -1;
+}
+
+static int bpf_case_start(struct bpf_case *c, const char *objfile,
+ const char *handler, const char *entry)
+{
+ return bpf_case_start_flags(c, objfile, handler, entry, NULL);
+}
+
+static void bpf_case_stop(struct bpf_case *c)
+{
+ unregister(c->entry);
+ bpf_link__destroy(c->link);
+ bpf_object__close(c->obj);
+}
+
+/* Activate @handler, run @target and check it produced @expect. */
+static int run_case(const char *objfile, const char *handler,
+ const char *entry, const char *target, const char *expect)
+{
+ struct bpf_case c;
+ int ret;
+
+ if (bpf_case_start(&c, objfile, handler, entry))
+ return -1;
+ ret = check_output(target, expect);
+ bpf_case_stop(&c);
+ return ret;
+}
+
+FIXTURE(bpf_handler) {
+ char obj[PATH_MAX]; /* struct_ops object of the case under test */
+};
+
+FIXTURE_SETUP(bpf_handler)
+{
+ char src[PATH_MAX];
+ const char *why = bpf_handler_unsupported();
+
+ if (why)
+ SKIP(return, "%s", why);
+
+ /* Shared test interpreter. */
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_bpf_interp"), 0);
+ ASSERT_EQ(copy_file(src, INTERP_PATH), 0);
+}
+
+FIXTURE_TEARDOWN(bpf_handler)
+{
+ unlink(INTERP_PATH);
+}
+
+/* The match program matches a synthetic header, the load program routes it. */
+TEST_F(bpf_handler, fixed_interpreter)
+{
+ ASSERT_EQ(create_fake_elf(AARCH64_PATH, EM_AARCH64), 0);
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "bpf_interp.bpf.o"), 0);
+ EXPECT_EQ(run_case(self->obj, "bpf_interp", "test_bpf_interp",
+ AARCH64_PATH, EXPECT), 0);
+ unlink(AARCH64_PATH);
+}
+
+/* A "$ORIGIN/..." PT_INTERP resolved to an interpreter next to the binary. */
+TEST_F(bpf_handler, origin_relative_interpreter)
+{
+ char src[PATH_MAX], app[PATH_MAX], interp[PATH_MAX];
+ char dir[] = RELOC_TEMPLATE;
+
+ ASSERT_NE(mkdtemp(dir), NULL);
+ snprintf(app, sizeof(app), "%s/app", dir);
+ snprintf(interp, sizeof(interp), "%s/binfmt_bpf_interp", dir);
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_bpf_app"), 0);
+ ASSERT_EQ(copy_file(src, app), 0);
+ ASSERT_EQ(copy_file(INTERP_PATH, interp), 0);
+
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "nix_origin.bpf.o"), 0);
+ EXPECT_EQ(run_case(self->obj, "nix_origin", "test_bpf_origin",
+ app, EXPECT), 0);
+
+ unlink(app);
+ unlink(interp);
+ rmdir(dir);
+}
+
+/* A transparent dispatch: the process presents as the binary, not the interp. */
+TEST_F(bpf_handler, transparent_dispatch)
+{
+ char src[PATH_MAX], cmd[PATH_MAX + 16];
+
+ /* Probe for transparent-mode support via its static counterpart. */
+ if (!binfmt_flag_supported('T'))
+ SKIP(return, "kernel without transparent mode");
+
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_transparent_interp"), 0);
+ ASSERT_EQ(copy_file(src, TRANS_INTERP), 0);
+ ASSERT_EQ(create_fake_elf(TRANS_PATH, EM_RISCV), 0);
+
+ setenv("BINFMT_TEST_BINARY", TRANS_PATH, 1);
+ snprintf(cmd, sizeof(cmd), "%s argone argtwo", TRANS_PATH);
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "transparent.bpf.o"), 0);
+ EXPECT_EQ(run_case(self->obj, "transparent", "test_bpf_transparent",
+ cmd, TRANS_EXPECT), 0);
+
+ unlink(TRANS_PATH);
+ unlink(TRANS_INTERP);
+}
+
+/* A per-exec loader substitution: the payload runs as a native exec. */
+TEST_F(bpf_handler, loader_substitution)
+{
+ char src[PATH_MAX], loader[PATH_MAX];
+ struct bpf_case c;
+ int status;
+
+ if (find_loader(loader, sizeof(loader)))
+ SKIP(return, "cannot determine own PT_INTERP");
+
+ ASSERT_EQ(copy_file(loader, LOADER_INTERP), 0);
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_loader_payload"), 0);
+ ASSERT_EQ(copy_file(src, LOADER_PATH), 0);
+ ASSERT_EQ(patch_file(LOADER_PATH, EI_PAD, LOADER_MARKER,
+ strlen(LOADER_MARKER)), 0);
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "loader.bpf.o"), 0);
+
+ setenv("BINFMT_TEST_BINARY", LOADER_PATH, 1);
+ setenv("BINFMT_TEST_INTERP", LOADER_INTERP, 1);
+
+ ASSERT_EQ(bpf_case_start(&c, self->obj, "loader", "test_bpf_loader"), 0);
+ status = run_payload(LOADER_PATH);
+ bpf_case_stop(&c);
+ EXPECT_EQ(status, 0);
+
+ unsetenv("BINFMT_TEST_INTERP");
+ unlink(LOADER_PATH);
+ unlink(LOADER_INTERP);
+}
+
+/* The errno an exec of @path fails with, 0 if it succeeded. */
+static int exec_errno(const char *path)
+{
+ int status;
+ pid_t pid;
+
+ pid = fork();
+ if (pid == 0) {
+ execl(path, path, (char *)NULL);
+ _exit(errno);
+ }
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/* Install a copy of the bound-interpreter test binary at @path. */
+static int install_interp(const char *path)
+{
+ char src[PATH_MAX];
+
+ if (artifact_path(src, sizeof(src), "binfmt_bind_interp"))
+ return -1;
+ return copy_file(src, path);
+}
+
+/* Bind @path to @entry under @name, the '+' command of a disabled entry. */
+static int entry_bind(const char *entry, const char *name, const char *path)
+{
+ char cmd[PATH_MAX];
+
+ snprintf(cmd, sizeof(cmd), "+%s %s\n", name, path);
+ return entry_command(entry, cmd);
+}
+
+/* Set the interpreter budget of this namespace. */
+static int write_interp_limit(const char *val)
+{
+ ssize_t n;
+ int fd;
+
+ fd = open(INTERP_LIMIT, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -1;
+ n = write(fd, val, strlen(val));
+ close(fd);
+ return n < 0 ? -1 : 0;
+}
+
+/*
+ * The errno a bind is refused with when the writer is a child that has spent
+ * the budget of a user namespace of its own, 0 if it succeeded and -1 if the
+ * child could not set itself up. The fd is opened here and inherited, so the
+ * interpreter is still opened with this process's credentials.
+ */
+static int bind_out_of_budget(const char *entry, const char *name,
+ const char *path)
+{
+ char cmd[PATH_MAX], file[PATH_MAX];
+ int fd, status, retval;
+ pid_t pid;
+
+ snprintf(file, sizeof(file), BINFMT_DIR "/%s", entry);
+ snprintf(cmd, sizeof(cmd), "+%s %s\n", name, path);
+
+ fd = open(file, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -1;
+
+ pid = fork();
+ if (pid == 0) {
+ ssize_t n;
+
+ /* A namespace of its own, with nothing left in it to spend. */
+ if (unshare(CLONE_NEWUSER) || write_interp_limit("0"))
+ _exit(BIND_NO_BUDGET);
+ n = write(fd, cmd, strlen(cmd));
+ _exit(n < 0 ? errno : 0);
+ }
+ close(fd);
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ retval = WEXITSTATUS(status);
+ return retval == BIND_NO_BUDGET ? -1 : retval;
+}
+
+FIXTURE(bound_interp) {
+ char obj[PATH_MAX];
+ struct bpf_case c;
+ bool started;
+};
+
+FIXTURE_SETUP(bound_interp)
+{
+ const char *why = bpf_handler_unsupported();
+
+ if (why)
+ SKIP(return, "%s", why);
+ if (!binfmt_flag_supported('D')) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'D' flag");
+ }
+
+ ASSERT_EQ(install_interp(BIND_FIRST), 0);
+ ASSERT_EQ(install_interp(BIND_SECOND), 0);
+
+ ASSERT_EQ(artifact_path(self->obj, sizeof(self->obj),
+ "interp_bind.bpf.o"), 0);
+
+ /*
+ * Registered disabled, so it cannot be matched yet and can still be
+ * given interpreters. Each path is resolved once, by its write(2);
+ * from here on the entry holds the files themselves.
+ */
+ ASSERT_EQ(bpf_case_start_flags(&self->c, self->obj, "interp_bind",
+ "test_interp_bind", "D"), 0);
+ self->started = true;
+
+ ASSERT_EQ(entry_bind("test_interp_bind", "first", BIND_FIRST), 0);
+ ASSERT_EQ(entry_bind("test_interp_bind", "second", BIND_SECOND), 0);
+}
+
+FIXTURE_TEARDOWN(bound_interp)
+{
+ if (self->started)
+ bpf_case_stop(&self->c);
+ unlink(BIND_FIRST);
+ unlink(BIND_SECOND);
+ unlink(AARCH64_PATH);
+ unlink(BIND_RISCV_PATH);
+ unlink(BIND_ARM_PATH);
+}
+
+/* Enabling is what makes the configured entry matchable. */
+static int activate(const char *entry)
+{
+ return entry_command(entry, "1\n");
+}
+
+/* One entry, one interpreter per guest architecture, picked per exec. */
+TEST_F(bound_interp, selects_by_name)
+{
+ ASSERT_EQ(create_fake_elf(AARCH64_PATH, EM_AARCH64), 0);
+ ASSERT_EQ(create_fake_elf(BIND_RISCV_PATH, EM_RISCV), 0);
+
+ /* Disabled, so it does not match and no format claims the binary. */
+ EXPECT_EQ(exec_errno(AARCH64_PATH), ENOEXEC);
+
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+ EXPECT_EQ(check_output(AARCH64_PATH, BIND_EXPECT BIND_FIRST), 0);
+ EXPECT_EQ(check_output(BIND_RISCV_PATH, BIND_EXPECT BIND_SECOND), 0);
+}
+
+/* What was bound is what runs, whatever the path holds afterwards. */
+TEST_F(bound_interp, path_no_longer_decides)
+{
+ char other[PATH_MAX];
+
+ ASSERT_EQ(create_fake_elf(AARCH64_PATH, EM_AARCH64), 0);
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+
+ /* Bound interpreters are pinned against writes, exactly like 'F'. */
+ EXPECT_TRUE(write_denied(BIND_FIRST));
+
+ /* Replace the path with a different binary: a new file, new inode. */
+ ASSERT_EQ(artifact_path(other, sizeof(other), "binfmt_bpf_interp"), 0);
+ ASSERT_EQ(unlink(BIND_FIRST), 0);
+ ASSERT_EQ(copy_file(other, BIND_FIRST), 0);
+
+ EXPECT_EQ(check_output(AARCH64_PATH, BIND_EXPECT BIND_FIRST), 0);
+}
+
+/* The entry reports what it bound, under the names it bound them as. */
+TEST_F(bound_interp, entry_reports_bindings)
+{
+ EXPECT_TRUE(entry_shows("test_interp_bind",
+ "bpf-interpreter first " BIND_FIRST));
+ EXPECT_TRUE(entry_shows("test_interp_bind",
+ "bpf-interpreter second " BIND_SECOND));
+}
+
+/* Selecting a name the entry did not bind fails the exec. */
+TEST_F(bound_interp, unbound_name_fails)
+{
+ ASSERT_EQ(create_fake_elf(BIND_ARM_PATH, EM_ARM), 0);
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+
+ EXPECT_EQ(exec_errno(BIND_ARM_PATH), ENOENT);
+}
+
+/* Activating seals it: what can be matched cannot be changed. */
+TEST_F(bound_interp, sealed_once_active)
+{
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+
+ EXPECT_EQ(entry_bind("test_interp_bind", "third", BIND_SECOND), -EBUSY);
+ EXPECT_FALSE(entry_shows("test_interp_bind",
+ "bpf-interpreter third " BIND_SECOND));
+}
+
+/* The seal is for good: disabling the entry again reopens nothing. */
+TEST_F(bound_interp, disable_does_not_unseal)
+{
+ ASSERT_EQ(activate("test_interp_bind"), 0);
+ ASSERT_EQ(entry_command("test_interp_bind", "0\n"), 0);
+
+ EXPECT_EQ(entry_bind("test_interp_bind", "third", BIND_SECOND), -EBUSY);
+}
+
+/* An entry registered without 'D' is sealed from the start. */
+TEST_F(bound_interp, born_sealed)
+{
+ /* A second entry for the handler the fixture already published. */
+ ASSERT_EQ(register_entry("test_born_sealed", "interp_bind", NULL), 0);
+
+ EXPECT_EQ(entry_bind("test_born_sealed", "first", BIND_FIRST), -EBUSY);
+ unregister("test_born_sealed");
+}
+
+/* A name is bound once; a second use of it is refused. */
+TEST_F(bound_interp, duplicate_name_refused)
+{
+ EXPECT_EQ(entry_bind("test_interp_bind", "first", BIND_SECOND), -EEXIST);
+}
+
+/* A name is a printable word: the entry file reports 'name path' lines. */
+TEST_F(bound_interp, name_must_be_printable)
+{
+ /* A control character would forge a line into the entry file. */
+ EXPECT_EQ(entry_bind("test_interp_bind", "a\tb", BIND_FIRST), -EINVAL);
+ EXPECT_EQ(entry_bind("test_interp_bind", "a\nb", BIND_FIRST), -EINVAL);
+
+ /* A space cannot even be spelled: the path starts after the first one. */
+ EXPECT_EQ(entry_bind("test_interp_bind", "a b", BIND_FIRST), -EINVAL);
+}
+
+/* The command ends at the write: bytes past an embedded nul are refused. */
+TEST_F(bound_interp, trailing_bytes_refused)
+{
+ char cmd[PATH_MAX];
+ size_t len;
+ int fd;
+
+ /* entry_command() cannot spell a nul, so write the buffer raw. */
+ snprintf(cmd, sizeof(cmd), "+nul %s", BIND_FIRST);
+ len = strlen(cmd) + 1;
+ memcpy(cmd + len, "junk", sizeof("junk"));
+ len += sizeof("junk");
+
+ fd = open(BINFMT_DIR "/test_interp_bind", O_WRONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(write(fd, cmd, len), -1);
+ EXPECT_EQ(errno, EINVAL);
+ close(fd);
+
+ EXPECT_FALSE(entry_shows("test_interp_bind",
+ "bpf-interpreter nul " BIND_FIRST));
+}
+
+/* An entry binds at most BIND_MAX interpreters. */
+TEST_F(bound_interp, capped_bindings)
+{
+ char name[16];
+ int i;
+
+ /* The fixture bound "first" and "second" already. */
+ for (i = 2; i < BIND_MAX; i++) {
+ snprintf(name, sizeof(name), "n%d", i);
+ ASSERT_EQ(entry_bind("test_interp_bind", name, BIND_FIRST), 0);
+ }
+ EXPECT_EQ(entry_bind("test_interp_bind", "over", BIND_FIRST), -ENOSPC);
+}
+
+/* A binding pins a file: it is charged, and refused once the budget is out. */
+TEST_F(bound_interp, bindings_are_charged)
+{
+ int err = bind_out_of_budget("test_interp_bind", "third", BIND_FIRST);
+
+ if (err < 0)
+ SKIP(return, "no user namespaces or no " INTERP_LIMIT);
+
+ /* The charge follows the writer, not the entry file it writes to. */
+ EXPECT_EQ(err, ENOSPC);
+
+ /* The budget was the only thing in the way. */
+ EXPECT_EQ(entry_bind("test_interp_bind", "third", BIND_FIRST), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_common.h b/tools/testing/selftests/exec/binfmt_misc_common.h
new file mode 100644
index 000000000000..745aff84dc78
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_common.h
@@ -0,0 +1,315 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/* Helpers shared by the binfmt_misc selftests. */
+#ifndef __SELFTESTS_EXEC_BINFMT_MISC_COMMON_H
+#define __SELFTESTS_EXEC_BINFMT_MISC_COMMON_H
+
+#include <elf.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <libgen.h>
+#include <limits.h>
+#include <link.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/types.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#define BINFMT_DIR "/proc/sys/fs/binfmt_misc"
+#define BINFMT_REG BINFMT_DIR "/register"
+
+/* comm holds 15 usable chars; a read of /proc/self/comm appends a newline. */
+#define TASK_COMM_LEN 16
+
+/* The canonical payload argv: run_payload() passes it, the payloads assert it. */
+#define PAYLOAD_ARGV0 "payload-argv0"
+#define PAYLOAD_ARG1 "argone"
+#define PAYLOAD_ARG2 "argtwo"
+
+/* Marker the loader tests poke into the payload's e_ident padding. */
+#define LOADER_MARKER "LDRTST"
+
+/* Exit status run_payload() reports when the exec was refused as unhandled. */
+#define RUN_ENOEXEC 42
+
+static inline int copy_file(const char *src, const char *dst)
+{
+ char buf[4096];
+ int in, out;
+ ssize_t n;
+
+ in = open(src, O_RDONLY);
+ if (in < 0)
+ return -1;
+ /* The tests share /tmp, so never write through a name they don't own. */
+ unlink(dst);
+ out = open(dst, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (out < 0) {
+ close(in);
+ return -1;
+ }
+ while ((n = read(in, buf, sizeof(buf))) > 0) {
+ if (write(out, buf, n) != n) {
+ close(in);
+ close(out);
+ return -1;
+ }
+ }
+ close(in);
+ close(out);
+ return n < 0 ? -1 : 0;
+}
+
+/* Write @rule to the register file, preserving the write's errno. */
+static inline int write_reg(const char *rule)
+{
+ int fd, saved;
+ ssize_t n;
+
+ fd = open(BINFMT_REG, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ n = write(fd, rule, strlen(rule));
+ saved = errno;
+ close(fd);
+ errno = saved;
+ return n < 0 ? -1 : 0;
+}
+
+static inline void unregister(const char *name)
+{
+ char path[PATH_MAX];
+ int fd;
+
+ snprintf(path, sizeof(path), BINFMT_DIR "/%s", name);
+ fd = open(path, O_WRONLY);
+ if (fd >= 0) {
+ if (write(fd, "-1", 2) < 0)
+ ; /* best effort */
+ close(fd);
+ }
+}
+
+/* Write @line to @entry's file, reporting the errno it was refused with. */
+static inline int entry_command(const char *entry, const char *line)
+{
+ char path[PATH_MAX];
+ int fd, retval = 0;
+ size_t len = strlen(line);
+
+ snprintf(path, sizeof(path), BINFMT_DIR "/%s", entry);
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -errno;
+ if (write(fd, line, len) != (ssize_t)len)
+ retval = -errno;
+ close(fd);
+ return retval;
+}
+
+/* Does @entry's file report @line? */
+static inline bool entry_shows(const char *entry, const char *line)
+{
+ char path[PATH_MAX], buf[PATH_MAX];
+ bool found = false;
+ FILE *fp;
+
+ snprintf(path, sizeof(path), BINFMT_DIR "/%s", entry);
+ fp = fopen(path, "r");
+ if (!fp)
+ return false;
+ while (fgets(buf, sizeof(buf), fp)) {
+ buf[strcspn(buf, "\n")] = '\0';
+ if (!strcmp(buf, line)) {
+ found = true;
+ break;
+ }
+ }
+ fclose(fp);
+ return found;
+}
+
+/* Mount binfmt_misc unless it already is, and report whether it is usable. */
+static inline bool binfmt_misc_available(void)
+{
+ if (access(BINFMT_REG, F_OK) < 0)
+ mount("binfmt_misc", BINFMT_DIR, "binfmt_misc", 0, NULL);
+ return access(BINFMT_REG, F_OK) == 0;
+}
+
+/* Absolute path of @name in the directory this test was built into. */
+static inline int artifact_path(char *out, size_t sz, const char *name)
+{
+ char exe[PATH_MAX];
+ ssize_t n;
+
+ n = readlink("/proc/self/exe", exe, sizeof(exe) - 1);
+ if (n < 0)
+ return -1;
+ exe[n] = '\0';
+ if ((size_t)snprintf(out, sz, "%s/%s", dirname(exe), name) >= sz)
+ return -1;
+ return 0;
+}
+
+/* Probe kernel support for a registration flag with a throwaway entry. */
+static inline bool binfmt_flag_supported(char flag)
+{
+ char rule[64];
+
+ snprintf(rule, sizeof(rule), ":bm_flag_probe:E::bmprobe::/bin/true:%c",
+ flag);
+ if (write_reg(rule))
+ return false;
+ unregister("bm_flag_probe");
+ return true;
+}
+
+/*
+ * Run @path with the canonical payload argv and return its exit status, or
+ * RUN_ENOEXEC when the exec itself was refused as unhandled.
+ */
+static inline int run_payload(const char *path)
+{
+ int status;
+ pid_t pid;
+
+ pid = fork();
+ if (pid == 0) {
+ execl(path, PAYLOAD_ARGV0, PAYLOAD_ARG1, PAYLOAD_ARG2,
+ (char *)NULL);
+ _exit(errno == ENOEXEC ? RUN_ENOEXEC : 126);
+ }
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/* Does the exe link name @path? */
+static inline bool exe_is(const char *path)
+{
+ char exe[PATH_MAX], real[PATH_MAX];
+ ssize_t n;
+
+ n = readlink("/proc/self/exe", exe, sizeof(exe) - 1);
+ if (n <= 0 || !realpath(path, real))
+ return false;
+ exe[n] = '\0';
+ return !strcmp(exe, real);
+}
+
+/* Is comm @name truncated to what a comm can hold? */
+static inline bool comm_is(const char *name)
+{
+ char comm[TASK_COMM_LEN + 2], expect[TASK_COMM_LEN];
+ ssize_t n;
+ int fd;
+
+ fd = open("/proc/self/comm", O_RDONLY);
+ if (fd < 0)
+ return false;
+ n = read(fd, comm, sizeof(comm) - 1);
+ close(fd);
+ if (n <= 0)
+ return false;
+ if (comm[n - 1] == '\n')
+ n--;
+ comm[n] = '\0';
+ snprintf(expect, sizeof(expect), "%s", name);
+ return !strcmp(comm, expect);
+}
+
+/* Opening @path for writing has to fail with ETXTBSY. */
+static inline bool write_denied(const char *path)
+{
+ int fd = open(path, O_WRONLY);
+
+ if (fd >= 0) {
+ close(fd);
+ return false;
+ }
+ return errno == ETXTBSY;
+}
+
+static inline int patch_file(const char *path, off_t off, const void *data, size_t len)
+{
+ ssize_t n;
+ int fd;
+
+ fd = open(path, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ n = pwrite(fd, data, len, off);
+ close(fd);
+ return n == (ssize_t)len ? 0 : -1;
+}
+
+/* start_code and end_code are the 26th and 27th fields of /proc/pid/stat. */
+static inline int stat_codes(pid_t pid, unsigned long *start_code,
+ unsigned long *end_code)
+{
+ char buf[4096], path[64], *p;
+ ssize_t n;
+ int fd, i;
+
+ snprintf(path, sizeof(path), "/proc/%d/stat", pid);
+ fd = open(path, O_RDONLY);
+ if (fd < 0)
+ return -1;
+ n = read(fd, buf, sizeof(buf) - 1);
+ close(fd);
+ if (n <= 0)
+ return -1;
+ buf[n] = '\0';
+
+ /* Skip "pid (comm)", then start_code is the 24th field after it. */
+ p = strrchr(buf, ')');
+ if (!p)
+ return -1;
+ p++;
+ for (i = 0; i < 23; i++) {
+ p = strchr(p + 1, ' ');
+ if (!p)
+ return -1;
+ }
+ if (sscanf(p, " %lu %lu", start_code, end_code) != 2)
+ return -1;
+ return 0;
+}
+
+/* Find the system loader through our own PT_INTERP. */
+static inline int find_loader(char *out, size_t sz)
+{
+ ElfW(Ehdr) eh;
+ ElfW(Phdr) ph;
+ int fd, i, ret = -1;
+
+ fd = open("/proc/self/exe", O_RDONLY);
+ if (fd < 0)
+ return -1;
+ if (pread(fd, &eh, sizeof(eh), 0) != sizeof(eh))
+ goto out;
+ for (i = 0; i < eh.e_phnum; i++) {
+ if (pread(fd, &ph, sizeof(ph),
+ eh.e_phoff + i * eh.e_phentsize) != sizeof(ph))
+ goto out;
+ if (ph.p_type != PT_INTERP)
+ continue;
+ if (!ph.p_filesz || ph.p_filesz > sz)
+ goto out;
+ if (pread(fd, out, ph.p_filesz, ph.p_offset) !=
+ (ssize_t)ph.p_filesz)
+ goto out;
+ out[ph.p_filesz - 1] = '\0';
+ ret = 0;
+ break;
+ }
+out:
+ close(fd);
+ return ret;
+}
+
+#endif /* __SELFTESTS_EXEC_BINFMT_MISC_COMMON_H */
diff --git a/tools/testing/selftests/exec/binfmt_misc_disabled.c b/tools/testing/selftests/exec/binfmt_misc_disabled.c
new file mode 100644
index 000000000000..47c9e8a4ee42
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_disabled.c
@@ -0,0 +1,172 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test the 'D' (register disabled) flag of binfmt_misc. An entry
+ * registered with it exists but cannot be matched until userspace enables
+ * it, which splits a registration into create and activate.
+ *
+ * Needs root for the registration; no bpf toolchain involved.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define MAGIC "#DISABLED-SELFTEST#"
+#define TARGET_PATH "/tmp/binfmt_disabled_target"
+#define INTERP_PATH "/tmp/binfmt_disabled_interp.sh"
+#define ENTRY "test_disabled"
+#define RULE(flags) ":" ENTRY ":M:0:" MAGIC "::" INTERP_PATH ":" flags
+
+/* The interpreter exits with a code the harness can recognise. */
+#define EXIT_INTERP 7
+
+/* The target only has to carry the magic; it is never actually loaded. */
+static int create_target(void)
+{
+ char buf[128] = MAGIC "\n";
+ int fd;
+
+ unlink(TARGET_PATH);
+ fd = open(TARGET_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, buf, sizeof(buf)) != (ssize_t)sizeof(buf)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+static int create_interp(void)
+{
+ char buf[64];
+ int fd;
+
+ unlink(INTERP_PATH);
+ fd = open(INTERP_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ snprintf(buf, sizeof(buf), "#!/bin/sh\nexit %d\n", EXIT_INTERP);
+ if (write(fd, buf, strlen(buf)) != (ssize_t)strlen(buf)) {
+ close(fd);
+ return -1;
+ }
+ return close(fd);
+}
+
+FIXTURE(disabled) {
+};
+
+FIXTURE_SETUP(disabled)
+{
+ if (getuid() != 0)
+ SKIP(return, "test must be run as root");
+ if (!binfmt_misc_available())
+ SKIP(return, "no binfmt_misc");
+
+ /* Skip the whole suite on a kernel that does not know 'D'. */
+ if (!binfmt_flag_supported('D')) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'D' flag");
+ }
+
+ ASSERT_EQ(create_interp(), 0);
+ ASSERT_EQ(create_target(), 0);
+}
+
+FIXTURE_TEARDOWN(disabled)
+{
+ unregister(ENTRY);
+ unlink(TARGET_PATH);
+ unlink(INTERP_PATH);
+}
+
+/* The entry exists but does not dispatch until it is enabled. */
+TEST_F(disabled, inert_until_enabled)
+{
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "disabled"));
+
+ /* Nothing matches it, so no binary format claims the target. */
+ EXPECT_EQ(run_payload(TARGET_PATH), RUN_ENOEXEC);
+
+ ASSERT_EQ(entry_command(ENTRY, "1\n"), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "enabled"));
+ EXPECT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+}
+
+/* Without 'D' an entry is matchable the moment it is registered. */
+TEST_F(disabled, enabled_without_the_flag)
+{
+ ASSERT_EQ(write_reg(RULE("")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "enabled"));
+ EXPECT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+}
+
+/* 'D' is spent on the registration: the entry does not report it back. */
+TEST_F(disabled, flag_not_reported)
+{
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ EXPECT_FALSE(entry_shows(ENTRY, "flags: D"));
+ EXPECT_TRUE(entry_shows(ENTRY, "flags: "));
+}
+
+/* A disabled entry can be disabled and enabled like any other. */
+TEST_F(disabled, toggles_like_any_entry)
+{
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+
+ ASSERT_EQ(entry_command(ENTRY, "1\n"), 0);
+ ASSERT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+ ASSERT_EQ(entry_command(ENTRY, "0\n"), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), RUN_ENOEXEC);
+ ASSERT_EQ(entry_command(ENTRY, "1\n"), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), EXIT_INTERP);
+}
+
+/* 'D' composes with the invocation flags a static entry can carry. */
+TEST_F(disabled, composes_with_invocation_flags)
+{
+ ASSERT_EQ(write_reg(RULE("PD")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "disabled"));
+ EXPECT_TRUE(entry_shows(ENTRY, "flags: P"));
+}
+
+/* '-1' to the status file sweeps a staged entry with everything else. */
+TEST_F(disabled, removed_by_remove_all)
+{
+ int fd;
+
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ EXPECT_TRUE(entry_shows(ENTRY, "disabled"));
+
+ fd = open(BINFMT_DIR "/status", O_WRONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(write(fd, "-1", 2), 2);
+ close(fd);
+
+ EXPECT_NE(access(BINFMT_DIR "/" ENTRY, F_OK), 0);
+}
+
+/* A file handle held across a removal cannot resurrect the entry. */
+TEST_F(disabled, no_resurrection_after_remove)
+{
+ int fd;
+
+ ASSERT_EQ(write_reg(RULE("D")), 0);
+ fd = open(BINFMT_DIR "/" ENTRY, O_WRONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+
+ ASSERT_EQ(write(fd, "-1", 2), 2);
+ EXPECT_NE(access(BINFMT_DIR "/" ENTRY, F_OK), 0);
+
+ /* Accepted like any toggle of a removed entry, but publishes nothing. */
+ EXPECT_EQ(write(fd, "1", 1), 1);
+ EXPECT_EQ(run_payload(TARGET_PATH), RUN_ENOEXEC);
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_interplimit.c b/tools/testing/selftests/exec/binfmt_misc_interplimit.c
new file mode 100644
index 000000000000..bf611c551784
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_interplimit.c
@@ -0,0 +1,232 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A pre-opened interpreter - what 'F' gives a static entry and what a 'B'
+ * entry binds - keeps a file open for as long as the entry lives, so it pins
+ * the mount it came from. It costs no file descriptor, and binfmt_misc is
+ * FS_USERNS_MOUNT, so an unprivileged user namespace can create them without
+ * bound. Check that UCOUNT_BINFMT_MISC_INTERPRETERS bounds it, that an entry
+ * that pre-opens nothing is not charged, that removing an entry gives the
+ * charge back, and that nesting a user namespace does not evade it.
+ *
+ * Runs unprivileged in a user namespace.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "../filesystems/utils.h"
+#include "kselftest_harness.h"
+
+#define MNT "/tmp/binfmt_interplimit"
+#define NESTED_MNT "/tmp/binfmt_interplimit_nested"
+#define LIMIT_SYSCTL "/proc/sys/user/max_binfmt_misc_interpreters"
+
+#define MAGIC "\\xde\\xad"
+/* Not on the instance, and unlike /bin/true it always exists. */
+#define INTERP "/proc/self/exe"
+
+/* Small enough to fill by hand, big enough that a refund is visible. */
+#define LIMIT 4
+
+/* What UCOUNT_ENTRY() lets a namespace raise its own limit to. */
+#define LIMIT_MAX "2147483647"
+
+static int ensure_dir(const char *path)
+{
+ if (mkdir(path, 0755) && errno != EEXIST)
+ return -1;
+ return 0;
+}
+
+/* Write @val to @path, preserving write(2)'s errno for the caller. */
+static int write_keep_errno(const char *path, const char *val)
+{
+ int fd, saved;
+ ssize_t n;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd < 0)
+ return -1;
+ n = write(fd, val, strlen(val));
+ saved = errno;
+ close(fd);
+ errno = saved;
+ return n < 0 ? -1 : 0;
+}
+
+static int set_limit(const char *val)
+{
+ return write_keep_errno(LIMIT_SYSCTL, val);
+}
+
+static int register_at(const char *mnt, const char *rule)
+{
+ char path[PATH_MAX];
+
+ snprintf(path, sizeof(path), "%s/register", mnt);
+ return write_keep_errno(path, rule);
+}
+
+/* An 'F' entry: one interpreter pre-opened at registration, one charge. */
+static int register_fixed(const char *mnt, const char *name)
+{
+ char rule[PATH_MAX];
+
+ snprintf(rule, sizeof(rule), ":%s:M::" MAGIC "::" INTERP ":F", name);
+ return register_at(mnt, rule);
+}
+
+/* The same entry without 'F': the interpreter is opened per exec instead. */
+static int register_plain(const char *mnt, const char *name)
+{
+ char rule[PATH_MAX];
+
+ snprintf(rule, sizeof(rule), ":%s:M::" MAGIC "::" INTERP ":", name);
+ return register_at(mnt, rule);
+}
+
+static int remove_entry(const char *mnt, const char *name)
+{
+ char path[PATH_MAX];
+
+ snprintf(path, sizeof(path), "%s/%s", mnt, name);
+ return write_keep_errno(path, "-1\n");
+}
+
+static bool entry_exists(const char *mnt, const char *name)
+{
+ char path[PATH_MAX];
+
+ snprintf(path, sizeof(path), "%s/%s", mnt, name);
+ return access(path, F_OK) == 0;
+}
+
+/* Register @n 'F' entries, each with a name of its own. */
+static int fill_budget(const char *mnt, unsigned int n)
+{
+ char name[32];
+ unsigned int i;
+
+ for (i = 0; i < n; i++) {
+ snprintf(name, sizeof(name), "fixed%u", i);
+ if (register_fixed(mnt, name))
+ return -1;
+ }
+ return 0;
+}
+
+FIXTURE(interp_limit) {
+};
+
+FIXTURE_SETUP(interp_limit)
+{
+ /* setup_userns() exits rather than returns if this is not there. */
+ if (access("/proc/self/ns/user", F_OK))
+ SKIP(return, "kernel without user namespaces");
+ ASSERT_EQ(setup_userns(), 0);
+
+ /* CAP_SYS_RESOURCE in this namespace is what makes it writable. */
+ if (set_limit(LIMIT_MAX)) {
+ if (errno == ENOENT)
+ SKIP(return, "kernel without " LIMIT_SYSCTL);
+ SKIP(return, "cannot set the limit: %s", strerror(errno));
+ }
+
+ ASSERT_EQ(ensure_dir(MNT), 0);
+ if (mount("binfmt_misc", MNT, "binfmt_misc", 0, NULL)) {
+ int saved = errno;
+
+ /* Teardown doesn't run when setup skips, so clean up here. */
+ rmdir(MNT);
+ SKIP(return, "no binfmt_misc: %s", strerror(saved));
+ }
+}
+
+FIXTURE_TEARDOWN(interp_limit)
+{
+ /* The namespaces go with the process; just don't litter /tmp. */
+ umount2(NESTED_MNT, MNT_DETACH);
+ umount2(MNT, MNT_DETACH);
+ rmdir(NESTED_MNT);
+ rmdir(MNT);
+}
+
+/* Every pre-opened interpreter is charged, and the budget is a hard stop. */
+TEST_F(interp_limit, fixed_interpreters_are_charged)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%u", LIMIT);
+ ASSERT_EQ(set_limit(buf), 0);
+
+ ASSERT_EQ(fill_budget(MNT, LIMIT), 0);
+
+ EXPECT_NE(register_fixed(MNT, "over"), 0);
+ EXPECT_EQ(errno, ENOSPC);
+
+ /* A refused registration leaves nothing behind. */
+ EXPECT_FALSE(entry_exists(MNT, "over"));
+}
+
+/* An entry that pre-opens nothing pins nothing, so it is not charged. */
+TEST_F(interp_limit, plain_entries_are_not_charged)
+{
+ ASSERT_EQ(set_limit("0"), 0);
+
+ EXPECT_EQ(register_plain(MNT, "plain"), 0);
+ EXPECT_TRUE(entry_exists(MNT, "plain"));
+
+ /* ... while the same entry with 'F' has nothing to spend. */
+ EXPECT_NE(register_fixed(MNT, "fixed"), 0);
+ EXPECT_EQ(errno, ENOSPC);
+}
+
+/* Removing an entry closes its interpreters and gives the charge back. */
+TEST_F(interp_limit, removal_refunds_the_charge)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%u", LIMIT);
+ ASSERT_EQ(set_limit(buf), 0);
+
+ ASSERT_EQ(fill_budget(MNT, LIMIT), 0);
+ ASSERT_NE(register_fixed(MNT, "over"), 0);
+
+ ASSERT_EQ(remove_entry(MNT, "fixed0"), 0);
+ EXPECT_EQ(register_fixed(MNT, "over"), 0);
+}
+
+/*
+ * The charge walks the ancestors, so a namespace cannot buy itself budget by
+ * nesting: it may raise only its own limit, and the parent it was created
+ * from is charged for every binding made below it.
+ */
+TEST_F(interp_limit, nesting_does_not_evade_it)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%u", LIMIT);
+ ASSERT_EQ(set_limit(buf), 0);
+ ASSERT_EQ(fill_budget(MNT, LIMIT), 0);
+
+ ASSERT_EQ(setup_userns(), 0);
+ ASSERT_EQ(set_limit(LIMIT_MAX), 0);
+
+ ASSERT_EQ(ensure_dir(NESTED_MNT), 0);
+ ASSERT_EQ(mount("binfmt_misc", NESTED_MNT, "binfmt_misc", 0, NULL), 0);
+
+ /* A fresh instance with an unlimited budget of its own, and yet: */
+ EXPECT_NE(register_fixed(NESTED_MNT, "nested"), 0);
+ EXPECT_EQ(errno, ENOSPC);
+
+ /* The nested instance works for anything that pins no file. */
+ EXPECT_EQ(register_plain(NESTED_MNT, "nested_plain"), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_loader.c b/tools/testing/selftests/exec/binfmt_misc_loader.c
new file mode 100644
index 000000000000..1e14dcd274af
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_loader.c
@@ -0,0 +1,372 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test the 'L' (loader substitution) flag of binfmt_misc. A matched
+ * binary runs as the MAIN image - a fully native exec - with the
+ * registered interpreter substituted for its PT_INTERP. The payload
+ * (binfmt_loader_payload) asserts the native identity from inside.
+ *
+ * The substitute is a copy of the system loader found via our own
+ * PT_INTERP; magic matching pokes a marker into the ELF header's
+ * e_ident padding, which kernel and loader ignore.
+ *
+ * Needs root for the registration; no bpf toolchain involved.
+ */
+#define _GNU_SOURCE
+#include <elf.h>
+#include <link.h>
+#include <signal.h>
+#include <stddef.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <sys/mman.h>
+#include <sys/ptrace.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define ENTRY "test_loader"
+#define INTERP_PATH "/tmp/binfmt_loader_interp"
+#define MOVED_PATH INTERP_PATH ".moved"
+#define TARGET_PATH "/tmp/binfmt_loader_target.ldrtest"
+#define STATIC_PATH "/tmp/binfmt_loader_static.ldrtest"
+#define FOREIGN_PATH "/tmp/binfmt_loader_foreign.ldrtest"
+#define SCRIPT_PATH "/tmp/binfmt_loader_script.ldrtest"
+#define M_RULE ":" ENTRY ":M:9:" LOADER_MARKER "::" INTERP_PATH ":L"
+#define E_RULE ":" ENTRY ":E::ldrtest::" INTERP_PATH ":L"
+#define FL_RULE ":" ENTRY ":E::ldrtest::" INTERP_PATH ":FL"
+
+/* Execute the binary from an inaccessible O_CLOEXEC memfd. */
+static int run_memfd(const char *path)
+{
+ int status;
+ pid_t pid;
+
+ pid = fork();
+ if (pid == 0) {
+ char *argv[] = { PAYLOAD_ARGV0, PAYLOAD_ARG1, PAYLOAD_ARG2, NULL };
+ char buf[4096];
+ int in, mfd;
+ ssize_t n;
+
+ mfd = memfd_create("loader-test", MFD_CLOEXEC);
+ in = open(path, O_RDONLY);
+ if (mfd < 0 || in < 0)
+ _exit(125);
+ while ((n = read(in, buf, sizeof(buf))) > 0)
+ if (write(mfd, buf, n) != n)
+ _exit(125);
+ close(in);
+ setenv("BINFMT_TEST_MEMFD", "1", 1);
+ unsetenv("BINFMT_TEST_BINARY");
+ syscall(SYS_execveat, mfd, "", argv, environ, AT_EMPTY_PATH);
+ _exit(126);
+ }
+ if (pid < 0 || waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/*
+ * The differentiator against the transparent mode: at PTRACE_EVENT_EXEC
+ * the identity is already complete - exe, auxv and the stat code markers
+ * are mutually consistent with no window a debugger could observe.
+ */
+static int ptrace_probe(const char *target)
+{
+ unsigned long auxv[2 * 64], base = 0, entry = 0, at_flags = 0;
+ unsigned long start_code = 0, end_code = 0;
+ int status, fd, execfd_seen = 0, failed = 0;
+ char path[64], buf[PATH_MAX];
+ ssize_t n;
+ pid_t pid;
+ int i;
+
+ pid = fork();
+ if (pid == 0) {
+ ptrace(PTRACE_TRACEME, 0, NULL, NULL);
+ raise(SIGSTOP);
+ execl(target, PAYLOAD_ARGV0, PAYLOAD_ARG1, PAYLOAD_ARG2, (char *)NULL);
+ _exit(126);
+ }
+ if (pid < 0)
+ return -1;
+ if (waitpid(pid, &status, 0) != pid || !WIFSTOPPED(status))
+ goto fail_kill;
+ if (ptrace(PTRACE_SETOPTIONS, pid, NULL, (void *)PTRACE_O_TRACEEXEC))
+ goto fail_kill;
+ if (ptrace(PTRACE_CONT, pid, NULL, NULL))
+ goto fail_kill;
+ if (waitpid(pid, &status, 0) != pid || !WIFSTOPPED(status) ||
+ status >> 8 != (SIGTRAP | (PTRACE_EVENT_EXEC << 8))) {
+ fprintf(stderr, "no exec stop (status %#x)\n", status);
+ goto fail_kill;
+ }
+
+ snprintf(path, sizeof(path), "/proc/%d/exe", pid);
+ n = readlink(path, buf, sizeof(buf) - 1);
+ if (n <= 0) {
+ failed = 1;
+ } else {
+ buf[n] = '\0';
+ if (strcmp(buf, target)) {
+ fprintf(stderr, "exe at exec stop: %s\n", buf);
+ failed = 1;
+ }
+ }
+
+ snprintf(path, sizeof(path), "/proc/%d/auxv", pid);
+ fd = open(path, O_RDONLY);
+ if (fd < 0) {
+ n = -1;
+ } else {
+ n = read(fd, auxv, sizeof(auxv));
+ close(fd);
+ }
+ if (n <= 0) {
+ failed = 1;
+ n = 0;
+ }
+ for (i = 0; i + 1 < (int)(n / sizeof(unsigned long)); i += 2) {
+ switch (auxv[i]) {
+ case AT_BASE:
+ base = auxv[i + 1];
+ break;
+ case AT_ENTRY:
+ entry = auxv[i + 1];
+ break;
+ case AT_FLAGS:
+ at_flags = auxv[i + 1];
+ break;
+ case AT_EXECFD:
+ execfd_seen = 1;
+ break;
+ }
+ }
+
+ if (stat_codes(pid, &start_code, &end_code))
+ failed = 1;
+
+ if (!base || execfd_seen || at_flags) {
+ fprintf(stderr, "auxv at exec stop not native\n");
+ failed = 1;
+ }
+ if (!start_code || entry < start_code || entry >= end_code) {
+ fprintf(stderr, "auxv/stat inconsistent at exec stop\n");
+ failed = 1;
+ }
+
+ if (ptrace(PTRACE_CONT, pid, NULL, NULL))
+ goto fail_kill;
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status) ||
+ WEXITSTATUS(status))
+ failed = 1;
+ return failed ? -1 : 0;
+
+fail_kill:
+ kill(pid, SIGKILL);
+ waitpid(pid, &status, 0);
+ return -1;
+}
+
+FIXTURE(loader) {
+ bool have_static;
+};
+
+FIXTURE_SETUP(loader)
+{
+ unsigned short foreign_machine = 0xdead;
+ char src[PATH_MAX], loader[PATH_MAX];
+
+ if (getuid() != 0)
+ SKIP(return, "test must be run as root");
+ if (!binfmt_misc_available())
+ SKIP(return, "no binfmt_misc");
+ if (find_loader(loader, sizeof(loader)))
+ SKIP(return, "cannot determine own PT_INTERP");
+
+ ASSERT_EQ(copy_file(loader, INTERP_PATH), 0);
+
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_loader_payload"), 0);
+ ASSERT_EQ(copy_file(src, TARGET_PATH), 0);
+ ASSERT_EQ(patch_file(TARGET_PATH, EI_PAD, LOADER_MARKER,
+ strlen(LOADER_MARKER)), 0);
+
+ /* The same payload with a machine type this kernel cannot load. */
+ ASSERT_EQ(copy_file(src, FOREIGN_PATH), 0);
+ ASSERT_EQ(patch_file(FOREIGN_PATH, EI_PAD, LOADER_MARKER,
+ strlen(LOADER_MARKER)), 0);
+ ASSERT_EQ(patch_file(FOREIGN_PATH, offsetof(ElfW(Ehdr), e_machine),
+ &foreign_machine, sizeof(foreign_machine)), 0);
+
+ self->have_static =
+ artifact_path(src, sizeof(src), "binfmt_loader_payload_static") == 0 &&
+ copy_file(src, STATIC_PATH) == 0;
+
+ setenv("BINFMT_TEST_BINARY", TARGET_PATH, 1);
+ setenv("BINFMT_TEST_INTERP", INTERP_PATH, 1);
+
+ /* Everything below needs the flag; find out once. */
+ if (write_reg(E_RULE)) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'L' flag");
+ }
+ unregister(ENTRY);
+}
+
+FIXTURE_TEARDOWN(loader)
+{
+ unregister(ENTRY);
+ if (access(MOVED_PATH, F_OK) == 0)
+ rename(MOVED_PATH, INTERP_PATH);
+ unlink(TARGET_PATH);
+ unlink(STATIC_PATH);
+ unlink(FOREIGN_PATH);
+ unlink(SCRIPT_PATH);
+ unlink(INTERP_PATH);
+}
+
+/* Grammar sanity check: the same entry without 'L' has to register. */
+TEST_F(loader, plain_entry_registers)
+{
+ ASSERT_EQ(write_reg(":" ENTRY ":E::ldrtest::" INTERP_PATH ":"), 0);
+}
+
+/* 'L' is a native exec: every classic-dispatch flag is rejected. */
+TEST_F(loader, rejects_classic_flags)
+{
+ static const char * const combos[] = { "LT", "LP", "LC", "LO" };
+ char rule[PATH_MAX];
+ unsigned int i;
+
+ for (i = 0; i < ARRAY_SIZE(combos); i++) {
+ int rc;
+
+ snprintf(rule, sizeof(rule),
+ ":" ENTRY ":E::ldrtest::" INTERP_PATH ":%s", combos[i]);
+ rc = write_reg(rule);
+ EXPECT_EQ(rc, -1)
+ TH_LOG("'%s' was not rejected", combos[i]);
+ if (rc == 0) {
+ unregister(ENTRY);
+ continue;
+ }
+ EXPECT_EQ(errno, EINVAL);
+ }
+}
+
+/*
+ * Without 'F' the interpreter is opened when the binary is executed, so a
+ * relative path would be resolved against the caller's working directory.
+ */
+TEST_F(loader, rejects_relative_interpreter)
+{
+ static const char * const flags[] = { "L", "C" };
+ char rule[PATH_MAX];
+ unsigned int i;
+
+ for (i = 0; i < ARRAY_SIZE(flags); i++) {
+ int rc;
+
+ snprintf(rule, sizeof(rule),
+ ":" ENTRY ":E::ldrtest::binfmt_loader_interp:%s",
+ flags[i]);
+ rc = write_reg(rule);
+ EXPECT_EQ(rc, -1)
+ TH_LOG("'%s' accepted a relative interpreter", flags[i]);
+ if (rc == 0) {
+ unregister(ENTRY);
+ continue;
+ }
+ EXPECT_EQ(errno, EINVAL);
+ }
+}
+
+TEST_F(loader, extension_matched)
+{
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+TEST_F(loader, magic_matched)
+{
+ ASSERT_EQ(write_reg(M_RULE), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+/*
+ * The differentiator against the transparent mode: at PTRACE_EVENT_EXEC the
+ * identity is already complete, with no window a debugger could observe.
+ */
+TEST_F(loader, exec_stop_consistency)
+{
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ EXPECT_EQ(ptrace_probe(TARGET_PATH), 0);
+}
+
+/* A binary without PT_INTERP drops the override and runs natively. */
+TEST_F(loader, static_binary_runs_natively)
+{
+ if (!self->have_static)
+ SKIP(return, "no static payload built");
+
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ setenv("BINFMT_TEST_BINARY", STATIC_PATH, 1);
+ setenv("BINFMT_TEST_STATIC", "1", 1);
+ EXPECT_EQ(run_payload(STATIC_PATH), 0);
+ unsetenv("BINFMT_TEST_STATIC");
+ setenv("BINFMT_TEST_BINARY", TARGET_PATH, 1);
+}
+
+/*
+ * A '#!' file that matched an 'L' entry is claimed by binfmt_script, which
+ * sits ahead of binfmt_elf. The substitute the entry staged has to be
+ * released when the interpreter replaces the file, not leaked.
+ */
+TEST_F(loader, script_claims_the_file)
+{
+ static const char script[] = "#!/bin/sh\nexit 0\n";
+ int fd;
+
+ unlink(SCRIPT_PATH);
+ fd = open(SCRIPT_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(write(fd, script, sizeof(script) - 1),
+ (ssize_t)sizeof(script) - 1);
+ ASSERT_EQ(close(fd), 0);
+
+ ASSERT_EQ(write_reg(E_RULE), 0);
+ EXPECT_EQ(run_payload(SCRIPT_PATH), 0);
+
+ /* A leaked substitute keeps its write denial on the loader. */
+ fd = open(INTERP_PATH, O_WRONLY);
+ EXPECT_GE(fd, 0)
+ TH_LOG("loader still write denied (errno %d)", errno);
+ if (fd >= 0)
+ close(fd);
+}
+
+/* Nothing needs the binary's path, so an inaccessible fd works. */
+TEST_F(loader, inaccessible_memfd)
+{
+ ASSERT_EQ(write_reg(M_RULE), 0);
+ EXPECT_EQ(run_memfd(TARGET_PATH), 0);
+}
+
+/* The whole exec of a wrong-arch binary fails as if unhandled. */
+TEST_F(loader, foreign_arch_enoexec)
+{
+ ASSERT_EQ(write_reg(M_RULE), 0);
+ EXPECT_EQ(run_payload(FOREIGN_PATH), RUN_ENOEXEC);
+}
+
+/* 'F' pre-opens the substitute, so it survives losing its path. */
+TEST_F(loader, fixed_interpreter_survives_rename)
+{
+ ASSERT_EQ(write_reg(FL_RULE), 0);
+ ASSERT_EQ(rename(INTERP_PATH, MOVED_PATH), 0);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_selfpin.c b/tools/testing/selftests/exec/binfmt_misc_selfpin.c
new file mode 100644
index 000000000000..5286b0604eed
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_selfpin.c
@@ -0,0 +1,158 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An 'F' entry keeps its interpreter open for as long as the entry exists,
+ * and the entry only goes away when the binfmt_misc superblock is destroyed.
+ * An interpreter that lives on a mount which in turn keeps that superblock
+ * alive therefore pins the instance that owns it, and nothing can break the
+ * cycle. Check the two ways userspace could arrange for that: an interpreter
+ * on the binfmt_misc instance itself, and one on a filesystem stacked on it.
+ *
+ * Runs unprivileged in a user namespace; binfmt_misc is FS_USERNS_MOUNT.
+ */
+#define _GNU_SOURCE
+#include <fcntl.h>
+#include <limits.h>
+#include <sched.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+
+#include "../filesystems/utils.h"
+#include "kselftest_harness.h"
+
+#define MNT "/tmp/binfmt_selfpin"
+#define BACKING "/tmp/binfmt_selfpin_back"
+#define LOWER BACKING "/lower"
+#define MERGED "/tmp/binfmt_selfpin_merged"
+
+#define MAGIC "\\xde\\xad"
+#define RULE(interp) ":selfpin:M::" MAGIC "::" interp ":F"
+/* Not on the instance, and unlike /bin/true it always exists. */
+#define INTERP "/proc/self/exe"
+
+#define OPTS_MAX (3 * PATH_MAX + 64)
+
+static int ensure_dir(const char *path)
+{
+ if (mkdir(path, 0755) && errno != EEXIST)
+ return -1;
+ return 0;
+}
+
+/* Write @rule to this instance's register file, preserving write(2)'s errno. */
+static int register_at(struct __test_metadata *_metadata, const char *rule)
+{
+ int fd, saved;
+ ssize_t n;
+
+ fd = open(MNT "/register", O_WRONLY);
+ ASSERT_GE(fd, 0);
+ n = write(fd, rule, strlen(rule));
+ saved = errno;
+ close(fd);
+ errno = saved;
+ return n < 0 ? -1 : 0;
+}
+
+/*
+ * Mount an overlay over @lower using a private upper/work pair, so the two
+ * mounts this test performs cannot interfere with each other and neither
+ * overlaps the lower layer.
+ */
+static int mount_overlay(const char *lower, int nr)
+{
+ char opts[OPTS_MAX], upper[PATH_MAX], work[PATH_MAX];
+
+ snprintf(upper, sizeof(upper), "%s/upper%d", BACKING, nr);
+ snprintf(work, sizeof(work), "%s/work%d", BACKING, nr);
+ if (mkdir(upper, 0755) || mkdir(work, 0755))
+ return -1;
+
+ snprintf(opts, sizeof(opts), "lowerdir=%s,upperdir=%s,workdir=%s",
+ lower, upper, work);
+ return mount("ovl", MERGED, "overlay", 0, opts);
+}
+
+FIXTURE(selfpin) {
+};
+
+FIXTURE_SETUP(selfpin)
+{
+ /* setup_userns() exits rather than returns if this is not there. */
+ if (access("/proc/self/ns/user", F_OK))
+ SKIP(return, "kernel without user namespaces");
+ ASSERT_EQ(setup_userns(), 0);
+
+ ASSERT_EQ(ensure_dir(MNT), 0);
+ if (mount("binfmt_misc", MNT, "binfmt_misc", 0, NULL)) {
+ int saved = errno;
+
+ /* Teardown doesn't run when setup skips, so clean up here. */
+ rmdir(MNT);
+ SKIP(return, "no binfmt_misc: %s", strerror(saved));
+ }
+}
+
+FIXTURE_TEARDOWN(selfpin)
+{
+ /* The namespaces go with the process; just don't litter /tmp. */
+ umount2(MERGED, MNT_DETACH);
+ umount2(BACKING, MNT_DETACH);
+ umount2(MNT, MNT_DETACH);
+ rmdir(MERGED);
+ rmdir(BACKING);
+ rmdir(MNT);
+}
+
+/*
+ * The instance's own files are regular files the mounter owns, so they can be
+ * made executable. Opening one for exec still has to fail, otherwise the entry
+ * pins the very superblock it lives in.
+ */
+TEST_F(selfpin, interpreter_on_the_instance)
+{
+ ASSERT_EQ(chmod(MNT "/status", 0755), 0);
+
+ ASSERT_NE(register_at(_metadata, RULE(MNT "/status")), 0);
+ EXPECT_EQ(errno, EACCES);
+}
+
+/* Same for an entry file rather than one of the control files. */
+TEST_F(selfpin, interpreter_on_an_entry)
+{
+ ASSERT_EQ(register_at(_metadata, ":victim:M::" MAGIC "::" INTERP ":"), 0);
+ ASSERT_EQ(chmod(MNT "/victim", 0755), 0);
+
+ ASSERT_NE(register_at(_metadata, RULE(MNT "/victim")), 0);
+ EXPECT_EQ(errno, EACCES);
+}
+
+/*
+ * A stacking filesystem holds a private clone of each layer for its whole
+ * lifetime, so an instance used as a layer can be pinned by an interpreter
+ * that does not live on it at all. Refuse to be a layer.
+ */
+TEST_F(selfpin, refuses_to_be_stacked_on)
+{
+ ASSERT_EQ(ensure_dir(BACKING), 0);
+ ASSERT_EQ(mount("tmpfs", BACKING, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mkdir(LOWER, 0755), 0);
+ ASSERT_EQ(ensure_dir(MERGED), 0);
+
+ /* Nothing to prove unless overlayfs works here at all. */
+ if (mount_overlay(LOWER, 1)) {
+ if (errno == ENODEV || errno == EPERM)
+ SKIP(return, "no unprivileged overlayfs");
+ SKIP(return, "overlayfs unusable here: %s", strerror(errno));
+ }
+ ASSERT_EQ(umount(MERGED), 0);
+
+ EXPECT_NE(mount_overlay(MNT, 2), 0);
+}
+
+/* An ordinary interpreter still registers with 'F'. */
+TEST_F(selfpin, ordinary_interpreter_still_works)
+{
+ EXPECT_EQ(register_at(_metadata, RULE(INTERP)), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_misc_transparent.c b/tools/testing/selftests/exec/binfmt_misc_transparent.c
new file mode 100644
index 000000000000..2ebf73de8018
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_transparent.c
@@ -0,0 +1,95 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test the static transparent flag 'T' of binfmt_misc. A magic-matched
+ * binary is dispatched to an interpreter with the argument vector left
+ * untouched, the binary passed through AT_EXECFD and mm->exe_file labeled
+ * with the binary. The asserting interpreter (binfmt_transparent_interp)
+ * verifies the constructed identity from inside the process and exits 0.
+ *
+ * Needs root for the registration; no bpf toolchain involved.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest_harness.h"
+
+#define MAGIC "#TRANSPARENT-SELFTEST#"
+#define TARGET_PATH "/tmp/binfmt_transparent_target"
+#define INTERP_PATH "/tmp/binfmt_transparent_interp"
+#define ENTRY "test_transparent"
+#define RULE(flags) ":" ENTRY ":M:0:" MAGIC "::" INTERP_PATH ":" flags
+
+/* The target only has to carry the magic; it is never actually loaded. */
+static int create_target(void)
+{
+ char buf[128] = MAGIC "\n";
+ int fd;
+
+ unlink(TARGET_PATH);
+ fd = open(TARGET_PATH, O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, buf, sizeof(buf)) != (ssize_t)sizeof(buf)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+FIXTURE(transparent) {
+};
+
+FIXTURE_SETUP(transparent)
+{
+ char src[PATH_MAX];
+
+ if (getuid() != 0)
+ SKIP(return, "test must be run as root");
+ if (!binfmt_misc_available())
+ SKIP(return, "no binfmt_misc");
+
+ ASSERT_EQ(artifact_path(src, sizeof(src), "binfmt_transparent_interp"), 0);
+ ASSERT_EQ(copy_file(src, INTERP_PATH), 0);
+ ASSERT_EQ(create_target(), 0);
+
+ /* Skip the whole suite on a kernel that does not know 'T'. */
+ if (!binfmt_flag_supported('T')) {
+ ASSERT_EQ(errno, EINVAL);
+ SKIP(return, "kernel without the 'T' flag");
+ }
+}
+
+FIXTURE_TEARDOWN(transparent)
+{
+ unregister(ENTRY);
+ unlink(TARGET_PATH);
+ unlink(INTERP_PATH);
+}
+
+/* Grammar sanity check: the same entry without 'T' has to register. */
+TEST_F(transparent, plain_entry_registers)
+{
+ ASSERT_EQ(write_reg(RULE("")), 0);
+}
+
+/* 'T' preserves the whole argv, so combining it with 'P' is rejected. */
+TEST_F(transparent, rejects_preserve_argv0)
+{
+ ASSERT_NE(write_reg(RULE("TP")), 0);
+ EXPECT_EQ(errno, EINVAL);
+}
+
+/* The interpreter asserts the identity the kernel built for it. */
+TEST_F(transparent, dispatch)
+{
+ ASSERT_EQ(write_reg(RULE("T")), 0);
+
+ setenv("BINFMT_TEST_BINARY", TARGET_PATH, 1);
+ setenv("BINFMT_TEST_ARGV0", PAYLOAD_ARGV0, 1);
+ EXPECT_EQ(run_payload(TARGET_PATH), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/exec/binfmt_transparent_interp.c b/tools/testing/selftests/exec/binfmt_transparent_interp.c
new file mode 100644
index 000000000000..d4c4a538c9aa
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_transparent_interp.c
@@ -0,0 +1,112 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Asserting interpreter for the transparent binfmt_misc mode. It runs in
+ * place of the dispatched binary and verifies the identity the kernel
+ * constructed: the aux vector contract, the exe link, argv, cmdline, comm
+ * and the write denial on the binary. BINFMT_TEST_BINARY names the binary;
+ * the harness execs it with the arguments "argone argtwo". Prints
+ * TRANSPARENT_OK and exits 0 when every check holds.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/auxv.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "binfmt_misc_common.h"
+#include "kselftest.h"
+
+#ifndef AT_FLAGS_TRANSPARENT_INTERP
+#define AT_FLAGS_TRANSPARENT_INTERP (1 << 1)
+#endif
+
+static int fail;
+
+static void ok(int cond, const char *what)
+{
+ if (!cond) {
+ fprintf(stderr, "TRANSPARENT_FAIL: %s (errno %d)\n", what, errno);
+ fail = 1;
+ }
+}
+
+int main(int argc, char **argv)
+{
+ const char *binary = getenv("BINFMT_TEST_BINARY");
+ const char *argv0 = getenv("BINFMT_TEST_ARGV0");
+ char expect[PATH_MAX + 32], buf[PATH_MAX];
+ unsigned long execfd;
+ struct stat stb, stfd;
+ const char *want[3];
+ const char *base;
+ size_t expect_len, i;
+ int fd, have_stb, have_stfd;
+ ssize_t n;
+
+ if (!binary) {
+ fprintf(stderr, "TRANSPARENT_FAIL: BINFMT_TEST_BINARY unset\n");
+ return 1;
+ }
+ /* Distinct from the binary path, so a classic argv splice is caught. */
+ want[0] = argv0 ? argv0 : binary;
+ want[1] = PAYLOAD_ARG1;
+ want[2] = PAYLOAD_ARG2;
+
+ /* The aux vector announces the transparent contract. */
+ ok(getauxval(AT_FLAGS) & AT_FLAGS_TRANSPARENT_INTERP,
+ "AT_FLAGS lacks AT_FLAGS_TRANSPARENT_INTERP");
+
+ /* AT_EXECFD refers to the very file that was executed. */
+ execfd = getauxval(AT_EXECFD);
+ ok(execfd > 2, "no AT_EXECFD");
+ have_stb = !stat(binary, &stb);
+ ok(have_stb, "cannot stat the binary");
+ have_stfd = !fstat((int)execfd, &stfd);
+ ok(have_stfd, "cannot fstat AT_EXECFD");
+ ok(have_stb && have_stfd && stb.st_dev == stfd.st_dev &&
+ stb.st_ino == stfd.st_ino, "AT_EXECFD is not the binary");
+
+ /* The exe link names the binary, not this interpreter. */
+ ok(exe_is(binary), "/proc/self/exe is not the binary");
+
+ /* argv arrived unspliced. */
+ ok(argc == (int)ARRAY_SIZE(want), "argv was rewritten");
+ for (i = 0; i < ARRAY_SIZE(want) && i < (size_t)argc; i++)
+ ok(!strcmp(argv[i], want[i]), "argv was rewritten");
+
+ /* And so did the kernel's copy of it: the same strings, NUL separated. */
+ for (i = 0, expect_len = 0; i < ARRAY_SIZE(want); i++) {
+ size_t len = strlen(want[i]) + 1;
+
+ if (expect_len + len > sizeof(expect)) {
+ ok(0, "argv does not fit the expectation buffer");
+ break;
+ }
+ memcpy(expect + expect_len, want[i], len);
+ expect_len += len;
+ }
+ fd = open("/proc/self/cmdline", O_RDONLY);
+ n = fd >= 0 ? read(fd, buf, sizeof(buf)) : -1;
+ if (fd >= 0)
+ close(fd);
+ ok(n == (ssize_t)expect_len && !memcmp(buf, expect, expect_len),
+ "/proc/self/cmdline was rewritten");
+
+ /* comm is the binary's basename. */
+ base = strrchr(binary, '/');
+ base = base ? base + 1 : binary;
+ ok(comm_is(base), "comm is not the binary's basename");
+
+ /* The binary is write-denied while it runs, like a direct exec. */
+ ok(write_denied(binary), "binary is writable while running");
+ ok(write_denied("/proc/self/exe"), "exe link is writable while running");
+
+ if (!fail)
+ printf("TRANSPARENT_OK\n");
+ return fail;
+}
diff --git a/tools/testing/selftests/exec/bpf_interp.bpf.c b/tools/testing/selftests/exec/bpf_interp.bpf.c
new file mode 100644
index 000000000000..8df2d2d01e25
--- /dev/null
+++ b/tools/testing/selftests/exec/bpf_interp.bpf.c
@@ -0,0 +1,61 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the selftest's fixed-interpreter case: match a
+ * 64-bit aarch64 ELF header from the prefetched buffer and route it to a fixed
+ * interpreter chosen by the program. This is the portable, self-contained
+ * equivalent of routing a foreign binary to an emulator: it matches
+ * programmatically and computes the interpreter, but points at a test binary
+ * the harness installs rather than a system emulator.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define EM_AARCH64 183
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+
+/*
+ * A magic-style decision needs nothing beyond the prefetched bprm->buf,
+ * even though the match program could read the file.
+ */
+SEC("struct_ops.s/match")
+bool BPF_PROG(bpf_interp_match, struct linux_binprm *bprm)
+{
+ __u16 machine;
+
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* e_machine is a 16-bit little-endian field at offset 18. */
+ machine = (__u8)bprm->buf[18] | ((__u16)(__u8)bprm->buf[19] << 8);
+ return machine == EM_AARCH64;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(bpf_interp_load, struct linux_binprm *bprm)
+{
+ /*
+ * Keep the path on the (writable) stack: bpf_binprm_set_interp() takes
+ * a sized memory arg and the verifier rejects a read-only .rodata
+ * buffer for it. The harness installs the interpreter at this path.
+ */
+ char interp[] = "/tmp/binfmt_bpf_interp";
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops bpf_interp = {
+ .match = (void *)bpf_interp_match,
+ .load = (void *)bpf_interp_load,
+ .name = "bpf_interp",
+};
diff --git a/tools/testing/selftests/exec/config b/tools/testing/selftests/exec/config
index c308079867b3..ea359a929ae8 100644
--- a/tools/testing/selftests/exec/config
+++ b/tools/testing/selftests/exec/config
@@ -1,2 +1,12 @@
CONFIG_BLK_DEV=y
CONFIG_BLK_DEV_LOOP=y
+CONFIG_BINFMT_MISC=y
+CONFIG_BINFMT_MISC_BPF=y
+CONFIG_BPF_JIT=y
+CONFIG_BPF_SYSCALL=y
+CONFIG_DEBUG_INFO=y
+CONFIG_DEBUG_INFO_BTF=y
+CONFIG_DEBUG_INFO_DWARF4=y
+CONFIG_OVERLAY_FS=y
+CONFIG_TMPFS=y
+CONFIG_USER_NS=y
diff --git a/tools/testing/selftests/exec/interp_bind.bpf.c b/tools/testing/selftests/exec/interp_bind.bpf.c
new file mode 100644
index 000000000000..1ce45cca215f
--- /dev/null
+++ b/tools/testing/selftests/exec/interp_bind.bpf.c
@@ -0,0 +1,76 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the selftest's bound-interpreter case: one
+ * handler, one entry, an interpreter per guest architecture - each bound to
+ * a file when the entry was registered rather than to a path resolved at
+ * exec time. The load program names the one it wants; a name the entry did
+ * not bind fails the exec, which the harness checks too.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define E_MACHINE_OFF 18
+#define EM_ARM 40
+#define EM_AARCH64 183
+#define EM_RISCV 243
+
+extern int bpf_binprm_select_interp(struct linux_binprm *bprm,
+ const char *name, size_t name__sz) __ksym;
+
+/* The guest architecture of a 64-bit ELF, or zero if it is not one. */
+static __u16 elf_machine(struct linux_binprm *bprm)
+{
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return 0;
+
+ /* Little-endian 16-bit field, read byte-wise for the verifier. */
+ return (__u8)bprm->buf[E_MACHINE_OFF] |
+ ((__u16)(__u8)bprm->buf[E_MACHINE_OFF + 1] << 8);
+}
+
+SEC("struct_ops.s/match")
+bool BPF_PROG(interp_bind_match, struct linux_binprm *bprm)
+{
+ __u16 machine = elf_machine(bprm);
+
+ return machine == EM_AARCH64 || machine == EM_RISCV ||
+ machine == EM_ARM;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(interp_bind_load, struct linux_binprm *bprm)
+{
+ /*
+ * Names, not paths: each one selects a file the entry pre-opened, so
+ * nothing is resolved here or later, in any namespace. The buffers
+ * are on the stack because the verifier rejects .rodata for a sized
+ * memory argument.
+ */
+ char first[] = "first";
+ char second[] = "second";
+ char unbound[] = "unbound";
+
+ switch (elf_machine(bprm)) {
+ case EM_AARCH64:
+ return bpf_binprm_select_interp(bprm, first, sizeof(first));
+ case EM_RISCV:
+ return bpf_binprm_select_interp(bprm, second, sizeof(second));
+ }
+
+ /* The entry bound nothing under this name: -ENOENT fails the exec. */
+ return bpf_binprm_select_interp(bprm, unbound, sizeof(unbound));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops interp_bind = {
+ .match = (void *)interp_bind_match,
+ .load = (void *)interp_bind_load,
+ .name = "interp_bind",
+};
diff --git a/tools/testing/selftests/exec/loader.bpf.c b/tools/testing/selftests/exec/loader.bpf.c
new file mode 100644
index 000000000000..108e51dd4961
--- /dev/null
+++ b/tools/testing/selftests/exec/loader.bpf.c
@@ -0,0 +1,56 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the loader-substitution case: match the
+ * marker the harness poked into the payload's e_ident padding and ask for
+ * the selected interpreter to be substituted for the binary's PT_INTERP,
+ * so the binary itself runs as a fully native exec.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define EI_PAD 9
+#define ELFCLASS64 2
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+extern int bpf_binprm_set_flags(struct linux_binprm *bprm,
+ enum bpf_binprm_flags flags) __ksym;
+
+SEC("struct_ops.s/match")
+bool BPF_PROG(loader_match, struct linux_binprm *bprm)
+{
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* The harness marks the payload with "LDRTST" at EI_PAD. */
+ return bprm->buf[EI_PAD + 0] == 'L' && bprm->buf[EI_PAD + 1] == 'D' &&
+ bprm->buf[EI_PAD + 2] == 'R' && bprm->buf[EI_PAD + 3] == 'T' &&
+ bprm->buf[EI_PAD + 4] == 'S' && bprm->buf[EI_PAD + 5] == 'T';
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(loader_load, struct linux_binprm *bprm)
+{
+ char interp[] = "/tmp/binfmt_loader_interp";
+ int err;
+
+ err = bpf_binprm_set_flags(bprm, BPF_BINPRM_LOADER);
+ if (err)
+ return err;
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops loader = {
+ .match = (void *)loader_match,
+ .load = (void *)loader_load,
+ .name = "loader",
+};
diff --git a/tools/testing/selftests/exec/nix_origin.bpf.c b/tools/testing/selftests/exec/nix_origin.bpf.c
new file mode 100644
index 000000000000..378e22a4c43b
--- /dev/null
+++ b/tools/testing/selftests/exec/nix_origin.bpf.c
@@ -0,0 +1,224 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * nix_origin.bpf.c - $ORIGIN-relative PT_INTERP resolution
+ *
+ * A binfmt_misc_ops handler that makes relocatable (Nix-style) ELF
+ * binaries work: if PT_INTERP starts with "$ORIGIN/", the loader is
+ * resolved relative to the directory of the binary being executed and
+ * selected via bpf_binprm_set_interp(). The match program reads the
+ * program headers itself, so anything else never commits to this
+ * handler and passes through untouched.
+ *
+ * Activate with:
+ * bpftool struct_ops register nix_origin.bpf.o /sys/fs/bpf
+ * echo ':nix-origin:B::::nix_origin:' > /proc/sys/fs/binfmt_misc/register
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define PATH_MAX 4096
+#define EI_CLASS 4
+#define ELFCLASSXX 2 /* ELFCLASS64; flip to 1 for 32-bit */
+#define PT_INTERP 3
+#define MAX_PHDRS 64
+
+#define ORIGIN "$ORIGIN"
+#define ORIGIN_LEN (sizeof(ORIGIN) - 1)
+
+#define ENOENT 2
+#define ENOEXEC 8
+#define ENAMETOOLONG 36
+
+extern int bpf_dynptr_from_file(struct file *file, __u32 flags,
+ struct bpf_dynptr *ptr__uninit) __ksym;
+extern int bpf_dynptr_file_discard(struct bpf_dynptr *dynptr) __ksym;
+extern int bpf_path_d_path(const struct path *path, char *buf,
+ size_t buf__sz) __ksym;
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+
+struct scratch {
+ char interp[PATH_MAX]; /* PT_INTERP as embedded in the binary */
+ char path[PATH_MAX]; /* d_path of the binary, becomes the result */
+};
+
+/* Keyed by pid: execs run concurrently and the programs can sleep. */
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(max_entries, 512);
+ __type(key, __u64);
+ __type(value, struct scratch);
+} scratch_map SEC(".maps");
+
+static const struct scratch zero_scratch;
+
+/* An ELF64 binary per the prefetched header? */
+static bool is_elf64(struct linux_binprm *bprm)
+{
+ return bprm->buf[0] == 0x7f && bprm->buf[1] == 'E' &&
+ bprm->buf[2] == 'L' && bprm->buf[3] == 'F' &&
+ bprm->buf[EI_CLASS] == ELFCLASSXX;
+}
+
+/* Locate PT_INTERP; false if the file has none or looks malformed. */
+static bool find_pt_interp(struct bpf_dynptr *dp, struct elf64_phdr *phdr)
+{
+ struct elf64_hdr ehdr;
+ bool found = false;
+ int i;
+
+ if (bpf_dynptr_read(&ehdr, sizeof(ehdr), dp, 0, 0))
+ return false;
+ if (ehdr.e_phentsize != sizeof(struct elf64_phdr))
+ return false;
+
+ bpf_for(i, 0, ehdr.e_phnum) {
+ if (i >= MAX_PHDRS)
+ break;
+ if (bpf_dynptr_read(phdr, sizeof(*phdr), dp,
+ ehdr.e_phoff + i * sizeof(*phdr), 0))
+ return false;
+ if (phdr->p_type == PT_INTERP) {
+ found = true;
+ break;
+ }
+ }
+ return found;
+}
+
+/*
+ * An ELF64 binary whose PT_INTERP starts with "$ORIGIN/" is ours. The
+ * match can sleep and read the file, so the decision is made here and
+ * regular binaries never commit to this handler: later binfmt_misc
+ * entries and binfmt_elf see them as if we did not exist.
+ */
+SEC("struct_ops.s/match")
+bool BPF_PROG(nix_origin_match, struct linux_binprm *bprm)
+{
+ char prefix[ORIGIN_LEN + 1] = {};
+ struct elf64_phdr phdr;
+ struct bpf_dynptr dp;
+ bool ours = false;
+
+ if (!is_elf64(bprm))
+ return false;
+
+ /* The dynptr must be discarded on every path once requested. */
+ if (bpf_dynptr_from_file(bprm->file, 0, &dp))
+ goto out;
+ if (find_pt_interp(&dp, &phdr) &&
+ phdr.p_filesz > ORIGIN_LEN + 1 &&
+ !bpf_dynptr_read(prefix, sizeof(prefix), &dp, phdr.p_offset, 0))
+ ours = !bpf_strncmp(prefix, sizeof(prefix), ORIGIN "/");
+out:
+ bpf_dynptr_file_discard(&dp);
+ return ours;
+}
+
+/*
+ * The match is committed and already vetted the "$ORIGIN/" prefix, so
+ * everything here reads the file again from scratch: -ENOEXEC only
+ * covers a binary that changed under us and stopped being ours.
+ */
+SEC("struct_ops.s/load")
+int BPF_PROG(nix_origin_load, struct linux_binprm *bprm)
+{
+ __u32 isz, sfx, rsz, slash;
+ struct elf64_phdr phdr;
+ struct bpf_dynptr dp;
+ struct scratch *sc;
+ __u64 id;
+ int ret = -ENOEXEC, len, i;
+
+ if (bpf_dynptr_from_file(bprm->file, 0, &dp))
+ goto out;
+
+ if (!find_pt_interp(&dp, &phdr))
+ goto out;
+
+ isz = phdr.p_filesz;
+ if (isz <= ORIGIN_LEN + 1 || isz >= sizeof(sc->interp))
+ goto out;
+ /*
+ * The range check above compiles to a test on a zero-extended copy of
+ * the u64 p_filesz, so the verifier does not carry the bound to the
+ * dynptr_read() length below ("unbounded memory access"). Mask isz to
+ * the buffer size (a power of two) and force the masked value to be
+ * materialized with a barrier so the read uses the bounded register.
+ */
+ isz &= sizeof(sc->interp) - 1;
+ barrier_var(isz);
+
+ id = bpf_get_current_pid_tgid();
+ if (bpf_map_update_elem(&scratch_map, &id, &zero_scratch, BPF_ANY))
+ goto out;
+ sc = bpf_map_lookup_elem(&scratch_map, &id);
+ if (!sc)
+ goto out_del;
+
+ if (bpf_dynptr_read(sc->interp, isz, &dp, phdr.p_offset, 0))
+ goto out_del;
+ if (sc->interp[isz - 1] != '\0')
+ goto out_del;
+
+ /* Not "$ORIGIN/..." anymore? Then it is not ours anymore either. */
+ if (sc->interp[0] != '$' || sc->interp[1] != 'O' ||
+ sc->interp[2] != 'R' || sc->interp[3] != 'I' ||
+ sc->interp[4] != 'G' || sc->interp[5] != 'I' ||
+ sc->interp[6] != 'N' || sc->interp[7] != '/')
+ goto out_del;
+
+ /*
+ * From here on resolution failures fail the exec instead of falling
+ * back to binfmt_elf, which would resolve the literal "$ORIGIN/..."
+ * relative to the caller's cwd.
+ */
+ ret = -ENOENT;
+ len = bpf_path_d_path(&bprm->file->f_path, sc->path, sizeof(sc->path));
+ if (len <= 0 || len > sizeof(sc->path))
+ goto out_del;
+ /* Unreachable or unlinked ("... (deleted)") binaries can't resolve. */
+ if (sc->path[0] != '/')
+ goto out_del;
+
+ /* $ORIGIN = dirname of the binary. */
+ slash = 0;
+ bpf_for(i, 1, len - 1) {
+ if (i >= sizeof(sc->path))
+ break;
+ if (sc->path[i] == '/')
+ slash = i;
+ }
+
+ /* Splice the suffix (leading '/' and NUL included) onto the dir. */
+ sfx = isz - ORIGIN_LEN;
+ rsz = slash + sfx;
+ if (rsz > sizeof(sc->path)) {
+ ret = -ENAMETOOLONG;
+ goto out_del;
+ }
+ bpf_for(i, 0, sfx) {
+ __u32 s = ORIGIN_LEN + i, d = slash + i;
+
+ if (s >= sizeof(sc->interp) || d >= sizeof(sc->path))
+ break;
+ sc->path[d] = sc->interp[s];
+ }
+
+ ret = bpf_binprm_set_interp(bprm, sc->path, rsz);
+out_del:
+ bpf_map_delete_elem(&scratch_map, &id);
+out:
+ bpf_dynptr_file_discard(&dp);
+ return ret;
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops nix_origin = {
+ .match = (void *)nix_origin_match,
+ .load = (void *)nix_origin_load,
+ .name = "nix_origin",
+};
diff --git a/tools/testing/selftests/exec/transparent.bpf.c b/tools/testing/selftests/exec/transparent.bpf.c
new file mode 100644
index 000000000000..7632019ebe69
--- /dev/null
+++ b/tools/testing/selftests/exec/transparent.bpf.c
@@ -0,0 +1,57 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the transparent-mode case: match a synthetic
+ * riscv ELF header and run the asserting interpreter transparently - the
+ * argument vector untouched, the binary in AT_EXECFD and mm->exe_file
+ * labeled with the binary.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define EM_RISCV 243
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+extern int bpf_binprm_set_flags(struct linux_binprm *bprm,
+ enum bpf_binprm_flags flags) __ksym;
+
+SEC("struct_ops.s/match")
+bool BPF_PROG(transparent_match, struct linux_binprm *bprm)
+{
+ __u16 machine;
+
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* e_machine is a 16-bit little-endian field at offset 18. */
+ machine = (__u8)bprm->buf[18] | ((__u16)(__u8)bprm->buf[19] << 8);
+ return machine == EM_RISCV;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(transparent_load, struct linux_binprm *bprm)
+{
+ char interp[] = "/tmp/binfmt_transparent_interp";
+ int err;
+
+ err = bpf_binprm_set_flags(bprm, BPF_BINPRM_TRANSPARENT);
+ if (err)
+ return err;
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops transparent = {
+ .match = (void *)transparent_match,
+ .load = (void *)transparent_load,
+ .name = "transparent",
+};