summaryrefslogtreecommitdiff
path: root/tools/testing
diff options
context:
space:
mode:
Diffstat (limited to 'tools/testing')
-rw-r--r--tools/testing/selftests/Makefile2
-rw-r--r--tools/testing/selftests/alsa/.gitignore1
-rw-r--r--tools/testing/selftests/alsa/Makefile2
-rw-r--r--tools/testing/selftests/alsa/aloop-test.c345
-rw-r--r--tools/testing/selftests/alsa/mixer-test.c235
-rw-r--r--tools/testing/selftests/bpf/Makefile12
-rw-r--r--tools/testing/selftests/bpf/Makefile.buildvars25
-rw-r--r--tools/testing/selftests/bpf/Makefile.skel18
-rw-r--r--tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c21
-rw-r--r--tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c2
-rw-r--r--tools/testing/selftests/bpf/prog_tests/btf.c132
-rw-r--r--tools/testing/selftests/bpf/prog_tests/btf_rust.c142
-rw-r--r--tools/testing/selftests/bpf/prog_tests/data_in_arena.c216
-rw-r--r--tools/testing/selftests/bpf/prog_tests/verifier.c2
-rw-r--r--tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c76
-rw-r--r--tools/testing/selftests/bpf/progs/data_in_arena.c112
-rw-r--r--tools/testing/selftests/bpf/progs/data_in_arena_decl.c37
-rw-r--r--tools/testing/selftests/bpf/progs/data_in_arena_extern.c20
-rw-r--r--tools/testing/selftests/bpf/progs/data_in_arena_fail.c20
-rw-r--r--tools/testing/selftests/bpf/progs/data_in_arena_nomap.c20
-rw-r--r--tools/testing/selftests/bpf/progs/data_in_arena_rust.rs73
-rw-r--r--tools/testing/selftests/bpf/progs/dynptr_fail.c99
-rw-r--r--tools/testing/selftests/bpf/progs/test_siphash.h2
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_arena_scalar.c912
-rw-r--r--tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c173
-rw-r--r--tools/testing/selftests/bpf/test_loader.c2
-rwxr-xr-xtools/testing/selftests/drivers/net/psp.py56
-rw-r--r--tools/testing/selftests/filesystems/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/empty_mntns/.gitignore3
-rw-r--r--tools/testing/selftests/filesystems/empty_mntns/Makefile5
-rw-r--r--tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c108
-rw-r--r--tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c129
-rw-r--r--tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c45
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/.gitignore8
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/Makefile12
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/config39
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c377
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c1542
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c565
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c193
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c190
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c181
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/settings1
-rw-r--r--tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c484
-rw-r--r--tools/testing/selftests/filesystems/open_tree_ns/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/open_tree_ns/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c195
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/Makefile1
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c174
-rw-r--r--tools/testing/selftests/filesystems/readdir_hold.h224
-rw-r--r--tools/testing/selftests/filesystems/rw_hint_test.c129
-rw-r--r--tools/testing/selftests/filesystems/umount_propagation/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c432
-rw-r--r--tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c211
-rw-r--r--tools/testing/selftests/kselftest/runner.sh6
-rw-r--r--tools/testing/selftests/kvm/arm64/vgic_init.c116
-rw-r--r--tools/testing/selftests/mm/uffd-unit-tests.c36
-rw-r--r--tools/testing/selftests/net/.gitignore1
-rw-r--r--tools/testing/selftests/net/Makefile1
-rw-r--r--tools/testing/selftests/net/config2
-rwxr-xr-xtools/testing/selftests/net/cork_fragsize.py5
-rw-r--r--tools/testing/selftests/net/so_reserve_mem.c448
-rw-r--r--tools/testing/selftests/nfsd/.gitignore1
-rw-r--r--tools/testing/selftests/nfsd/Makefile6
-rw-r--r--tools/testing/selftests/nfsd/config14
-rw-r--r--tools/testing/selftests/nfsd/nfsd_netlink_listener.c1323
-rw-r--r--tools/testing/selftests/nfsd/settings1
-rw-r--r--tools/testing/selftests/nommu/Makefile8
-rw-r--r--tools/testing/selftests/nommu/local.mk7
-rw-r--r--tools/testing/selftests/nommu/nommu_mmap_test.c261
-rw-r--r--tools/testing/selftests/nommu/nommu_mremap_test.c366
-rw-r--r--tools/testing/selftests/seccomp/seccomp_bpf.c413
-rw-r--r--tools/testing/vma/include/dup.h22
75 files changed, 10951 insertions, 100 deletions
diff --git a/tools/testing/selftests/Makefile b/tools/testing/selftests/Makefile
index 79a00e9ee46d..6e7a5c0658e3 100644
--- a/tools/testing/selftests/Makefile
+++ b/tools/testing/selftests/Makefile
@@ -45,6 +45,7 @@ TARGETS += filesystems/open_tree_ns
TARGETS += filesystems/overlayfs
TARGETS += filesystems/statmount
TARGETS += filesystems/mount-notify
+TARGETS += filesystems/mount_cycle
TARGETS += filesystems/nsfs
TARGETS += filesystems/fuse
TARGETS += filesystems/move_mount
@@ -97,6 +98,7 @@ TARGETS += net/packetdrill
TARGETS += net/ppp
TARGETS += net/rds
TARGETS += net/tcp_ao
+TARGETS += nfsd
TARGETS += nolibc
TARGETS += pci_endpoint
TARGETS += pcie_bwctrl
diff --git a/tools/testing/selftests/alsa/.gitignore b/tools/testing/selftests/alsa/.gitignore
index 3dd8e1176b89..7b0e1e9ebf1b 100644
--- a/tools/testing/selftests/alsa/.gitignore
+++ b/tools/testing/selftests/alsa/.gitignore
@@ -1,3 +1,4 @@
+aloop-test
global-timer
mixer-test
pcm-test
diff --git a/tools/testing/selftests/alsa/Makefile b/tools/testing/selftests/alsa/Makefile
index 8dab90ad22bb..afd64a679dc1 100644
--- a/tools/testing/selftests/alsa/Makefile
+++ b/tools/testing/selftests/alsa/Makefile
@@ -16,7 +16,7 @@ LDLIBS+=-lpthread
OVERRIDE_TARGETS = 1
-TEST_GEN_PROGS := mixer-test pcm-test test-pcmtest-driver utimer-test
+TEST_GEN_PROGS := aloop-test mixer-test pcm-test test-pcmtest-driver utimer-test
TEST_GEN_PROGS_EXTENDED := libatest.so global-timer
diff --git a/tools/testing/selftests/alsa/aloop-test.c b/tools/testing/selftests/alsa/aloop-test.c
new file mode 100644
index 000000000000..a58b9fdbdc17
--- /dev/null
+++ b/tools/testing/selftests/alsa/aloop-test.c
@@ -0,0 +1,345 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Tests for the hw constraints between the two ends of an snd-aloop cable,
+ * with and without the "PCM Notify" control, and for the "PCM Slave"
+ * controls that report the playback side's parameters.
+ *
+ * Needs snd-aloop loaded with the default card id "Loopback". The tests use
+ * the cable between hw:Loopback,0,0 (playback) and hw:Loopback,1,0 (capture).
+ */
+#include <errno.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <string.h>
+#include <alsa/asoundlib.h>
+#include "kselftest_harness.h"
+
+#define FRAMES 1024
+#define MAX_CHANNELS 4
+
+struct stream_params {
+ snd_pcm_access_t access;
+ snd_pcm_format_t format;
+ unsigned int channels;
+ unsigned int rate;
+};
+
+/* The playback params differ from the capture params in every field. */
+static const struct stream_params capture_params = {
+ SND_PCM_ACCESS_RW_INTERLEAVED, SND_PCM_FORMAT_S16_LE, 2, 44100
+};
+
+static const struct stream_params playback_params = {
+ SND_PCM_ACCESS_RW_NONINTERLEAVED, SND_PCM_FORMAT_S32_LE, 4, 96000
+};
+
+struct probe_result {
+ unsigned int rate_min, rate_max;
+ unsigned int channels_min, channels_max;
+ bool format_ok;
+};
+
+/* Value events seen on the cable's controls */
+enum {
+ EV_ACTIVE = 1 << 0,
+ EV_FORMAT = 1 << 1,
+ EV_RATE = 1 << 2,
+ EV_CHANNELS = 1 << 3,
+ EV_ACCESS = 1 << 4,
+};
+
+FIXTURE(aloop) {
+ char play_name[32];
+ char capt_name[32];
+ snd_ctl_t *ctl;
+ bool saved_notify;
+ bool notify_saved;
+};
+
+/* The cable's controls are on the capture side device. */
+static void cable_ctl_id(snd_ctl_elem_value_t *value, const char *name)
+{
+ snd_ctl_elem_value_set_interface(value, SND_CTL_ELEM_IFACE_PCM);
+ snd_ctl_elem_value_set_name(value, name);
+ snd_ctl_elem_value_set_device(value, 1);
+ snd_ctl_elem_value_set_subdevice(value, 0);
+}
+
+static long cable_ctl_get(snd_ctl_t *ctl, const char *name)
+{
+ snd_ctl_elem_value_t *value;
+
+ snd_ctl_elem_value_alloca(&value);
+ cable_ctl_id(value, name);
+ if (snd_ctl_elem_read(ctl, value) < 0)
+ return -1;
+ if (!strcmp(name, "PCM Slave Access Mode"))
+ return snd_ctl_elem_value_get_enumerated(value, 0);
+ return snd_ctl_elem_value_get_integer(value, 0);
+}
+
+static int set_notify(snd_ctl_t *ctl, bool on)
+{
+ snd_ctl_elem_value_t *value;
+
+ snd_ctl_elem_value_alloca(&value);
+ cable_ctl_id(value, "PCM Notify");
+ snd_ctl_elem_value_set_boolean(value, 0, on);
+ return snd_ctl_elem_write(ctl, value);
+}
+
+/* Read all pending control events and return the EV_* bits for the cable's controls. */
+static unsigned int read_events(snd_ctl_t *ctl)
+{
+ static const struct {
+ const char *name;
+ unsigned int bit;
+ } names[] = {
+ { "PCM Slave Active", EV_ACTIVE },
+ { "PCM Slave Format", EV_FORMAT },
+ { "PCM Slave Rate", EV_RATE },
+ { "PCM Slave Channels", EV_CHANNELS },
+ { "PCM Slave Access Mode", EV_ACCESS },
+ };
+ snd_ctl_event_t *event;
+ unsigned int seen = 0;
+ int i;
+
+ snd_ctl_event_alloca(&event);
+ while (snd_ctl_read(ctl, event) > 0) {
+ if (snd_ctl_event_get_type(event) != SND_CTL_EVENT_ELEM ||
+ !(snd_ctl_event_elem_get_mask(event) & SND_CTL_EVENT_MASK_VALUE) ||
+ snd_ctl_event_elem_get_device(event) != 1 ||
+ snd_ctl_event_elem_get_subdevice(event) != 0)
+ continue;
+ for (i = 0; i < ARRAY_SIZE(names); i++)
+ if (!strcmp(snd_ctl_event_elem_get_name(event), names[i].name))
+ seen |= names[i].bit;
+ }
+ return seen;
+}
+
+/* Open and configure a stream. snd_pcm_hw_params() also prepares it. */
+static int open_pcm(snd_pcm_t **pcm, const char *name, snd_pcm_stream_t stream,
+ const struct stream_params *p)
+{
+ unsigned int buffer_time = 100000;
+ snd_pcm_hw_params_t *hw;
+ int err;
+
+ snd_pcm_hw_params_alloca(&hw);
+ err = snd_pcm_open(pcm, name, stream, 0);
+ if (err < 0)
+ return err;
+ err = snd_pcm_hw_params_any(*pcm, hw);
+ if (err >= 0)
+ err = snd_pcm_hw_params_set_access(*pcm, hw, p->access);
+ if (err >= 0)
+ err = snd_pcm_hw_params_set_format(*pcm, hw, p->format);
+ if (err >= 0)
+ err = snd_pcm_hw_params_set_channels(*pcm, hw, p->channels);
+ if (err >= 0)
+ err = snd_pcm_hw_params_set_rate(*pcm, hw, p->rate, 0);
+ if (err >= 0)
+ err = snd_pcm_hw_params_set_buffer_time_near(*pcm, hw, &buffer_time, NULL);
+ if (err >= 0)
+ err = snd_pcm_hw_params(*pcm, hw);
+ if (err < 0) {
+ snd_pcm_close(*pcm);
+ *pcm = NULL;
+ }
+ return err;
+}
+
+/* What a client probing the device sees, and whether it may use the given format */
+static int probe_pcm(const char *name, snd_pcm_stream_t stream, snd_pcm_format_t format,
+ struct probe_result *res)
+{
+ snd_pcm_hw_params_t *hw;
+ snd_pcm_t *pcm;
+ int err;
+
+ snd_pcm_hw_params_alloca(&hw);
+ err = snd_pcm_open(&pcm, name, stream, 0);
+ if (err < 0)
+ return err;
+ err = snd_pcm_hw_params_any(pcm, hw);
+ if (err >= 0)
+ err = snd_pcm_hw_params_get_rate_min(hw, &res->rate_min, NULL);
+ if (err >= 0)
+ err = snd_pcm_hw_params_get_rate_max(hw, &res->rate_max, NULL);
+ if (err >= 0)
+ err = snd_pcm_hw_params_get_channels_min(hw, &res->channels_min);
+ if (err >= 0)
+ err = snd_pcm_hw_params_get_channels_max(hw, &res->channels_max);
+ if (err >= 0)
+ res->format_ok = !snd_pcm_hw_params_test_format(pcm, hw, format);
+ snd_pcm_close(pcm);
+ return err;
+}
+
+static int start_playback(snd_pcm_t *pcm, const struct stream_params *p)
+{
+ static char silence[FRAMES * MAX_CHANNELS * 4];
+ void *bufs[MAX_CHANNELS];
+ snd_pcm_sframes_t written;
+ unsigned int i;
+
+ if (p->access == SND_PCM_ACCESS_RW_NONINTERLEAVED) {
+ for (i = 0; i < p->channels; i++)
+ bufs[i] = silence + i * FRAMES * 4;
+ written = snd_pcm_writen(pcm, bufs, FRAMES);
+ } else {
+ written = snd_pcm_writei(pcm, silence, FRAMES);
+ }
+ if (written < 0)
+ return written;
+ if (snd_pcm_state(pcm) == SND_PCM_STATE_PREPARED)
+ return snd_pcm_start(pcm);
+ return 0;
+}
+
+FIXTURE_SETUP(aloop) {
+ char ctl_name[32];
+ snd_pcm_t *pcm;
+ int card, err;
+
+ card = snd_card_get_index("Loopback");
+ if (card < 0)
+ SKIP(return, "No Loopback card, snd-aloop is probably not loaded");
+
+ sprintf(ctl_name, "hw:%d", card);
+ sprintf(self->play_name, "hw:%d,0,0", card);
+ sprintf(self->capt_name, "hw:%d,1,0", card);
+
+ err = snd_pcm_open(&pcm, self->capt_name, SND_PCM_STREAM_CAPTURE, SND_PCM_NONBLOCK);
+ if (err == -EBUSY)
+ SKIP(return, "%s is in use", self->capt_name);
+ ASSERT_EQ(err, 0);
+ snd_pcm_close(pcm);
+ err = snd_pcm_open(&pcm, self->play_name, SND_PCM_STREAM_PLAYBACK, SND_PCM_NONBLOCK);
+ if (err == -EBUSY)
+ SKIP(return, "%s is in use", self->play_name);
+ ASSERT_EQ(err, 0);
+ snd_pcm_close(pcm);
+
+ ASSERT_EQ(snd_ctl_open(&self->ctl, ctl_name, SND_CTL_NONBLOCK), 0);
+ ASSERT_EQ(snd_ctl_subscribe_events(self->ctl, 1), 0);
+ self->saved_notify = cable_ctl_get(self->ctl, "PCM Notify");
+ self->notify_saved = true;
+}
+
+FIXTURE_TEARDOWN(aloop) {
+ if (self->notify_saved)
+ set_notify(self->ctl, self->saved_notify);
+ if (self->ctl)
+ snd_ctl_close(self->ctl);
+}
+
+/* Without notify, a playback opened while a capture is set up is pinned to its parameters. */
+TEST_F(aloop, playback_constrained_without_notify) {
+ struct probe_result res;
+ snd_pcm_t *capt;
+
+ ASSERT_EQ(set_notify(self->ctl, false), 0);
+ ASSERT_EQ(open_pcm(&capt, self->capt_name, SND_PCM_STREAM_CAPTURE, &capture_params), 0);
+
+ ASSERT_EQ(probe_pcm(self->play_name, SND_PCM_STREAM_PLAYBACK, playback_params.format,
+ &res), 0);
+ EXPECT_EQ(res.rate_min, capture_params.rate);
+ EXPECT_EQ(res.rate_max, capture_params.rate);
+ EXPECT_EQ(res.channels_min, capture_params.channels);
+ EXPECT_EQ(res.channels_max, capture_params.channels);
+ EXPECT_FALSE(res.format_ok);
+
+ snd_pcm_close(capt);
+}
+
+/* With notify, the playback side is free to pick other parameters. */
+TEST_F(aloop, playback_unconstrained_with_notify) {
+ struct probe_result res;
+ snd_pcm_t *capt;
+
+ ASSERT_EQ(set_notify(self->ctl, true), 0);
+ ASSERT_EQ(open_pcm(&capt, self->capt_name, SND_PCM_STREAM_CAPTURE, &capture_params), 0);
+
+ ASSERT_EQ(probe_pcm(self->play_name, SND_PCM_STREAM_PLAYBACK, playback_params.format,
+ &res), 0);
+ EXPECT_LE(res.rate_min, capture_params.rate);
+ EXPECT_GE(res.rate_max, playback_params.rate);
+ EXPECT_LE(res.channels_min, capture_params.channels);
+ EXPECT_GE(res.channels_max, playback_params.channels);
+ EXPECT_TRUE(res.format_ok);
+
+ snd_pcm_close(capt);
+}
+
+/*
+ * With notify, starting a playback with different parameters stops the
+ * running capture, and the "PCM Slave" controls report the new parameters
+ * with a value event for each one that changed.
+ */
+TEST_F(aloop, params_change_stops_capture_with_notify) {
+ snd_pcm_t *capt, *play;
+ unsigned int events;
+
+ /*
+ * The controls keep the last playback's parameters, and only notify on
+ * a change. Start a playback with the capture's parameters while no
+ * capture is open, so that every control changes below.
+ */
+ ASSERT_EQ(open_pcm(&play, self->play_name, SND_PCM_STREAM_PLAYBACK, &capture_params), 0);
+ ASSERT_EQ(start_playback(play, &capture_params), 0);
+ snd_pcm_close(play);
+
+ ASSERT_EQ(set_notify(self->ctl, true), 0);
+ ASSERT_EQ(open_pcm(&capt, self->capt_name, SND_PCM_STREAM_CAPTURE, &capture_params), 0);
+ ASSERT_EQ(snd_pcm_start(capt), 0);
+ ASSERT_EQ(snd_pcm_state(capt), SND_PCM_STATE_RUNNING);
+ EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Active"), 0);
+ read_events(self->ctl);
+
+ ASSERT_EQ(open_pcm(&play, self->play_name, SND_PCM_STREAM_PLAYBACK, &playback_params), 0)
+ TH_LOG("Playback refused other parameters while the capture is running");
+ ASSERT_EQ(start_playback(play, &playback_params), 0);
+
+ /* loopback_check_format() stops the capture from the playback's start trigger. */
+ EXPECT_NE(snd_pcm_state(capt), SND_PCM_STATE_RUNNING);
+
+ EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Active"), 1);
+ EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Format"), playback_params.format);
+ EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Rate"), playback_params.rate);
+ EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Channels"), playback_params.channels);
+ EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Access Mode"), 1);
+
+ events = read_events(self->ctl);
+ EXPECT_TRUE(events & EV_ACTIVE);
+ EXPECT_TRUE(events & EV_FORMAT);
+ EXPECT_TRUE(events & EV_RATE);
+ EXPECT_TRUE(events & EV_CHANNELS);
+ EXPECT_TRUE(events & EV_ACCESS);
+
+ snd_pcm_close(play);
+ snd_pcm_close(capt);
+}
+
+/* With notify, a capture opened second is still pinned to the playback's parameters. */
+TEST_F(aloop, capture_constrained_with_notify) {
+ struct probe_result res;
+ snd_pcm_t *play;
+
+ ASSERT_EQ(set_notify(self->ctl, true), 0);
+ ASSERT_EQ(open_pcm(&play, self->play_name, SND_PCM_STREAM_PLAYBACK, &playback_params), 0);
+
+ ASSERT_EQ(probe_pcm(self->capt_name, SND_PCM_STREAM_CAPTURE, capture_params.format,
+ &res), 0);
+ EXPECT_EQ(res.rate_min, playback_params.rate);
+ EXPECT_EQ(res.rate_max, playback_params.rate);
+ EXPECT_EQ(res.channels_min, playback_params.channels);
+ EXPECT_EQ(res.channels_max, playback_params.channels);
+ EXPECT_FALSE(res.format_ok);
+
+ snd_pcm_close(play);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/alsa/mixer-test.c b/tools/testing/selftests/alsa/mixer-test.c
index 53a72753bb08..966ef83354b9 100644
--- a/tools/testing/selftests/alsa/mixer-test.c
+++ b/tools/testing/selftests/alsa/mixer-test.c
@@ -28,7 +28,7 @@
#include "kselftest.h"
#include "alsa-local.h"
-#define TESTS_PER_CONTROL 7
+#define TESTS_PER_CONTROL 8
/* Suffixes of the SNDRV_CTL_NAME_IEC958() names, not exported to userspace */
#define IEC958_DEFAULT "Default"
@@ -52,9 +52,14 @@ struct ctl_data {
snd_ctl_elem_id_t *id;
snd_ctl_elem_info_t *info;
snd_ctl_elem_value_t *def_val;
+ snd_ctl_elem_value_t *snapshot;
+ bool snapshot_valid;
+ bool moved;
int elem;
int event_missing;
int event_spurious;
+ int side_effects;
+ unsigned int ev_mask;
struct card_data *card;
struct ctl_data *next;
};
@@ -159,6 +164,10 @@ static void find_controls(void)
if (err < 0)
ksft_exit_fail_msg("Out of memory\n");
+ err = snd_ctl_elem_value_malloc(&ctl_data->snapshot);
+ if (err < 0)
+ ksft_exit_fail_msg("Out of memory\n");
+
snd_ctl_elem_list_get_id(card_data->ctls, ctl,
ctl_data->id);
snd_ctl_elem_info_set_id(ctl_data->info, ctl_data->id);
@@ -207,6 +216,20 @@ static void find_controls(void)
snd_config_delete(config);
}
+/* The control on the same card that an event's numid refers to */
+static struct ctl_data *find_ctl_by_numid(struct card_data *card,
+ unsigned int numid)
+{
+ struct ctl_data *ctl;
+
+ for (ctl = ctl_list; ctl != NULL; ctl = ctl->next)
+ if (ctl->card == card &&
+ snd_ctl_elem_info_get_numid(ctl->info) == numid)
+ return ctl;
+
+ return NULL;
+}
+
/*
* Block for up to timeout ms for an event, returns a negative value
* on error, 0 for no event and 1 for an event.
@@ -265,8 +288,21 @@ static int wait_for_event(struct ctl_data *ctl, int timeout)
mask = snd_ctl_event_elem_get_mask(event);
ev_id = snd_ctl_event_elem_get_numid(event);
if (ev_id != snd_ctl_elem_info_get_numid(ctl->info)) {
+ struct ctl_data *other = find_ctl_by_numid(ctl->card,
+ ev_id);
+
+ /*
+ * Remember that the driver announced this one.
+ * test_ctl_write_side_effects() uses that to tell a
+ * deliberate link from a silent register collision.
+ */
+ if (other)
+ other->ev_mask |= mask;
+
ksft_print_msg("Event for unexpected ctl %s\n",
snd_ctl_event_elem_get_name(event));
+ /* The loop condition must not see the other control's mask */
+ mask = 0;
continue;
}
@@ -1147,6 +1183,202 @@ static void test_ctl_write_valid(struct ctl_data *ctl)
ctl->card->card_name, ctl->elem);
}
+/*
+ * Build the smallest or the largest value the control offers. The smallest
+ * clears the control's register field and the largest sets its top bit, which
+ * is the bit a mask one bit too wide puts in its neighbour.
+ */
+static bool set_limit_value(struct ctl_data *ctl, snd_ctl_elem_value_t *val,
+ bool max)
+{
+ int i, count = snd_ctl_elem_info_get_count(ctl->info);
+
+ snd_ctl_elem_value_set_id(val, ctl->id);
+
+ switch (snd_ctl_elem_info_get_type(ctl->info)) {
+ case SND_CTL_ELEM_TYPE_BOOLEAN:
+ for (i = 0; i < count; i++)
+ snd_ctl_elem_value_set_boolean(val, i, max);
+ return true;
+
+ case SND_CTL_ELEM_TYPE_INTEGER:
+ for (i = 0; i < count; i++)
+ snd_ctl_elem_value_set_integer(val, i, max ?
+ snd_ctl_elem_info_get_max(ctl->info) :
+ snd_ctl_elem_info_get_min(ctl->info));
+ return true;
+
+ case SND_CTL_ELEM_TYPE_INTEGER64:
+ for (i = 0; i < count; i++)
+ snd_ctl_elem_value_set_integer64(val, i, max ?
+ snd_ctl_elem_info_get_max64(ctl->info) :
+ snd_ctl_elem_info_get_min64(ctl->info));
+ return true;
+
+ case SND_CTL_ELEM_TYPE_ENUMERATED:
+ for (i = 0; i < count; i++)
+ snd_ctl_elem_value_set_enumerated(val, i, max ?
+ snd_ctl_elem_info_get_items(ctl->info) - 1 : 0);
+ return true;
+
+ default:
+ /* Nothing sensible to write for the rest */
+ return false;
+ }
+}
+
+/* Note every control on the card that no longer reads as it did */
+static void find_moved_ctls(struct ctl_data *ctl, snd_ctl_elem_value_t *val)
+{
+ struct ctl_data *other;
+ int err;
+
+ for (other = ctl_list; other != NULL; other = other->next) {
+ if (!other->snapshot_valid)
+ continue;
+
+ /*
+ * The buffer is shared and compare() looks at all of it, so
+ * clear what the last control left in the slots this one
+ * does not use.
+ */
+ snd_ctl_elem_value_clear(val);
+ snd_ctl_elem_value_set_id(val, other->id);
+ err = snd_ctl_elem_read(ctl->card->handle, val);
+ if (err < 0) {
+ ksft_print_msg("snd_ctl_elem_read() failed for %s: %s\n",
+ other->name, snd_strerror(err));
+ continue;
+ }
+
+ if (snd_ctl_elem_value_compare(other->snapshot, val))
+ other->moved = true;
+ }
+}
+
+/*
+ * Write one control and look for others on the same card that moved with it.
+ * A driver that links two controls on purpose tells userspace about both, so
+ * only an unannounced change is counted. That is what a control whose
+ * register mask covers bits belonging to its neighbour looks like from here.
+ */
+static void test_ctl_write_side_effects(struct ctl_data *ctl)
+{
+ struct ctl_data *other;
+ snd_ctl_elem_value_t *min_val, *max_val, *read_val;
+ int err;
+
+ snd_ctl_elem_value_alloca(&min_val);
+ snd_ctl_elem_value_alloca(&max_val);
+ snd_ctl_elem_value_alloca(&read_val);
+
+ /* Without a readable default there is nothing to put back */
+ if (snd_ctl_elem_info_is_inactive(ctl->info) ||
+ !snd_ctl_elem_info_is_writable(ctl->info) ||
+ !snd_ctl_elem_info_is_readable(ctl->info) ||
+ !set_limit_value(ctl, min_val, false) ||
+ !set_limit_value(ctl, max_val, true)) {
+ ksft_test_result_skip("write_side_effects.%s.%d\n",
+ ctl->card->card_name, ctl->elem);
+ return;
+ }
+
+ /* Drain first, a stale event would look like the driver announced it */
+ drop_events(ctl);
+
+ /*
+ * Record what the rest of the card reads as. A volatile control can
+ * move on its own so there is nothing to compare it against.
+ */
+ for (other = ctl_list; other != NULL; other = other->next) {
+ other->snapshot_valid = false;
+ other->moved = false;
+ other->ev_mask = 0;
+
+ if (other == ctl || other->card != ctl->card)
+ continue;
+ if (!snd_ctl_elem_info_is_readable(other->info) ||
+ snd_ctl_elem_info_is_volatile(other->info))
+ continue;
+
+ snd_ctl_elem_value_clear(other->snapshot);
+ snd_ctl_elem_value_set_id(other->snapshot, other->id);
+ err = snd_ctl_elem_read(ctl->card->handle, other->snapshot);
+ if (err < 0) {
+ ksft_print_msg("snd_ctl_elem_read() failed for %s: %s\n",
+ other->name, snd_strerror(err));
+ continue;
+ }
+
+ other->snapshot_valid = true;
+ }
+
+ /*
+ * Compare against the snapshot after each write, before anything is
+ * put back. Restoring the control we wrote goes through the same
+ * mask, so doing it first would hide the change we are looking for.
+ */
+ err = snd_ctl_elem_write(ctl->card->handle, min_val);
+ if (err >= 0) {
+ drop_events(ctl);
+ find_moved_ctls(ctl, read_val);
+ err = snd_ctl_elem_write(ctl->card->handle, max_val);
+ }
+ if (err < 0) {
+ ksft_print_msg("snd_ctl_elem_write() failed for %s: %s\n",
+ ctl->name, snd_strerror(err));
+ } else {
+ drop_events(ctl);
+ find_moved_ctls(ctl, read_val);
+ }
+
+ for (other = ctl_list; other != NULL; other = other->next) {
+ if (!other->moved)
+ continue;
+
+ if (other->ev_mask & SND_CTL_EVENT_MASK_VALUE) {
+ ksft_print_msg("Writing %s changed %s, the driver said so\n",
+ ctl->name, other->name);
+ } else {
+ ksft_print_msg("Writing %s silently changed %s\n",
+ ctl->name, other->name);
+ ctl->side_effects++;
+ }
+ }
+
+ /*
+ * The control we wrote goes back first so its mask stops moving the
+ * rest. A plain write keeps this out of the event counters, they
+ * belong to the tests that check them.
+ */
+ snd_ctl_elem_write(ctl->card->handle, ctl->def_val);
+
+ for (other = ctl_list; other != NULL; other = other->next) {
+ if (!other->snapshot_valid ||
+ !snd_ctl_elem_info_is_writable(other->info))
+ continue;
+
+ snd_ctl_elem_value_clear(read_val);
+ snd_ctl_elem_value_set_id(read_val, other->id);
+ if (snd_ctl_elem_read(ctl->card->handle, read_val) < 0)
+ continue;
+
+ if (snd_ctl_elem_value_compare(other->snapshot, read_val))
+ snd_ctl_elem_write(ctl->card->handle, other->snapshot);
+ }
+
+ /* Our own restores queue events, the next test must not see them */
+ drop_events(ctl);
+
+ if (err < 0)
+ ksft_test_result_skip("write_side_effects.%s.%d\n",
+ ctl->card->card_name, ctl->elem);
+ else
+ ksft_test_result(!ctl->side_effects,
+ "write_side_effects.%s.%d\n",
+ ctl->card->card_name, ctl->elem);
+}
+
static bool test_ctl_write_invalid_value(struct ctl_data *ctl,
snd_ctl_elem_value_t *val)
{
@@ -1390,6 +1622,7 @@ int main(void)
test_ctl_name(ctl);
test_ctl_write_default(ctl);
test_ctl_write_valid(ctl);
+ test_ctl_write_side_effects(ctl);
test_ctl_write_invalid(ctl);
test_ctl_event_missing(ctl);
test_ctl_event_spurious(ctl);
diff --git a/tools/testing/selftests/bpf/Makefile b/tools/testing/selftests/bpf/Makefile
index afa589a27b15..a22be7efd1fa 100644
--- a/tools/testing/selftests/bpf/Makefile
+++ b/tools/testing/selftests/bpf/Makefile
@@ -422,6 +422,16 @@ $(LIBARENA_ASAN_SKEL): $(INCLUDE_DIR)/vmlinux.h $(BPFOBJ) $(LIBARENA_BPF_DEPS)
+$(MAKE) -C libarena libarena_asan.skel.h $(LIBARENA_MAKE_ARGS)
endif
+# #![no_std] looks for compiler_builtins too. Nothing of it is used.
+ifneq ($(RUST_CORE),)
+$(RUST_CORE): $(RUST_CORE_SRC)
+ $(call msg,RUSTC,,$@)
+ $(Q)mkdir -p $(@D)
+ +$(Q)$(RUSTC_BPF) -A warnings --edition 2024 --crate-name core --out-dir $(@D) $<
+ +$(Q)echo '#![feature(compiler_builtins)] #![compiler_builtins] #![no_std]' | \
+ $(RUSTC_BPF) -A warnings --crate-name compiler_builtins --out-dir $(@D) -
+endif
+
# Generated test list headers
define gen_tests_hdr
@@ -457,7 +467,7 @@ RUNNER_PREREQS := $(INCLUDE_DIR)/vmlinux.h $(BPFOBJ) $(BPFTOOL) \
$(VERIFY_SIG_HDR) $(PRIVATE_KEY) $(VERIFICATION_CERT) \
$(LIBARENA_SKEL) $(LIBARENA_ASAN_SKEL) \
prog_tests/tests.h map_tests/tests.h \
- $(RUNNER_OBJS)
+ $(RUNNER_OBJS) $(RUST_CORE)
# Runtime fixtures for each test_progs flavor.
RUNNER_EXTRA_FILES := $(OUTPUT)/urandom_read \
diff --git a/tools/testing/selftests/bpf/Makefile.buildvars b/tools/testing/selftests/bpf/Makefile.buildvars
index d2a0c0031b87..ff3476bc40a4 100644
--- a/tools/testing/selftests/bpf/Makefile.buildvars
+++ b/tools/testing/selftests/bpf/Makefile.buildvars
@@ -109,6 +109,31 @@ HOST_INCLUDE_DIR := $(INCLUDE_DIR)
endif
RESOLVE_BTFIDS := $(HOST_BUILD_DIR)/resolve_btfids/resolve_btfids
+# Programs in Rust are built by upstream rustc. It has no prebuilt core for
+# the bpf target, so it has to come with the source of core:
+# rustup component add rust-src
+# core is built as edition 2024, which it is since rustc 1.87.
+# rustc emits LLVM bitcode and clang makes the object of it, so clang has to be
+# 23 or newer and not older than LLVM of rustc.
+# Otherwise RUST_CORE is empty and the tests are skipped.
+RUSTC ?= rustc
+RUST_CORE_SRC := $(wildcard $(shell $(RUSTC) --print sysroot 2>/dev/null)$\
+ /lib/rustlib/src/rust/library/core/src/lib.rs)
+ifneq ($(RUST_CORE_SRC),)
+ifeq ($(shell { clang=$$(echo __clang_major__ | $(CLANG) -E -P -x c -) && \
+ llvm=$$($(srctree)/scripts/rustc-llvm-version.sh $(RUSTC)) && \
+ [ $$($(srctree)/scripts/rustc-version.sh $(RUSTC)) -ge 108700 ] && \
+ [ $$clang -ge 23 ] && [ $$clang -ge $$((llvm / 10000)) ]; } \
+ 2>/dev/null && echo y),y)
+RUST_CORE := $(BUILD_DIR)/rust/libcore.rlib
+endif
+endif
+# RUSTC_BOOTSTRAP=1 is to build core with a stable rustc, like the kernel does.
+# panic=abort is a stop gap until panic=unwind is supported.
+RUSTC_BPF = RUSTC_BOOTSTRAP=1 $(RUSTC) -O -C panic=abort --crate-type rlib \
+ --target $(if $(IS_LITTLE_ENDIAN),bpfel,bpfeb)-unknown-none \
+ -L $(dir $(RUST_CORE))
+
DEFAULT_BPFTOOL := $(HOST_SCRATCH_DIR)/sbin/bpftool
ifneq ($(CROSS_COMPILE),)
CROSS_BPFTOOL := $(SCRATCH_DIR)/sbin/bpftool
diff --git a/tools/testing/selftests/bpf/Makefile.skel b/tools/testing/selftests/bpf/Makefile.skel
index 2e22bb901bf3..06e297a17156 100644
--- a/tools/testing/selftests/bpf/Makefile.skel
+++ b/tools/testing/selftests/bpf/Makefile.skel
@@ -20,7 +20,8 @@ ifneq ($(BPF_CC),)
BPF_SRCS := $(notdir $(wildcard progs/*.c))
BPF_OBJS := $(patsubst %.c,$(RDIR)/%.bpf.o,$(BPF_SRCS))
-SKEL_BLACKLIST := btf__% test_pinning_invalid.c test_sk_assign.c
+SKEL_BLACKLIST := btf__% test_pinning_invalid.c test_sk_assign.c \
+ data_in_arena_extern.c data_in_arena_nomap.c
LINKED_SKELS := test_static_linked.skel.h linked_funcs.skel.h \
linked_vars.skel.h linked_maps.skel.h linked_arena.skel.h \
@@ -137,4 +138,19 @@ $(LINKED_SKELS_H): $(RDIR)/%.skel.h: $$(addprefix $(RDIR)/,$$($$*.skel.h-deps))
$(BPFTOOL) $(GEN_SKEL) | $(RDIR)
$(Q)$(cmd_bpf_link_skel)
+# Programs in Rust, see Makefile.buildvars. No skeletons: the objects may be absent.
+ifneq ($(RUST_CORE),)
+ifeq ($(BPF_CC),$(CLANG))
+RUST_OBJS := $(patsubst progs/%.rs,$(RDIR)/%.bpf.o,$(wildcard progs/*.rs))
+BPF_OBJS += $(RUST_OBJS)
+
+$(RUST_OBJS): $(RDIR)/%.bpf.o: progs/%.rs $(RUST_CORE) | $(RDIR)
+ $(call msg,RUSTC,$(BINARY),$@)
+ +$(Q)$(RUSTC_BPF) --edition 2021 -C debuginfo=2 --emit=llvm-bc \
+ $(patsubst -mcpu=%,-C target-cpu=%,$(filter -mcpu=%,$(BPF_CC_FLAGS))) \
+ -o $(dir $(RUST_CORE))$(BINARY)-$*.bc $< && \
+ $(BPF_CC) $(BPF_CC_FLAGS) -c $(dir $(RUST_CORE))$(BINARY)-$*.bc -o $@ $(call skip_on_fail,BPF)
+endif
+endif
+
endif # BPF_CC
diff --git a/tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c b/tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c
new file mode 100644
index 000000000000..2ac2e9a652fe
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c
@@ -0,0 +1,21 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <test_progs.h>
+#include "sysctl_helpers.h"
+#include "verifier_arena_scalar.skel.h"
+
+/* The same tests with constants of the programs blinded */
+void serial_test_arena_scalar_blinded(void)
+{
+ const char *harden = "/proc/sys/net/core/bpf_jit_harden";
+ char old[16] = {};
+
+ if (!is_jit_enabled()) {
+ test__skip();
+ return;
+ }
+ if (sysctl_set_or_fail(harden, old, "2"))
+ return;
+ RUN_TESTS(verifier_arena_scalar);
+ sysctl_set_or_fail(harden, NULL, old);
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c b/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c
index 6dbd1487343c..122ecb7e98e2 100644
--- a/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c
+++ b/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c
@@ -11,6 +11,7 @@
#include "bpf_qdisc_fail__invalid_dynptr.skel.h"
#include "bpf_qdisc_fail__invalid_dynptr_slice.skel.h"
#include "bpf_qdisc_fail__invalid_dynptr_cross_frame.skel.h"
+#include "bpf_qdisc_fail__invalid_dynptr_returned_slice.skel.h"
#include "bpf_qdisc_fail__untrusted_write.skel.h"
#include "bpf_qdisc_dynptr_use_after_invalidate_clone.skel.h"
@@ -230,6 +231,7 @@ void test_ns_bpf_qdisc(void)
test_incompl_ops();
RUN_TESTS(bpf_qdisc_fail__invalid_dynptr);
RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_cross_frame);
+ RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_returned_slice);
RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_slice);
RUN_TESTS(bpf_qdisc_fail__untrusted_write);
RUN_TESTS(bpf_qdisc_dynptr_use_after_invalidate_clone);
diff --git a/tools/testing/selftests/bpf/prog_tests/btf.c b/tools/testing/selftests/bpf/prog_tests/btf.c
index df6ad38d287d..24ab62b2834a 100644
--- a/tools/testing/selftests/bpf/prog_tests/btf.c
+++ b/tools/testing/selftests/bpf/prog_tests/btf.c
@@ -424,7 +424,7 @@ static struct btf_raw_test raw_tests[] = {
.err_str = "Invalid type",
},
{
- .descr = "global data test #8, invalid var size",
+ .descr = "global data test #8, var is smaller than its type",
.raw_types = {
/* int */
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
@@ -457,8 +457,6 @@ static struct btf_raw_test raw_tests[] = {
.key_type_id = 0,
.value_type_id = 7,
.max_entries = 1,
- .btf_load_err = true,
- .err_str = "Invalid size",
},
{
.descr = "global data test #9, invalid var size",
@@ -498,7 +496,7 @@ static struct btf_raw_test raw_tests[] = {
.err_str = "Invalid size",
},
{
- .descr = "global data test #10, invalid var size",
+ .descr = "global data test #10, section is smaller than map value",
.raw_types = {
/* int */
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
@@ -531,8 +529,7 @@ static struct btf_raw_test raw_tests[] = {
.key_type_id = 0,
.value_type_id = 7,
.max_entries = 1,
- .btf_load_err = true,
- .err_str = "Invalid size",
+ .map_create_err = true,
},
{
.descr = "global data test #11, multiple section members",
@@ -1987,14 +1984,14 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "typedef (invalid name, invalid identifier)",
+ .descr = "typedef (invalid name, not printable)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPEDEF_ENC(NAME_TBD, 1), /* [2] */
BTF_END_RAW,
},
- .str_sec = "\0__!int",
- .str_sec_size = sizeof("\0__!int"),
+ .str_sec = "\0__\7int",
+ .str_sec_size = sizeof("\0__\7int"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "typedef_check_btf",
.key_size = sizeof(int),
@@ -2112,15 +2109,15 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "fwd type (invalid name, invalid identifier)",
+ .descr = "fwd type (invalid name, not printable)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_ENC(NAME_TBD,
BTF_INFO_ENC(BTF_KIND_FWD, 0, 0), 0), /* [2] */
BTF_END_RAW,
},
- .str_sec = "\0__!skb",
- .str_sec_size = sizeof("\0__!skb"),
+ .str_sec = "\0__\7skb",
+ .str_sec_size = sizeof("\0__\7skb"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "fwd_type_check_btf",
.key_size = sizeof(int),
@@ -2175,7 +2172,7 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "struct type (invalid name, invalid identifier)",
+ .descr = "struct type (invalid name, not printable)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_ENC(NAME_TBD,
@@ -2183,8 +2180,8 @@ static struct btf_raw_test raw_tests[] = {
BTF_MEMBER_ENC(NAME_TBD, 1, 0),
BTF_END_RAW,
},
- .str_sec = "\0A!\0B",
- .str_sec_size = sizeof("\0A!\0B"),
+ .str_sec = "\0A\7\0B",
+ .str_sec_size = sizeof("\0A\7\0B"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "struct_type_check_btf",
.key_size = sizeof(int),
@@ -2217,7 +2214,7 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "struct member (invalid name, invalid identifier)",
+ .descr = "struct member (invalid name, not printable)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_ENC(NAME_TBD,
@@ -2225,8 +2222,8 @@ static struct btf_raw_test raw_tests[] = {
BTF_MEMBER_ENC(NAME_TBD, 1, 0),
BTF_END_RAW,
},
- .str_sec = "\0A\0B*",
- .str_sec_size = sizeof("\0A\0B*"),
+ .str_sec = "\0A\0B\7",
+ .str_sec_size = sizeof("\0A\0B\7"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "struct_type_check_btf",
.key_size = sizeof(int),
@@ -2260,7 +2257,7 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "enum type (invalid name, invalid identifier)",
+ .descr = "enum type (invalid name, not printable)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_ENC(NAME_TBD,
@@ -2269,8 +2266,8 @@ static struct btf_raw_test raw_tests[] = {
BTF_ENUM_ENC(NAME_TBD, 0),
BTF_END_RAW,
},
- .str_sec = "\0A!\0B",
- .str_sec_size = sizeof("\0A!\0B"),
+ .str_sec = "\0A\7\0B",
+ .str_sec_size = sizeof("\0A\7\0B"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "enum_type_check_btf",
.key_size = sizeof(int),
@@ -2306,7 +2303,7 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "enum member (invalid name, invalid identifier)",
+ .descr = "enum member (invalid name, not printable)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_ENC(0,
@@ -2315,8 +2312,8 @@ static struct btf_raw_test raw_tests[] = {
BTF_ENUM_ENC(NAME_TBD, 0),
BTF_END_RAW,
},
- .str_sec = "\0A!",
- .str_sec_size = sizeof("\0A!"),
+ .str_sec = "\0A\7",
+ .str_sec_size = sizeof("\0A\7"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "enum_type_check_btf",
.key_size = sizeof(int),
@@ -2625,14 +2622,14 @@ static struct btf_raw_test raw_tests[] = {
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_INT_ENC(0, 0, 0, 32, 4), /* [2] */
- /* void (*)(int a, unsigned int !!!) */
+ /* void (*)(int a, unsigned int \7) */
BTF_FUNC_PROTO_ENC(0, 2), /* [3] */
BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1),
BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 2),
BTF_END_RAW,
},
- .str_sec = "\0a\0!!!",
- .str_sec_size = sizeof("\0a\0!!!"),
+ .str_sec = "\0a\0\7",
+ .str_sec_size = sizeof("\0a\0\7"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "func_proto_type_check_btf",
.key_size = sizeof(int),
@@ -2775,12 +2772,12 @@ static struct btf_raw_test raw_tests[] = {
BTF_FUNC_PROTO_ENC(0, 2), /* [3] */
BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1),
BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 2),
- /* void !!!(int a, unsigned int b) */
+ /* void \7(int a, unsigned int b) */
BTF_FUNC_ENC(NAME_TBD, 3), /* [4] */
BTF_END_RAW,
},
- .str_sec = "\0a\0b\0!!!",
- .str_sec_size = sizeof("\0a\0b\0!!!"),
+ .str_sec = "\0a\0b\0\7",
+ .str_sec_size = sizeof("\0a\0b\0\7"),
.map_type = BPF_MAP_TYPE_ARRAY,
.map_name = "func_type_check_btf",
.key_size = sizeof(int),
@@ -2793,7 +2790,7 @@ static struct btf_raw_test raw_tests[] = {
},
{
- .descr = "func (Some arg has no name)",
+ .descr = "func (Some arg of global func has no name)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
BTF_TYPE_INT_ENC(0, 0, 0, 32, 4), /* [2] */
@@ -2802,7 +2799,8 @@ static struct btf_raw_test raw_tests[] = {
BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1),
BTF_FUNC_PROTO_ARG_ENC(0, 2),
/* void func(int a, unsigned int) */
- BTF_FUNC_ENC(NAME_TBD, 3), /* [4] */
+ BTF_TYPE_ENC(NAME_TBD, /* [4] */
+ BTF_INFO_ENC(BTF_KIND_FUNC, 0, BTF_FUNC_GLOBAL), 3),
BTF_END_RAW,
},
.str_sec = "\0a\0func",
@@ -2819,6 +2817,57 @@ static struct btf_raw_test raw_tests[] = {
},
{
+ .descr = "func (Some arg of static func has no name)",
+ .raw_types = {
+ /* int */
+ BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
+ /* unsigned int */
+ BTF_TYPE_INT_ENC(0, 0, 0, 32, 4), /* [2] */
+ /* void (*)(int a, unsigned int) */
+ BTF_FUNC_PROTO_ENC(0, 2), /* [3] */
+ BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1),
+ BTF_FUNC_PROTO_ARG_ENC(0, 2),
+ /* static void func(int a, unsigned int) */
+ BTF_FUNC_ENC(NAME_TBD, 3), /* [4] */
+ BTF_END_RAW,
+ },
+ .str_sec = "\0a\0func",
+ .str_sec_size = sizeof("\0a\0func"),
+ .map_type = BPF_MAP_TYPE_ARRAY,
+ .map_name = "func_type_check_btf",
+ .key_size = sizeof(int),
+ .value_size = sizeof(int),
+ .key_type_id = 1,
+ .value_type_id = 1,
+ .max_entries = 4,
+},
+
+{
+ .descr = "func (vararg of global func has no name)",
+ .raw_types = {
+ /* int */
+ BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
+ /* void (*)(int a, ...) */
+ BTF_FUNC_PROTO_ENC(0, 2), /* [2] */
+ BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1),
+ BTF_FUNC_PROTO_ARG_ENC(0, 0),
+ /* void func(int a, ...) */
+ BTF_TYPE_ENC(NAME_TBD, /* [3] */
+ BTF_INFO_ENC(BTF_KIND_FUNC, 0, BTF_FUNC_GLOBAL), 2),
+ BTF_END_RAW,
+ },
+ .str_sec = "\0a\0func",
+ .str_sec_size = sizeof("\0a\0func"),
+ .map_type = BPF_MAP_TYPE_ARRAY,
+ .map_name = "func_type_check_btf",
+ .key_size = sizeof(int),
+ .value_size = sizeof(int),
+ .key_type_id = 1,
+ .value_type_id = 1,
+ .max_entries = 4,
+},
+
+{
.descr = "func (Non zero vlen)",
.raw_types = {
BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */
@@ -3585,15 +3634,28 @@ static struct btf_raw_test raw_tests[] = {
.btf_load_err = true,
},
{
- .descr = "type name '?foo' is not ok",
+ .descr = "type name '?foo' is ok",
.raw_types = {
/* union ?foo; */
BTF_TYPE_ENC(1, BTF_INFO_ENC(BTF_KIND_FWD, 1, 0), 0), /* [1] */
BTF_END_RAW,
},
BTF_STR_SEC("\0?foo"),
- .err_str = "Invalid name",
- .btf_load_err = true,
+},
+{
+ .descr = "names of Rust types and functions are ok",
+ .raw_types = {
+ BTF_TYPE_INT_ENC(NAME_NTH(1), 0, 0, 32, 4), /* [1] */
+ BTF_STRUCT_ENC(NAME_NTH(2), 1, 4), /* [2] */
+ BTF_MEMBER_ENC(NAME_NTH(3), 1, 0),
+ BTF_FWD_ENC(NAME_NTH(4), 0), /* [3] */
+ BTF_TYPEDEF_ENC(NAME_NTH(5), 2), /* [4] */
+ BTF_FUNC_PROTO_ENC(0, 1), /* [5] */
+ BTF_FUNC_PROTO_ARG_ENC(NAME_NTH(6), 1),
+ BTF_FUNC_ENC(NAME_NTH(7), 5), /* [6] */
+ BTF_END_RAW,
+ },
+ BTF_STR_SEC("\0u32\0Option<&str>\0__0\0*const str\0{impl#9}<[u8; 4]>\0self\0fmt<str>"),
},
{
diff --git a/tools/testing/selftests/bpf/prog_tests/btf_rust.c b/tools/testing/selftests/bpf/prog_tests/btf_rust.c
new file mode 100644
index 000000000000..daf777cadfda
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/btf_rust.c
@@ -0,0 +1,142 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <test_progs.h>
+#include <bpf/btf.h>
+#include "bpftool_helpers.h"
+
+#define FUNC_NAME "write_fmt<scx_simple::BpfStream>"
+#define KSYM_NAME "write_fmt_scx_simple__BpfStream_"
+
+/* The name of a function of Rust is a part of the name of the program in kallsyms */
+static void test_func_name(void)
+{
+ struct bpf_insn insns[] = {
+ BPF_MOV64_IMM(BPF_REG_0, 0),
+ BPF_EXIT_INSN(),
+ };
+ LIBBPF_OPTS(bpf_prog_load_opts, opts);
+ int int_id, proto_id, prog_fd = -1, i;
+ struct bpf_func_info func_info = {};
+ struct bpf_prog_info info = {};
+ unsigned long long addr;
+ __u32 len = sizeof(info);
+ char sym[128], *p = sym;
+ struct btf *btf;
+
+ btf = btf__new_empty();
+ if (!ASSERT_OK_PTR(btf, "btf"))
+ return;
+ int_id = btf__add_int(btf, "i32", 4, BTF_INT_SIGNED);
+ ASSERT_GT(int_id, 0, "int");
+ proto_id = btf__add_func_proto(btf, int_id);
+ ASSERT_GT(proto_id, 0, "proto");
+ ASSERT_OK(btf__add_func_param(btf, "ctx", int_id), "param");
+ func_info.type_id = btf__add_func(btf, FUNC_NAME, BTF_FUNC_STATIC, proto_id);
+ ASSERT_GT(func_info.type_id, 0, "func");
+ if (!ASSERT_OK(btf__load_into_kernel(btf), "btf load"))
+ goto out;
+
+ opts.prog_btf_fd = btf__fd(btf);
+ opts.func_info = &func_info;
+ opts.func_info_cnt = 1;
+ opts.func_info_rec_size = sizeof(func_info);
+ prog_fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, NULL, "GPL", insns,
+ ARRAY_SIZE(insns), &opts);
+ if (!ASSERT_GE(prog_fd, 0, "prog load"))
+ goto out;
+ if (!ASSERT_OK(bpf_prog_get_info_by_fd(prog_fd, &info, &len), "prog info"))
+ goto out;
+ if (!info.jited_prog_len) {
+ test__skip();
+ goto out;
+ }
+
+ p += sprintf(p, "bpf_prog_");
+ for (i = 0; i < BPF_TAG_SIZE; i++)
+ p += sprintf(p, "%02x", info.tag[i]);
+ sprintf(p, "_%s", KSYM_NAME);
+ ASSERT_OK(kallsyms_find(sym, &addr), sym);
+out:
+ if (prog_fd >= 0)
+ close(prog_fd);
+ btf__free(btf);
+}
+
+#define PIN_PATH "/sys/fs/bpf/btf_rust_piece"
+
+/*
+ * A piece of a static that LLVM split has the type of the whole static.
+ * It's the last variable in the section, so its type ends past the map value.
+ */
+static void test_piece(void)
+{
+ LIBBPF_OPTS(bpf_map_create_opts, opts);
+ int int_id, struct_id, var_id, piece_id, sec_id, map_fd = -1, key = 0;
+ __u32 value[2] = { 0x11111111, 0x22222222 };
+ char line[256] = {}, out[1024] = {};
+ struct btf *btf;
+ FILE *f = NULL;
+
+ btf = btf__new_empty();
+ if (!ASSERT_OK_PTR(btf, "btf"))
+ return;
+ int_id = btf__add_int(btf, "u32", 4, 0);
+ ASSERT_GT(int_id, 0, "int");
+ struct_id = btf__add_struct(btf, "Whole", 8);
+ ASSERT_GT(struct_id, 0, "struct");
+ ASSERT_OK(btf__add_field(btf, "a", int_id, 0, 0), "field");
+ ASSERT_OK(btf__add_field(btf, "b", int_id, 32, 0), "field");
+ var_id = btf__add_var(btf, "CNT", BTF_VAR_STATIC, int_id);
+ ASSERT_GT(var_id, 0, "var");
+ piece_id = btf__add_var(btf, "WHOLE.1", BTF_VAR_STATIC, struct_id);
+ ASSERT_GT(piece_id, 0, "piece");
+ sec_id = btf__add_datasec(btf, ".bss", sizeof(value));
+ ASSERT_GT(sec_id, 0, "datasec");
+ ASSERT_OK(btf__add_datasec_var_info(btf, var_id, 0, 4), "var info");
+ ASSERT_OK(btf__add_datasec_var_info(btf, piece_id, 4, 4), "piece info");
+ if (!ASSERT_OK(btf__load_into_kernel(btf), "btf load"))
+ goto out;
+
+ opts.btf_fd = btf__fd(btf);
+ opts.btf_value_type_id = sec_id;
+ map_fd = bpf_map_create(BPF_MAP_TYPE_ARRAY, ".bss", sizeof(key), sizeof(value), 1, &opts);
+ if (!ASSERT_GE(map_fd, 0, "map create"))
+ goto out;
+ if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, value, 0), "map update"))
+ goto out;
+
+ /* the variable is printed, the piece is not */
+ unlink(PIN_PATH);
+ if (!ASSERT_OK(bpf_obj_pin(map_fd, PIN_PATH), "pin"))
+ goto out;
+ f = fopen(PIN_PATH, "r");
+ if (!ASSERT_OK_PTR(f, "open"))
+ goto out;
+ while (fgets(line, sizeof(line), f) && line[0] == '#')
+ ;
+ ASSERT_HAS_SUBSTR(line, "286331153", "var");
+ ASSERT_NULL(strstr(line, "572662306"), "piece");
+
+ /* the same for bpftool */
+ if (!ASSERT_OK(get_bpftool_command_output("map dump pinned " PIN_PATH, out, sizeof(out)),
+ "bpftool"))
+ goto out;
+ ASSERT_HAS_SUBSTR(out, "CNT", "var");
+ ASSERT_NULL(strstr(out, "WHOLE.1"), "piece");
+out:
+ if (f)
+ fclose(f);
+ unlink(PIN_PATH);
+ if (map_fd >= 0)
+ close(map_fd);
+ btf__free(btf);
+}
+
+/* Serial: programs are not in kallsyms while another test sets bpf_jit_harden */
+void serial_test_btf_rust(void)
+{
+ if (test__start_subtest("func_name"))
+ test_func_name();
+ if (test__start_subtest("piece"))
+ test_piece();
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/data_in_arena.c b/tools/testing/selftests/bpf/prog_tests/data_in_arena.c
new file mode 100644
index 000000000000..cb1023507c01
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/data_in_arena.c
@@ -0,0 +1,216 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <test_progs.h>
+#include "data_in_arena.skel.h"
+#include "data_in_arena_decl.skel.h"
+#include "data_in_arena_fail.skel.h"
+
+static int run_prog(struct bpf_program *prog)
+{
+ LIBBPF_OPTS(bpf_test_run_opts, topts);
+
+ if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(prog), &topts), "test_run"))
+ return -1;
+ return topts.retval;
+}
+
+static void run(struct data_in_arena *skel, int counter)
+{
+ int i;
+
+ ASSERT_EQ(run_prog(skel->progs.use_data), counter + 7 + 1 + 2, "retval");
+ ASSERT_EQ(skel->bss->sum, counter + 7 + 1 + 2, "sum");
+ ASSERT_EQ(skel->data->counter, counter + 1, "counter");
+ ASSERT_EQ(skel->data->pair[1], 5, "pair[1]");
+ for (i = 0; i < 4; i++)
+ ASSERT_EQ(skel->bss->table[i], 10 * (i + 1) + i, "table");
+}
+
+static void test_in_arena(void)
+{
+ struct bpf_map_info info = {};
+ struct data_in_arena *skel;
+ __u32 len = sizeof(info);
+ struct bpf_map *arena;
+ size_t sz;
+
+ skel = data_in_arena__open();
+ if (!ASSERT_OK_PTR(skel, "open"))
+ return;
+
+ arena = bpf_object__find_map_by_name(skel->obj, "arena");
+ if (!ASSERT_OK_PTR(arena, "arena"))
+ goto out;
+ ASSERT_EQ(bpf_map__type(arena), BPF_MAP_TYPE_ARENA, "arena type");
+ ASSERT_EQ(bpf_map__max_entries(arena), 1, "arena pages");
+ ASSERT_OK(bpf_map__set_max_entries(arena, 8), "arena resize");
+ ASSERT_FALSE(bpf_map__autocreate(skel->maps.data), "data autocreate");
+ ASSERT_FALSE(bpf_map__autocreate(skel->maps.bss), "bss autocreate");
+ ASSERT_FALSE(bpf_map__autocreate(skel->maps.rodata), "rodata autocreate");
+ ASSERT_EQ(bpf_map__set_autocreate(skel->maps.data, true), -EOPNOTSUPP, "set_autocreate");
+ ASSERT_EQ(bpf_map__set_value_size(skel->maps.bss, 4096), -EOPNOTSUPP, "set_value_size");
+ ASSERT_EQ(bpf_map__initial_value(skel->maps.data, &sz), skel->data, "initial_value");
+ ASSERT_EQ(sz, sizeof(*skel->data), "initial_value size");
+
+ /* initial values are set the usual way */
+ skel->data->counter = 100;
+
+ if (!ASSERT_OK(data_in_arena__load(skel), "load"))
+ goto out;
+ /* there are no maps behind the sections */
+ ASSERT_ERR(bpf_map_get_info_by_fd(bpf_map__fd(skel->maps.data), &info, &len), "data map");
+ ASSERT_ERR(bpf_map_get_info_by_fd(bpf_map__fd(skel->maps.bss), &info, &len), "bss map");
+ ASSERT_ERR(bpf_map_get_info_by_fd(bpf_map__fd(skel->maps.rodata), &info, &len),
+ "rodata map");
+ run(skel, 100);
+
+ /* pointers to data next to pointers to functions */
+ ASSERT_EQ(run_prog(skel->progs.use_ops), 42 + 1 + 'e', "use_ops");
+ ASSERT_EQ(skel->data->counter, 102, "counter");
+
+ /* alignment of sections and pointers to data in data */
+ ASSERT_EQ((unsigned long)&skel->bss->aligned64 % 64, 0, "alignment");
+ ASSERT_EQ(run_prog(skel->progs.use_ptrs), 0, "use_ptrs");
+ ASSERT_EQ(skel->data->x, 43, "x");
+ ASSERT_EQ(skel->bss->aligned64.v[7], 7, "aligned64");
+ ASSERT_EQ(skel->data->px, &skel->data->x, "px");
+
+ /* format strings of bpf_printk() and BPF_SNPRINTF() */
+ ASSERT_EQ(run_prog(skel->progs.use_printk), sizeof("43-7"), "use_printk");
+ ASSERT_STREQ(skel->bss->out, "43-7", "out");
+out:
+ data_in_arena__destroy(skel);
+}
+
+/* The object has an arena map and __arena variables */
+static void test_declared_arena(void)
+{
+ struct data_in_arena_decl *skel;
+ struct bpf_map *map;
+ int arenas = 0;
+
+ skel = data_in_arena_decl__open();
+ if (!ASSERT_OK_PTR(skel, "open"))
+ return;
+ bpf_object__for_each_map(map, skel->obj)
+ arenas += bpf_map__type(map) == BPF_MAP_TYPE_ARENA;
+ ASSERT_EQ(arenas, 1, "no second arena");
+ skel->data->counter = 6;
+ if (!ASSERT_OK(data_in_arena_decl__load(skel), "load"))
+ goto out;
+ ASSERT_EQ(run_prog(skel->progs.use_data), 6 + 7 + 11, "retval");
+ ASSERT_EQ(run_prog(skel->progs.use_data), 7 + 7 + 12, "retval");
+ ASSERT_EQ(skel->bss->sum, 7 + 7 + 12, "sum");
+ ASSERT_EQ(skel->data->counter, 8, "counter");
+out:
+ data_in_arena_decl__destroy(skel);
+}
+
+/* A pointer in data that can't be made an address of arena fails the load */
+static void test_ptr_to_map(void)
+{
+ struct data_in_arena_fail *skel;
+
+ skel = data_in_arena_fail__open();
+ if (!ASSERT_OK_PTR(skel, "open"))
+ return;
+ ASSERT_ERR(data_in_arena_fail__load(skel), "load");
+ data_in_arena_fail__destroy(skel);
+}
+
+/* So does a pointer to a variable of the kernel. There is no skeleton: the open fails. */
+static void test_ptr_to_extern(void)
+{
+ struct bpf_object *obj;
+
+ obj = bpf_object__open_file("./data_in_arena_extern.bpf.o", NULL);
+ if (!ASSERT_ERR_PTR(obj, "open"))
+ bpf_object__close(obj);
+}
+
+/* __arena variables and no arena map. There is no skeleton: the open fails. */
+static void test_arena_var_no_map(void)
+{
+ struct bpf_object *obj;
+
+ obj = bpf_object__open_file("./data_in_arena_nomap.bpf.o", NULL);
+ if (!ASSERT_ERR_PTR(obj, "open"))
+ bpf_object__close(obj);
+}
+
+/* The address of an arena that libbpf doesn't create is not known. No pointers in data then. */
+static void test_not_my_arena(bool pin)
+{
+ LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = BPF_F_MMAPABLE);
+ struct data_in_arena *skel;
+ struct bpf_map *arena;
+ int fd = -1;
+
+ skel = data_in_arena__open();
+ if (!ASSERT_OK_PTR(skel, "open"))
+ return;
+ arena = bpf_object__find_map_by_name(skel->obj, "arena");
+ if (!ASSERT_OK_PTR(arena, "arena"))
+ goto out;
+ if (pin) {
+ ASSERT_OK(bpf_map__set_pin_path(arena, "/sys/fs/bpf/data_in_arena"), "pin_path");
+ } else {
+ fd = bpf_map_create(BPF_MAP_TYPE_ARENA, "arena", 0, 0, 1, &opts);
+ if (!ASSERT_GE(fd, 0, "map_create"))
+ goto out;
+ ASSERT_OK(bpf_map__reuse_fd(arena, fd), "reuse_fd");
+ }
+ ASSERT_EQ(data_in_arena__load(skel), -ENOTSUP, "load");
+out:
+ if (fd >= 0)
+ close(fd);
+ data_in_arena__destroy(skel);
+}
+
+/* The object is there if rustc and clang can build it, see Makefile.buildvars */
+static void test_rust(void)
+{
+ const char *file = "./data_in_arena_rust.bpf.o";
+ struct bpf_object *obj;
+
+ if (access(file, R_OK)) {
+ test__skip();
+ return;
+ }
+ obj = bpf_object__open_file(file, NULL);
+ if (!ASSERT_OK_PTR(obj, "open"))
+ return;
+ if (!ASSERT_OK(bpf_object__load(obj), "load"))
+ goto out;
+ /* libbpf goes on without BTF when the kernel doesn't take it */
+ ASSERT_GE(bpf_object__btf_fd(obj), 0, "btf_fd");
+ ASSERT_EQ(run_prog(bpf_object__find_program_by_name(obj, "list_in_data")),
+ 100 + 20 + 3, "retval");
+out:
+ bpf_object__close(obj);
+}
+
+void test_data_in_arena(void)
+{
+#if !defined(__x86_64__) && !defined(__aarch64__)
+ /* other JITs don't take BPF_F_ARENA_SCALAR */
+ test__skip();
+ return;
+#endif
+ if (test__start_subtest("arena"))
+ test_in_arena();
+ if (test__start_subtest("declared_arena"))
+ test_declared_arena();
+ if (test__start_subtest("ptr_to_map"))
+ test_ptr_to_map();
+ if (test__start_subtest("ptr_to_extern"))
+ test_ptr_to_extern();
+ if (test__start_subtest("arena_var_no_map"))
+ test_arena_var_no_map();
+ if (test__start_subtest("pinned_arena"))
+ test_not_my_arena(true);
+ if (test__start_subtest("reused_arena"))
+ test_not_my_arena(false);
+ if (test__start_subtest("rust"))
+ test_rust();
+}
diff --git a/tools/testing/selftests/bpf/prog_tests/verifier.c b/tools/testing/selftests/bpf/prog_tests/verifier.c
index 8a6d341b754a..460ad10ddc02 100644
--- a/tools/testing/selftests/bpf/prog_tests/verifier.c
+++ b/tools/testing/selftests/bpf/prog_tests/verifier.c
@@ -12,6 +12,7 @@
#include "verifier_and.skel.h"
#include "verifier_arena.skel.h"
#include "verifier_arena_large.skel.h"
+#include "verifier_arena_scalar.skel.h"
#include "verifier_arena_globals1.skel.h"
#include "verifier_arena_globals2.skel.h"
#include "verifier_array_access.skel.h"
@@ -198,6 +199,7 @@ void test_verifier_align(void) { RUN(verifier_align); }
void test_verifier_and(void) { RUN(verifier_and); }
void test_verifier_arena(void) { RUN(verifier_arena); }
void test_verifier_arena_large(void) { RUN(verifier_arena_large); }
+void test_verifier_arena_scalar(void) { RUN(verifier_arena_scalar); }
void test_verifier_arena_globals1(void) { RUN(verifier_arena_globals1); }
void test_verifier_arena_globals2(void) { RUN(verifier_arena_globals2); }
void test_verifier_basic_stack(void) { RUN(verifier_basic_stack); }
diff --git a/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c b/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c
new file mode 100644
index 000000000000..8217f4c4c00c
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c
@@ -0,0 +1,76 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include "bpf_experimental.h"
+#include "bpf_qdisc_common.h"
+#include "bpf_misc.h"
+
+char _license[] SEC("license") = "GPL";
+
+int proto;
+
+static __noinline struct ethhdr *slice_in_subprog(struct sk_buff *skb)
+{
+ struct bpf_dynptr ptr;
+
+ bpf_dynptr_from_skb((struct __sk_buff *)skb, 0, &ptr);
+ return bpf_dynptr_slice(&ptr, 0, NULL, sizeof(struct ethhdr));
+}
+
+SEC("struct_ops")
+__failure __msg("invalid mem access 'scalar'")
+int BPF_PROG(invalid_dynptr_returned_slice, struct sk_buff *skb,
+ struct Qdisc *sch, struct bpf_sk_buff_ptr *to_free)
+{
+ struct ethhdr *hdr;
+
+ hdr = slice_in_subprog(skb);
+ if (!hdr) {
+ bpf_qdisc_skb_drop(skb, to_free);
+ return NET_XMIT_DROP;
+ }
+
+ /* this should fail */
+ proto = hdr->h_proto;
+
+ bpf_qdisc_skb_drop(skb, to_free);
+
+ return NET_XMIT_DROP;
+}
+
+SEC("struct_ops")
+__auxiliary
+struct sk_buff *BPF_PROG(bpf_qdisc_test_dequeue, struct Qdisc *sch)
+{
+ return NULL;
+}
+
+SEC("struct_ops")
+__auxiliary
+int BPF_PROG(bpf_qdisc_test_init, struct Qdisc *sch, struct nlattr *opt,
+ struct netlink_ext_ack *extack)
+{
+ return 0;
+}
+
+SEC("struct_ops")
+__auxiliary
+void BPF_PROG(bpf_qdisc_test_reset, struct Qdisc *sch)
+{
+}
+
+SEC("struct_ops")
+__auxiliary
+void BPF_PROG(bpf_qdisc_test_destroy, struct Qdisc *sch)
+{
+}
+
+SEC(".struct_ops")
+struct Qdisc_ops test = {
+ .enqueue = (void *)invalid_dynptr_returned_slice,
+ .dequeue = (void *)bpf_qdisc_test_dequeue,
+ .init = (void *)bpf_qdisc_test_init,
+ .reset = (void *)bpf_qdisc_test_reset,
+ .destroy = (void *)bpf_qdisc_test_destroy,
+ .id = "bpf_qdisc_test",
+};
diff --git a/tools/testing/selftests/bpf/progs/data_in_arena.c b/tools/testing/selftests/bpf/progs/data_in_arena.c
new file mode 100644
index 000000000000..18f38103c0fc
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/data_in_arena.c
@@ -0,0 +1,112 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+
+/* global data of the object is in arena */
+char data_in_arena SEC(".arena.data");
+
+int counter = 5;
+long pair[2] = { 1, 2 };
+int x = 42;
+long sum;
+long table[4];
+struct {
+ long v[8];
+} aligned64 __attribute__((aligned(64)));
+const volatile int ro = 7;
+const volatile long ro_table[4] = { 10, 20, 30, 40 };
+
+/* const strings stay in a map for helpers and kfuncs, a copy of them is in arena */
+const char hello[] SEC(".rodata.str.hello") = "hello";
+
+/* pointers to data that are stored in data */
+int *px = &x;
+const char *str = hello;
+int *const volatile cpx SEC(".data.rel.ro") = &x;
+
+SEC("syscall")
+int use_data(void *ctx)
+{
+ int i;
+
+ for (i = 0; i < 4; i++)
+ table[i] = ro_table[i] + i;
+ sum = counter + ro + pair[0] + pair[1];
+ counter++;
+ __sync_fetch_and_add(&pair[1], 3);
+ return sum;
+}
+
+typedef int (*op_fn)(int);
+
+static __noinline int add1(int v)
+{
+ return v + 1;
+}
+
+/*
+ * Pointers to functions and to data in read-only data of a program with callx.
+ * Volatile, so that the compiler doesn't replace the pointers with what
+ * they point to.
+ */
+static const volatile struct {
+ op_fn fn;
+ int *data;
+ const char *name;
+} ops SEC(".data.rel.ro") = { add1, &x, hello };
+
+SEC("syscall")
+int use_ops(void *ctx)
+{
+ /* a program has the arena when its code refers to it */
+ counter++;
+#ifdef __clang__
+ return ops.fn(*ops.data) + ops.name[1];
+#else
+ /* gcc doesn't support indirect calls */
+ return add1(*ops.data) + ops.name[1];
+#endif
+}
+
+SEC("syscall")
+int use_ptrs(void *ctx)
+{
+ unsigned long addr = (unsigned long)&aligned64;
+ char local[4] = "abc";
+
+ /* hide the address from the compiler, it knows that '& 63' is 0 */
+ asm volatile ("" : "+r"(addr));
+ if (addr & 63)
+ return 1;
+ aligned64.v[7] = 7;
+ if (*px != 42)
+ return 2;
+ *px = 43;
+ if (x != 43 || *cpx != 43)
+ return 3;
+ if (str[0] != 'h' || str[4] != 'o' || str[5])
+ return 4;
+ /* the literal is in a map */
+ if (bpf_strncmp(local, sizeof(local), "abc"))
+ return 5;
+ return 0;
+}
+
+char out[16];
+
+/* format strings are in a map */
+SEC("syscall")
+int use_printk(void *ctx)
+{
+ char buf[sizeof(out)];
+ int i, n;
+
+ bpf_printk("counter %d", counter);
+ n = BPF_SNPRINTF(buf, sizeof(buf), "%d-%d", x, ro);
+ for (i = 0; i < sizeof(out); i++)
+ out[i] = buf[i];
+ return n;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_decl.c b/tools/testing/selftests/bpf/progs/data_in_arena_decl.c
new file mode 100644
index 000000000000..01bc4fea8f0c
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/data_in_arena_decl.c
@@ -0,0 +1,37 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#define BPF_NO_KFUNC_PROTOTYPES
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_experimental.h"
+#include <bpf_arena_common.h>
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARENA);
+ __uint(map_flags, BPF_F_MMAPABLE);
+ __uint(max_entries, 4);
+} arena SEC(".maps");
+
+/* global data of the object is in arena */
+char data_in_arena SEC(".arena.data");
+
+int counter = 5;
+long sum;
+const volatile int ro = 7;
+
+#if defined(__BPF_FEATURE_ADDR_SPACE_CAST)
+int __arena avar = 11;
+#else
+int avar = 11;
+#endif
+
+SEC("syscall")
+int use_data(void *ctx)
+{
+ sum = counter + ro + avar;
+ counter++;
+ avar++;
+ return sum;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_extern.c b/tools/testing/selftests/bpf/progs/data_in_arena_extern.c
new file mode 100644
index 000000000000..dc1ad5ca3bdd
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/data_in_arena_extern.c
@@ -0,0 +1,20 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+
+/* global data of the object is in arena */
+char data_in_arena SEC(".arena.data");
+
+extern const int bpf_prog_active __ksym;
+
+/* the variable of the kernel is not in arena */
+const void *kp = &bpf_prog_active;
+
+SEC("syscall")
+int ptr_to_extern(void *ctx)
+{
+ return *(int *)kp;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_fail.c b/tools/testing/selftests/bpf/progs/data_in_arena_fail.c
new file mode 100644
index 000000000000..ad8afe9afc96
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/data_in_arena_fail.c
@@ -0,0 +1,20 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+
+/* global data of the object is in arena */
+char data_in_arena SEC(".arena.data");
+
+int x = 42;
+/* the table is read-only data with pointers. It's not in arena. */
+int *const volatile tbl[1] SEC(".data.rel.ro") = { &x };
+int *const volatile *pp = tbl;
+
+SEC("syscall")
+int ptr_to_map(void *ctx)
+{
+ return **pp;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_nomap.c b/tools/testing/selftests/bpf/progs/data_in_arena_nomap.c
new file mode 100644
index 000000000000..ccf66321eb36
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/data_in_arena_nomap.c
@@ -0,0 +1,20 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+#include "bpf_arena_common.h"
+
+/* global data of the object is in arena */
+char data_in_arena SEC(".arena.data");
+
+int counter = 5;
+/* needs an arena map that is declared. The one that libbpf creates won't do. */
+int __arena_global avar = 11;
+
+SEC("syscall")
+int arena_var(void *ctx)
+{
+ return avar + counter;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_rust.rs b/tools/testing/selftests/bpf/progs/data_in_arena_rust.rs
new file mode 100644
index 000000000000..3dd9d4208232
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/data_in_arena_rust.rs
@@ -0,0 +1,73 @@
+// SPDX-License-Identifier: GPL-2.0
+
+// Why .data, .bss and .rodata of a program in Rust are in arena.
+//
+// A reference in Rust is an address. It doesn't say what it points to and it
+// can be stored in data. The list below has a node in each of the sections.
+// The nodes are linked by references that are in the data:
+// IN_BSS.next is stored by the program,
+// IN_DATA.next is a relocation in .data against .rodata.
+// sum() loads the references back and reads the three nodes with the same insn.
+//
+// When the sections are array maps libbpf skips the relocation in .data, and
+// what sum() loads from a node is a number that can't be dereferenced:
+// R1 invalid mem access 'scalar'
+// In arena the address of a node is a number to begin with.
+
+#![no_std]
+#![no_main]
+
+// Tell libbpf to keep .data, .bss and .rodata in arena.
+#[used]
+#[link_section = ".arena.data"]
+static DATA_IN_ARENA: u8 = 0;
+
+#[used]
+#[link_section = "license"]
+static LICENSE: [u8; 4] = *b"GPL\0";
+
+// panic=abort is a stop gap until panic=unwind is supported.
+// Nothing here panics, so the handler is not a part of the program.
+#[panic_handler]
+fn panic(_info: &core::panic::PanicInfo) -> ! {
+ loop {}
+}
+
+pub struct Node {
+ val: u32,
+ next: Option<&'static Node>,
+}
+
+// no_mangle makes them visible outside, so LLVM can't fold the list into a constant.
+#[no_mangle]
+static IN_RODATA: Node = Node { val: 3, next: None };
+#[no_mangle]
+static mut IN_DATA: Node = Node {
+ val: 20,
+ next: Some(&IN_RODATA),
+};
+#[no_mangle]
+static mut IN_BSS: Node = Node { val: 0, next: None };
+
+#[inline(never)]
+fn sum(mut node: Option<&Node>) -> u32 {
+ let mut sum = 0;
+ // The verifier wants a bound.
+ for _ in 0..8 {
+ let Some(n) = node else { break };
+ sum += n.val;
+ node = n.next;
+ }
+ sum
+}
+
+#[no_mangle]
+#[link_section = "syscall"]
+pub extern "C" fn list_in_data(_ctx: *mut u8) -> u32 {
+ unsafe {
+ let head = &mut *&raw mut IN_BSS;
+ head.val = 100;
+ head.next = Some(&*&raw const IN_DATA);
+ sum(Some(head))
+ }
+}
diff --git a/tools/testing/selftests/bpf/progs/dynptr_fail.c b/tools/testing/selftests/bpf/progs/dynptr_fail.c
index 148cf4417322..c2247b38c849 100644
--- a/tools/testing/selftests/bpf/progs/dynptr_fail.c
+++ b/tools/testing/selftests/bpf/progs/dynptr_fail.c
@@ -125,9 +125,9 @@ static int missing_release_callback_fn(__u32 index, void *data)
return 0;
}
-/* Any dynptr initialized within a callback must have bpf_dynptr_put called */
+/* A callback cannot return with the last dynptr for a referenced resource. */
SEC("?raw_tp")
-__failure __msg("Unreleased reference id")
+__failure __msg("cannot overwrite referenced dynptr")
int ringbuf_missing_release_callback(void *ctx)
{
bpf_loop(10, missing_release_callback_fn, NULL, 0);
@@ -1895,6 +1895,101 @@ int clone_invalidate4(void *ctx)
return 0;
}
+static __noinline void clone_slice_in_subprog(struct bpf_dynptr *ptr, int **data)
+{
+ struct bpf_dynptr clone;
+
+ bpf_dynptr_clone(ptr, &clone);
+ *data = bpf_dynptr_data(&clone, 0, sizeof(val));
+}
+
+static __noinline void caller_slice_in_subprog(struct bpf_dynptr *ptr, int **data)
+{
+ struct bpf_dynptr clone;
+
+ *data = bpf_dynptr_data(ptr, 0, sizeof(val));
+ bpf_dynptr_clone(ptr, &clone);
+}
+
+static __noinline void reserve_dynptr_in_subprog(void)
+{
+ struct bpf_dynptr ptr;
+
+ bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr);
+}
+
+/* A subprogram cannot lose the last dynptr that can release a resource. */
+SEC("?raw_tp")
+__failure __msg("cannot overwrite referenced dynptr")
+int referenced_dynptr_lost_on_subprog_return(void *ctx)
+{
+ reserve_dynptr_in_subprog();
+
+ return 0;
+}
+
+/*
+ * Destroying a callee-local clone on return must not invalidate a slice whose
+ * source dynptr belongs to the caller.
+ */
+SEC("?raw_tp")
+__success
+int caller_dynptr_slice_across_subprog_valid(void *ctx)
+{
+ struct bpf_dynptr ptr;
+ int *data = NULL;
+
+ bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr);
+ caller_slice_in_subprog(&ptr, &data);
+ if (data)
+ *data = 123;
+ bpf_ringbuf_submit_dynptr(&ptr, 0);
+
+ return 0;
+}
+
+/*
+ * A slice derived from a callee-local clone is invalid after the subprogram
+ * returns.
+ */
+SEC("?raw_tp")
+__failure __msg("invalid mem access 'scalar'")
+int callee_dynptr_slice_invalid_after_return(void *ctx)
+{
+ struct bpf_dynptr ptr;
+ int *data = NULL;
+
+ bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr);
+ clone_slice_in_subprog(&ptr, &data);
+ if (data)
+ /* this should fail */
+ *data = 123;
+ bpf_ringbuf_submit_dynptr(&ptr, 0);
+
+ return 0;
+}
+
+/*
+ * A slice from a caller-owned dynptr survives the subprogram return, but
+ * releasing the shared reservation must invalidate it.
+ */
+SEC("?raw_tp")
+__failure __msg("invalid mem access 'scalar'")
+int caller_dynptr_slice_release_after_subprog_invalid(void *ctx)
+{
+ struct bpf_dynptr ptr;
+ int *data = NULL;
+
+ bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr);
+ caller_slice_in_subprog(&ptr, &data);
+ bpf_ringbuf_submit_dynptr(&ptr, 0);
+ if (data)
+ /* this should fail */
+ *data = 123;
+
+ return 0;
+}
+
/* Invalidating a dynptr should invalidate any data slices
* of its parent
*/
diff --git a/tools/testing/selftests/bpf/progs/test_siphash.h b/tools/testing/selftests/bpf/progs/test_siphash.h
index 5d3a7ec36780..9a85670a9dea 100644
--- a/tools/testing/selftests/bpf/progs/test_siphash.h
+++ b/tools/testing/selftests/bpf/progs/test_siphash.h
@@ -22,7 +22,7 @@ static inline u64 rol64(u64 word, unsigned int shift)
#define SIPHASH_CONST_2 0x6c7967656e657261ULL
#define SIPHASH_CONST_3 0x7465646279746573ULL
-/* lib/siphash.c */
+/* lib/crypto/siphash.c */
#define SIPROUND SIPHASH_PERMUTATION(v0, v1, v2, v3)
#define PREAMBLE(len) \
diff --git a/tools/testing/selftests/bpf/progs/verifier_arena_scalar.c b/tools/testing/selftests/bpf/progs/verifier_arena_scalar.c
new file mode 100644
index 000000000000..bebc37f501b7
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/verifier_arena_scalar.c
@@ -0,0 +1,912 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/bpf.h>
+#include <bpf/bpf_helpers.h>
+#include "../../../include/linux/filter.h"
+#include "bpf_misc.h"
+
+void *bpf_arena_alloc_pages(void *map, void *addr, __u32 page_cnt, int node_id,
+ __u64 flags) __ksym;
+
+#ifdef __TARGET_ARCH_arm64
+#define ARENA_VM_START (1ull << 32)
+#else
+#define ARENA_VM_START (1ull << 44)
+#endif
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARENA);
+ __uint(map_flags, BPF_F_MMAPABLE);
+ __uint(max_entries, 4);
+ __ulong(map_extra, ARENA_VM_START);
+} arena SEC(".maps");
+
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(max_entries, 1);
+ __type(key, int);
+ __type(value, long long);
+} hash SEC(".maps");
+
+/* JITs that take BPF_F_ARENA_SCALAR */
+#define __arena_scalar __flag(BPF_F_ARENA_SCALAR) __arch_x86_64 __arch_arm64
+
+/* BTF FUNC records are not generated for kfuncs referenced from inline assembly */
+void __kfunc_btf_root(void)
+{
+ bpf_arena_alloc_pages(0, 0, 0, 0, 0);
+}
+
+/* Tests start with r6 = address of a new page as the user space sees it, a number */
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: load and store of every size")
+__success __retval(0)
+__load_if_JITed()
+__naked void ld_st_sizes(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r7 = r6; \
+ r1 = 0x1122334455667788 ll; \
+ *(u64 *)(r6 + 0) = r1; \
+ *(u32 *)(r6 + 8) = r1; \
+ *(u16 *)(r6 + 12) = r1; \
+ *(u8 *)(r6 + 14) = r1; \
+ *(u64 *)(r6 + 16) = 0x1234; \
+ *(u32 *)(r6 + 24) = 0x5678; \
+ *(u16 *)(r6 + 28) = 0x9a; \
+ *(u8 *)(r6 + 30) = 0xbc; \
+ r0 = 1; \
+ r2 = *(u64 *)(r6 + 0); \
+ if r2 != r1 goto 9f; \
+ r0 = 2; \
+ r2 = *(u32 *)(r6 + 8); \
+ if r2 != 0x55667788 goto 9f; \
+ r0 = 3; \
+ r2 = *(u16 *)(r6 + 12); \
+ if r2 != 0x7788 goto 9f; \
+ r0 = 4; \
+ r2 = *(u8 *)(r6 + 14); \
+ if r2 != 0x88 goto 9f; \
+ r0 = 5; \
+ r2 = *(u64 *)(r6 + 16); \
+ if r2 != 0x1234 goto 9f; \
+ r0 = 6; \
+ r2 = *(u32 *)(r6 + 24); \
+ if r2 != 0x5678 goto 9f; \
+ r0 = 7; \
+ r2 = *(u16 *)(r6 + 28); \
+ if r2 != 0x9a goto 9f; \
+ r0 = 8; \
+ r2 = *(u8 *)(r6 + 30); \
+ if r2 != 0xbc goto 9f; \
+ /* the address is what it was */ \
+ r0 = 9; \
+ if r6 != r7 goto 9f; \
+ r0 = 10; \
+ r7 >>= 32; \
+ if r7 == 0 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: load into the register that holds the address")
+__success __retval(0)
+__load_if_JITed()
+__naked void ld_into_base(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r1 = 77; \
+ *(u64 *)(r6 + 0) = r1; \
+ r6 = *(u64 *)(r6 + 0); \
+ r0 = 1; \
+ if r6 != 77 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: sign extending load")
+__success __retval(0)
+__load_if_JITed()
+__naked void ldsx(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r7 = r6; \
+ *(u64 *)(r6 + 0) = 0x80; \
+ .8byte %[ldsx_insn]; /* r2 = *(s8 *)(r6 + 0) */ \
+ r0 = 1; \
+ if r2 != -128 goto 9f; \
+ r0 = 2; \
+ if r6 != r7 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(ldsx_insn, BPF_RAW_INSN(BPF_LDX | BPF_MEMSX | BPF_B,
+ BPF_REG_2, BPF_REG_6, 0, 0))
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: address loaded from arena")
+__success __retval(0)
+__load_if_JITed()
+__naked void ptr_chase(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ /* page[0] = &page[64]; page[64] = 5; */ \
+ r1 = r6; \
+ r1 += 64; \
+ *(u64 *)(r6 + 0) = r1; \
+ *(u64 *)(r1 + 0) = 5; \
+ r2 = *(u64 *)(r6 + 0); \
+ r0 = 1; \
+ if r2 != r1 goto 9f; \
+ r3 = *(u64 *)(r2 + 0); \
+ r0 = 2; \
+ if r3 != 5 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: atomics")
+__success __retval(0)
+__load_if_JITed()
+__naked void atomics(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r7 = r6; \
+ *(u64 *)(r6 + 8) = 1; \
+ r1 = 2; \
+ lock *(u64 *)(r6 + 8) += r1; \
+ r1 = 4; \
+ .8byte %[fetch_add_insn]; /* r1 = atomic_fetch_add((u64 *)(r6 + 8), r1) */ \
+ r0 = 1; \
+ if r1 != 3 goto 9f; \
+ r1 = 8; \
+ .8byte %[xchg_insn]; /* r1 = xchg_64(r6 + 8, r1) */ \
+ r0 = 2; \
+ if r1 != 7 goto 9f; \
+ r0 = 8; \
+ r1 = 16; \
+ .8byte %[cmpxchg_insn]; /* r0 = cmpxchg_64(r6 + 8, r0, r1) */ \
+ r2 = r0; \
+ r0 = 3; \
+ if r2 != 8 goto 9f; \
+ r2 = *(u64 *)(r6 + 8); \
+ r0 = 4; \
+ if r2 != 16 goto 9f; \
+ r0 = 5; \
+ if r6 != r7 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(fetch_add_insn, BPF_ATOMIC_OP(BPF_DW, BPF_ADD | BPF_FETCH,
+ BPF_REG_6, BPF_REG_1, 8)),
+ __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_6, BPF_REG_1, 8)),
+ __imm_insn(cmpxchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_CMPXCHG, BPF_REG_6, BPF_REG_1, 8))
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: cmpxchg through r0")
+__success __retval(0)
+__load_if_JITed()
+__naked void cmpxchg_r0(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ /* The address is in r0. The value at the address is not equal to it. */ \
+ *(u64 *)(r6 + 0) = 3; \
+ r0 = r6; \
+ r1 = 5; \
+ .8byte %[cmpxchg_insn]; /* r0 = cmpxchg_64(r0 + 0, r0, r1) */ \
+ r2 = r0; \
+ r0 = 1; \
+ if r2 != 3 goto 9f; \
+ r2 = *(u64 *)(r6 + 0); \
+ r0 = 2; \
+ if r2 != 3 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(cmpxchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_CMPXCHG, BPF_REG_0, BPF_REG_1, 0))
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: xchg into the register that holds the address")
+__success __retval(0)
+__load_if_JITed()
+__naked void xchg_into_base(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ *(u64 *)(r6 + 0) = 3; \
+ r1 = r6; \
+ .8byte %[xchg_insn]; /* r1 = xchg_64(r1 + 0, r1) */ \
+ r0 = 1; \
+ if r1 != 3 goto 9f; \
+ r2 = *(u64 *)(r6 + 0); \
+ r0 = 2; \
+ if r2 != r6 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_1, BPF_REG_1, 0))
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: xchg into the register that holds a pointer to stack")
+__failure __msg("misaligned access off (0x0; 0xffffffffffffffff)+0 size 8")
+__naked void xchg_into_stack_ptr(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r1 = 0; \
+ *(u64 *)(r10 - 8) = r1; \
+ r1 = r10; \
+ r1 += -8; \
+ .8byte %[xchg_insn]; /* r1 = xchg_64(r1 + 0, r1) */ \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_addr(arena),
+ __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_1, BPF_REG_1, 0))
+ : __clobber_all);
+}
+
+#ifdef CAN_USE_LOAD_ACQ_STORE_REL
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: load-acquire and store-release")
+__success __retval(0)
+__load_if_JITed()
+__naked void load_acq_store_rel(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r7 = r6; \
+ r1 = 0x1234; \
+ .8byte %[store_release_insn]; /* store_release((u64 *)(r6 + 8), r1) */ \
+ .8byte %[load_acquire_insn]; /* r2 = load_acquire((u64 *)(r6 + 8)) */ \
+ r0 = 1; \
+ if r2 != 0x1234 goto 9f; \
+ .8byte %[load_acquire8_insn]; /* w2 = load_acquire((u8 *)(r6 + 8)) */ \
+ r0 = 2; \
+ if r2 != 0x34 goto 9f; \
+ r0 = 3; \
+ if r6 != r7 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(store_release_insn,
+ BPF_ATOMIC_OP(BPF_DW, BPF_STORE_REL, BPF_REG_6, BPF_REG_1, 8)),
+ __imm_insn(load_acquire_insn,
+ BPF_ATOMIC_OP(BPF_DW, BPF_LOAD_ACQ, BPF_REG_2, BPF_REG_6, 8)),
+ __imm_insn(load_acquire8_insn,
+ BPF_ATOMIC_OP(BPF_B, BPF_LOAD_ACQ, BPF_REG_2, BPF_REG_6, 8))
+ : __clobber_all);
+}
+
+#endif /* CAN_USE_LOAD_ACQ_STORE_REL */
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: number that is not an address in arena")
+__success __retval(0)
+__load_if_JITed()
+__naked void not_in_arena(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ /* nothing is allocated: all loads read 0, stores are dropped */ \
+ r6 = 0xdeadbeef00000000 ll; \
+ r0 = 1; \
+ r2 = *(u64 *)(r6 + 0); \
+ if r2 != 0 goto 9f; \
+ r0 = 2; \
+ r2 = *(u8 *)(r6 - 32768); \
+ if r2 != 0 goto 9f; \
+ r6 = 0x12345678ffffffff ll; \
+ r0 = 3; \
+ r2 = *(u64 *)(r6 + 32760); \
+ if r2 != 0 goto 9f; \
+ *(u64 *)(r6 + 32760) = 1; \
+ *(u8 *)(r6 + 32767) = r2; \
+ r1 = 1; \
+ lock *(u64 *)(r6 + 32760) += r1; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: store through the address of the stack as a number")
+__success __retval(0)
+__load_if_JITed()
+__naked void stack_addr_as_number(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ *(u64 *)(r10 - 8) = 5; \
+ r6 = r10; \
+ r6 |= 0; \
+ /* r6 is a number now. The store goes to arena, not to the stack. */ \
+ *(u64 *)(r6 - 8) = 7; \
+ r1 = 9; \
+ *(u64 *)(r6 - 8) = r1; \
+ lock *(u64 *)(r6 - 8) += r1; \
+ r2 = *(u64 *)(r10 - 8); \
+ r0 = 1; \
+ if r2 != 5 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: 64-bit math on the address stays 64-bit")
+__success __retval(0)
+__load_if_JITed()
+__naked void alu64_after_access(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r2 = *(u64 *)(r6 + 0); \
+ r7 = r6; \
+ r7 += 8; \
+ r7 -= r6; \
+ r0 = 1; \
+ if r7 != 8 goto 9f; \
+ r7 = r6; \
+ r7 >>= 32; \
+ r0 = 2; \
+ if r7 == 0 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: st of an immediate changes no register")
+__success __retval(0)
+__load_if_JITed()
+__naked void st_keeps_regs(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r0 = r6; \
+ r1 = 0x1111; \
+ r2 = 0x2222; \
+ *(u64 *)(r0 + 0) = 5; \
+ r3 = r0; \
+ r0 = 1; \
+ if r3 != r6 goto 9f; \
+ r0 = 2; \
+ if r1 != 0x1111 goto 9f; \
+ r0 = 3; \
+ if r2 != 0x2222 goto 9f; \
+ r0 = 0x3333; \
+ r1 = r6; \
+ *(u32 *)(r1 + 8) = -7; \
+ r3 = r0; \
+ r0 = 4; \
+ if r3 != 0x3333 goto 9f; \
+ r0 = 5; \
+ if r1 != r6 goto 9f; \
+ r0 = 6; \
+ r3 = *(u64 *)(r6 + 0); \
+ if r3 != 5 goto 9f; \
+ r0 = 7; \
+ r3 = *(u64 *)(r6 + 8); \
+ r4 = 0xfffffff9 ll; \
+ if r3 != r4 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: access through a number with garbage in the upper half")
+__success __retval(0)
+__load_if_JITed()
+__naked void st_value_garbage(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r1 = 0xdeadbeef00001000 ll; \
+ *(u64 *)(r6 + 2048) = r1; \
+ r7 = *(u64 *)(r6 + 2048); \
+ r8 = *(u64 *)(r6 + 2048); \
+ r1 = 3; \
+ *(u64 *)(r8 + 0) = 1; \
+ r0 = 1; \
+ if r8 != r7 goto 9f; \
+ *(u8 *)(r8 + 1) = 1; \
+ r0 = 2; \
+ if r8 != r7 goto 9f; \
+ *(u32 *)(r8 + 4) = r1; \
+ r0 = 3; \
+ if r8 != r7 goto 9f; \
+ r2 = *(u16 *)(r8 + 2); \
+ r0 = 4; \
+ if r8 != r7 goto 9f; \
+ r0 = 5; \
+ if r1 != 3 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: access through a number without the upper half")
+__success __retval(0)
+__load_if_JITed()
+__naked void st_value_small(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r1 = 0x1000 ll; \
+ *(u64 *)(r6 + 2048) = r1; \
+ r7 = *(u64 *)(r6 + 2048); \
+ r8 = *(u64 *)(r6 + 2048); \
+ r1 = 3; \
+ *(u64 *)(r8 + 0) = 1; \
+ r0 = 1; \
+ if r8 != r7 goto 9f; \
+ *(u8 *)(r8 + 1) = 1; \
+ r0 = 2; \
+ if r8 != r7 goto 9f; \
+ *(u32 *)(r8 + 4) = r1; \
+ r0 = 3; \
+ if r8 != r7 goto 9f; \
+ r2 = *(u16 *)(r8 + 2); \
+ r0 = 4; \
+ if r8 != r7 goto 9f; \
+ r0 = 5; \
+ if r1 != 3 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: atomics through a number that is not an address in arena")
+__success __retval(0)
+__load_if_JITed()
+__naked void atomic_value(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r1 = 0xdeadbeef00001000 ll; \
+ *(u64 *)(r6 + 2048) = r1; \
+ r7 = *(u64 *)(r6 + 2048); \
+ r8 = *(u64 *)(r6 + 2048); \
+ r1 = 3; \
+ lock *(u64 *)(r8 + 0) += r1; \
+ r0 = 1; \
+ if r8 != r7 goto 9f; \
+ .8byte %[fetch_add_insn]; \
+ r0 = 2; \
+ if r8 != r7 goto 9f; \
+ .8byte %[xchg_insn]; \
+ r0 = 3; \
+ if r8 != r7 goto 9f; \
+ r0 = 0; \
+ .8byte %[cmpxchg_insn]; \
+ r0 = 4; \
+ if r8 != r7 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(fetch_add_insn, BPF_ATOMIC_OP(BPF_DW, BPF_ADD | BPF_FETCH,
+ BPF_REG_8, BPF_REG_1, 0)),
+ __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_8, BPF_REG_1, 0)),
+ __imm_insn(cmpxchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_CMPXCHG, BPF_REG_8, BPF_REG_1, 0))
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: number and pointer to arena at the same insn")
+__success __retval(0)
+__load_if_JITed()
+__naked void mixed_number_arena(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ *(u64 *)(r6 + 8) = 0x1234; \
+ /* a number that the verifier does not know, 0 at run time */ \
+ r7 = *(u64 *)(r6 + 16); \
+ r8 = r6; \
+ if r7 != 0 goto 1f; \
+ .8byte %[cast_kern_insn]; \
+1: *(u8 *)(r8 + 0) = 1; \
+ *(u8 *)(r8 + 1) = r7; \
+ r2 = *(u8 *)(r8 + 1); \
+ lock *(u64 *)(r8 + 24) += r7; \
+ r0 = 0; \
+ if r7 != 0 goto 9f; \
+ /* pointer to arena only: JIT adds all 64 bits of r8 to the base */ \
+ r0 = 2; \
+ r2 = *(u64 *)(r8 + 8); \
+ if r2 != 0x1234 goto 9f; \
+ r0 = 3; \
+ r2 = *(u8 *)(r6 + 0); \
+ if r2 != 1 goto 9f; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm_insn(cast_kern_insn, BPF_RAW_INSN(BPF_ALU64 | BPF_MOV | BPF_X,
+ BPF_REG_8, BPF_REG_8, 1, 1))
+ : __clobber_all);
+}
+
+SEC("syscall")
+__description("arena_scalar: no flag, no access through a number")
+__failure __msg("R6 invalid mem access 'scalar'")
+__naked void no_flag(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r6 = 0x100000000000 ll; \
+ r0 = *(u64 *)(r6 + 0); \
+ exit; \
+" :
+ : __imm_addr(arena)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: no arena, no access through a number")
+__failure __msg("R6 invalid mem access 'scalar'")
+__naked void no_arena(void)
+{
+ asm volatile (" \
+ r6 = 0x100000000000 ll; \
+ r0 = *(u64 *)(r6 + 0); \
+ exit; \
+" ::: __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: pointer that is NULL is an address in arena")
+__success __retval(0)
+__load_if_JITed()
+__naked void null_ptr(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r1 = 0; \
+ *(u32 *)(r10 - 4) = r1; \
+ r2 = r10; \
+ r2 += -4; \
+ r1 = %[hash] ll; \
+ call %[bpf_map_lookup_elem]; \
+ r1 = r0; \
+ r0 = 1; \
+ if r1 != 0 goto 9f; \
+ /* nothing is allocated: the load reads 0 */ \
+ r0 = *(u64 *)(r1 + 0); \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm_addr(hash),
+ __imm(bpf_map_lookup_elem)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: pointer that is NULL with an offset is an address in arena")
+__success __retval(0)
+__load_if_JITed()
+__naked void null_ptr_off(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r1 = 0; \
+ *(u32 *)(r10 - 4) = r1; \
+ r2 = r10; \
+ r2 += -4; \
+ r1 = %[hash] ll; \
+ call %[bpf_map_lookup_elem]; \
+ r1 = r0; \
+ r0 = 1; \
+ if r1 != 0 goto 9f; \
+ r1 += 8; \
+ *(u64 *)(r1 + 0) = 5; \
+ r0 = *(u64 *)(r1 + 0); \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm_addr(hash),
+ __imm(bpf_map_lookup_elem)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: number that is less than a page is an address in arena")
+__success __retval(0)
+__load_if_JITed()
+__naked void small_number(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ call %[bpf_get_prandom_u32]; \
+ r1 = r0; \
+ r1 &= 0xfff; \
+ r0 = *(u64 *)(r1 + 0); \
+ exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: number is not a pointer for a helper")
+__failure __msg("R2 type=scalar expected=")
+__naked void helper_arg(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ r1 = %[hash] ll; \
+ r2 = r6; \
+ call %[bpf_map_lookup_elem]; \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm_addr(hash),
+ __imm(bpf_arena_alloc_pages),
+ __imm(bpf_map_lookup_elem)
+ : __clobber_all);
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: number and pointer to stack at the same insn")
+__failure __msg("same insn cannot be used with different pointers")
+__load_if_JITed()
+__naked void mixed_number_stack(void)
+{
+ asm volatile (" \
+ r1 = %[arena] ll; \
+ r2 = 0; \
+ r3 = 1; \
+ r4 = -1; \
+ r5 = 0; \
+ call %[bpf_arena_alloc_pages]; \
+ r6 = r0; \
+ r0 = 100; \
+ if r6 == 0 goto 9f; \
+ *(u64 *)(r10 - 8) = 0; \
+ call %[bpf_get_prandom_u32]; \
+ if w0 != 0 goto 1f; \
+ r6 = r10; \
+ r6 += -8; \
+1: r0 = *(u64 *)(r6 + 0); \
+ r0 = 0; \
+9: exit; \
+" :
+ : __imm_addr(arena),
+ __imm(bpf_arena_alloc_pages),
+ __imm(bpf_get_prandom_u32)
+ : __clobber_all);
+}
+
+static int st_cb(__u64 idx, void *ctx)
+{
+ volatile long *p = *(volatile long **)ctx;
+
+ p[idx] = 7;
+ return 0;
+}
+
+SEC("syscall")
+__arena_scalar
+__description("arena_scalar: store through a number in a callback")
+__success __retval(0)
+__load_if_JITed()
+int st_in_callback(void *unused)
+{
+ volatile long *p = bpf_arena_alloc_pages(&arena, NULL, 1, -1, 0);
+
+ if (!p)
+ return 100;
+ bpf_loop(4, st_cb, &p, 0);
+ return p[0] + p[3] + p[4] - 14;
+}
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c b/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c
index 4d8273c258d5..31663338866d 100644
--- a/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c
+++ b/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c
@@ -22,8 +22,8 @@ struct {
SEC("socket")
__description("map element value illegal alu op, 1")
-__failure __msg("R0 bitwise operator &= on pointer")
-__failure_unpriv
+__failure __msg("R0 invalid mem access 'scalar'")
+__failure_unpriv __msg_unpriv("R0 bitwise operator &= on pointer")
__naked void value_illegal_alu_op_1(void)
{
asm volatile (" \
@@ -70,8 +70,8 @@ l0_%=: exit; \
SEC("socket")
__description("map element value illegal alu op, 3")
-__failure __msg("R0 pointer arithmetic with /= operator")
-__failure_unpriv
+__failure __msg("R0 invalid mem access 'scalar'")
+__failure_unpriv __msg_unpriv("R0 pointer arithmetic with /= operator")
__naked void value_illegal_alu_op_3(void)
{
asm volatile (" \
@@ -165,6 +165,171 @@ __naked void map_ptr_illegal_alu_op(void)
: __clobber_all);
}
+SEC("socket")
+__description("tag in the low bit of a pointer, and, shift")
+__success __retval(0)
+__failure_unpriv __msg_unpriv("R1 bitwise operator &= on pointer")
+__naked void ptr_tag_and_shift(void)
+{
+ asm volatile (" \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r1 = %[map_hash_48b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto l0_%=; \
+ r1 = r0; \
+ r1 &= 1; \
+ r2 = r0; \
+ r2 >>= 1; \
+ r3 = r0; \
+ r3 |= 1; \
+ r3 ^= 1; \
+ r0 = *(u32*)(r0 + 0); \
+ r0 = 0; \
+l0_%=: exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("tag in the low bit of a pointer, CAP_BPF without CAP_PERFMON")
+__success __retval(0)
+__failure_unpriv __msg_unpriv("R1 bitwise operator &= on pointer")
+__caps_unpriv(CAP_BPF)
+__naked void ptr_tag_cap_bpf(void)
+{
+ asm volatile (" \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r1 = %[map_hash_48b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto l0_%=; \
+ r1 = r0; \
+ r1 &= 1; \
+ r0 = 0; \
+l0_%=: exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("number op= pointer")
+__success __retval(0)
+__failure_unpriv __msg_unpriv("R1 pointer arithmetic with *= operator")
+__naked void number_mul_ptr(void)
+{
+ asm volatile (" \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r1 = %[map_hash_48b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto l0_%=; \
+ r1 = 7; \
+ r1 *= r0; \
+ r0 = 0; \
+l0_%=: exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("pointer with the tag cleared is a number")
+__failure __msg("R0 invalid mem access 'scalar'")
+__failure_unpriv __msg_unpriv("R0 bitwise operator |= on pointer")
+__naked void ptr_tag_cleared_deref(void)
+{
+ asm volatile (" \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r1 = %[map_hash_48b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto l0_%=; \
+ r0 |= 1; \
+ r0 ^= 1; \
+ r0 = *(u32*)(r0 + 0); \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("shift of a pointer that may be NULL")
+__failure __msg("R0 pointer arithmetic on map_value_or_null prohibited, null-check it first")
+__failure_unpriv
+__naked void ptr_or_null_shift(void)
+{
+ asm volatile (" \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r1 = %[map_hash_48b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ r0 >>= 1; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("and of a pointer to map")
+__failure __msg("R0 pointer arithmetic on map_ptr prohibited")
+__failure_unpriv
+__naked void map_ptr_and(void)
+{
+ asm volatile (" \
+ r0 = %[map_hash_48b] ll; \
+ r0 &= 1; \
+ r0 = 0; \
+ exit; \
+" :
+ : __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
+SEC("socket")
+__description("32-bit and of a pointer")
+__success __retval(0)
+__failure_unpriv __msg_unpriv("R0 32-bit pointer arithmetic prohibited")
+__naked void ptr_and32(void)
+{
+ asm volatile (" \
+ r2 = r10; \
+ r2 += -8; \
+ r1 = 0; \
+ *(u64*)(r2 + 0) = r1; \
+ r1 = %[map_hash_48b] ll; \
+ call %[bpf_map_lookup_elem]; \
+ if r0 == 0 goto l0_%=; \
+ w0 &= 1; \
+l0_%=: r0 = 0; \
+ exit; \
+" :
+ : __imm(bpf_map_lookup_elem),
+ __imm_addr(map_hash_48b)
+ : __clobber_all);
+}
+
SEC("flow_dissector")
__description("flow_keys illegal alu op with variable offset")
__failure __msg("R7 pointer arithmetic on flow_keys prohibited")
diff --git a/tools/testing/selftests/bpf/test_loader.c b/tools/testing/selftests/bpf/test_loader.c
index 25eeb1c1248b..a89890cd56d8 100644
--- a/tools/testing/selftests/bpf/test_loader.c
+++ b/tools/testing/selftests/bpf/test_loader.c
@@ -580,6 +580,8 @@ static int parse_test_spec(struct test_loader *tester,
update_flags(&spec->prog_flags, BPF_F_XDP_HAS_FRAGS, clear);
} else if (strcmp(val, "BPF_F_TEST_REG_INVARIANTS") == 0) {
update_flags(&spec->prog_flags, BPF_F_TEST_REG_INVARIANTS, clear);
+ } else if (strcmp(val, "BPF_F_ARENA_SCALAR") == 0) {
+ update_flags(&spec->prog_flags, BPF_F_ARENA_SCALAR, clear);
} else /* assume numeric value */ {
err = parse_int(val, &flags, "test prog flags");
if (err)
diff --git a/tools/testing/selftests/drivers/net/psp.py b/tools/testing/selftests/drivers/net/psp.py
index 5a81f40cac7d..473500901879 100755
--- a/tools/testing/selftests/drivers/net/psp.py
+++ b/tools/testing/selftests/drivers/net/psp.py
@@ -11,6 +11,8 @@ import struct
import termios
import time
+from contextlib import contextmanager
+
from lib.py import defer
from lib.py import ksft_run, ksft_exit, ksft_pr
from lib.py import ksft_true, ksft_eq, ksft_ne, ksft_gt, ksft_raises
@@ -58,6 +60,17 @@ def _make_psp_conn(cfg, version=0, ipver=None):
return s
+@contextmanager
+def _make_lo_conn():
+ # After tx-assoc, the client's egress is dropped, since lo has no
+ # psp_dev, so its FIN never reaches the server. Closing the server
+ # resets the unaccepted child, and the client accepts the cleartext
+ # RST because it hasn't received any PSP traffic yet.
+ with socket.create_server(("localhost", 0)) as srv, \
+ socket.create_connection(srv.getsockname()[:2]) as s:
+ yield s
+
+
def _close_conn(cfg, s):
_send_with_ack(cfg, b'data close\0')
s.close()
@@ -200,20 +213,18 @@ def dev_rotate_spi(cfg):
_init_psp_dev(cfg)
top_a = top_b = 0
- with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
+ with _make_lo_conn() as s:
assoc_a = cfg.pspnl.rx_assoc({"version": 0,
"dev-id": cfg.psp_dev_id,
"sock-fd": s.fileno()})
top_a = assoc_a['rx-key']['spi'] >> 31
- s.close()
rot = cfg.pspnl.key_rotate({"id": cfg.psp_dev_id})
- with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
+ with _make_lo_conn() as s:
ksft_eq(rot['id'], cfg.psp_dev_id)
assoc_b = cfg.pspnl.rx_assoc({"version": 0,
"dev-id": cfg.psp_dev_id,
"sock-fd": s.fileno()})
top_b = assoc_b['rx-key']['spi'] >> 31
- s.close()
ksft_ne(top_a, top_b)
@@ -221,7 +232,7 @@ def assoc_basic(cfg):
""" Test creating associations """
_init_psp_dev(cfg)
- with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
+ with _make_lo_conn() as s:
assoc = cfg.pspnl.rx_assoc({"version": 0,
"dev-id": cfg.psp_dev_id,
"sock-fd": s.fileno()})
@@ -234,7 +245,6 @@ def assoc_basic(cfg):
"tx-key": assoc['rx-key'],
"sock-fd": s.fileno()})
ksft_eq(len(assoc), 0)
- s.close()
def assoc_bad_dev(cfg):
@@ -309,6 +319,32 @@ def assoc_sk_only_unconn(cfg):
ksft_eq(the_exception.nl_msg.error, -errno.EINVAL)
+def assoc_rx_unconnected(cfg):
+ """ Test that an Rx assoc is rejected on an unconnected socket """
+ _init_psp_dev(cfg)
+
+ with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
+ with ksft_raises(NlError) as cm:
+ cfg.pspnl.rx_assoc({"version": 0,
+ "dev-id": cfg.psp_dev_id,
+ "sock-fd": s.fileno()})
+ ksft_eq(cm.exception.nl_msg.error, -errno.ENOTCONN)
+ ksft_eq(cm.exception.nl_msg.extack['bad-attr'], ".sock-fd")
+
+
+def assoc_rx_listener(cfg):
+ """ Test that an Rx assoc is rejected on a listening socket """
+ _init_psp_dev(cfg)
+
+ with socket.create_server(("localhost", 0)) as s:
+ with ksft_raises(NlError) as cm:
+ cfg.pspnl.rx_assoc({"version": 0,
+ "dev-id": cfg.psp_dev_id,
+ "sock-fd": s.fileno()})
+ ksft_eq(cm.exception.nl_msg.error, -errno.ENOTCONN)
+ ksft_eq(cm.exception.nl_msg.extack['bad-attr'], ".sock-fd")
+
+
def assoc_version_mismatch(cfg):
""" Test creating associations where Rx and Tx PSP versions do not match """
_init_psp_dev(cfg)
@@ -320,7 +356,7 @@ def assoc_version_mismatch(cfg):
# Translate versions to integers
versions = [cfg.pspnl.consts["version"].entries[v].value for v in versions]
- with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
+ with _make_lo_conn() as s:
rx = cfg.pspnl.rx_assoc({"version": versions[0],
"dev-id": cfg.psp_dev_id,
"sock-fd": s.fileno()})
@@ -393,7 +429,7 @@ def assoc_twice(cfg):
return assoc
- with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
+ with _make_lo_conn() as s:
assoc = rx_assoc_check(s)
tx = cfg.pspnl.tx_assoc({"dev-id": cfg.psp_dev_id,
"version": 0,
@@ -402,7 +438,7 @@ def assoc_twice(cfg):
ksft_eq(len(tx), 0)
# Use the same Tx assoc second time
- with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s2:
+ with _make_lo_conn() as s2:
rx_assoc_check(s2)
tx = cfg.pspnl.tx_assoc({"dev-id": cfg.psp_dev_id,
"version": 0,
@@ -410,8 +446,6 @@ def assoc_twice(cfg):
"sock-fd": s2.fileno()})
ksft_eq(len(tx), 0)
- s.close()
-
def _data_basic_send(cfg, version, ipver):
""" Test basic data send """
diff --git a/tools/testing/selftests/filesystems/.gitignore b/tools/testing/selftests/filesystems/.gitignore
index 57f5bbdbedff..cf79000e4092 100644
--- a/tools/testing/selftests/filesystems/.gitignore
+++ b/tools/testing/selftests/filesystems/.gitignore
@@ -6,3 +6,4 @@ anon_inode_test
kernfs_test
idmapped_tmpfile
ustat_test
+rw_hint_test
diff --git a/tools/testing/selftests/filesystems/Makefile b/tools/testing/selftests/filesystems/Makefile
index bc4bfb677589..fbd5c27505ef 100644
--- a/tools/testing/selftests/filesystems/Makefile
+++ b/tools/testing/selftests/filesystems/Makefile
@@ -1,7 +1,7 @@
# SPDX-License-Identifier: GPL-2.0
CFLAGS += $(KHDR_INCLUDES)
-TEST_GEN_PROGS := devpts_pts anon_inode_test kernfs_test fclog ustat_test
+TEST_GEN_PROGS := devpts_pts anon_inode_test kernfs_test fclog ustat_test rw_hint_test
TEST_GEN_PROGS += idmapped_tmpfile
TEST_GEN_PROGS_EXTENDED := dnotify_test
diff --git a/tools/testing/selftests/filesystems/empty_mntns/.gitignore b/tools/testing/selftests/filesystems/empty_mntns/.gitignore
index 99f89d329db2..27bcbcaeb1d1 100644
--- a/tools/testing/selftests/filesystems/empty_mntns/.gitignore
+++ b/tools/testing/selftests/filesystems/empty_mntns/.gitignore
@@ -2,3 +2,6 @@
clone3_empty_mntns_test
empty_mntns_test
overmount_chroot_test
+internal_sb_reconfigure_test
+nullfs_atime_test
+root_readdir_test
diff --git a/tools/testing/selftests/filesystems/empty_mntns/Makefile b/tools/testing/selftests/filesystems/empty_mntns/Makefile
index 22e3fb915e81..22af2164c54e 100644
--- a/tools/testing/selftests/filesystems/empty_mntns/Makefile
+++ b/tools/testing/selftests/filesystems/empty_mntns/Makefile
@@ -4,9 +4,14 @@ CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) $(TOOLS_INCLUDES)
LDLIBS += -lcap
TEST_GEN_PROGS := empty_mntns_test overmount_chroot_test clone3_empty_mntns_test
+TEST_GEN_PROGS += internal_sb_reconfigure_test nullfs_atime_test root_readdir_test
+
+LOCAL_HDRS += ../readdir_hold.h
include ../../lib.mk
$(OUTPUT)/empty_mntns_test: ../utils.c
$(OUTPUT)/overmount_chroot_test: ../utils.c
$(OUTPUT)/clone3_empty_mntns_test: ../utils.c
+$(OUTPUT)/internal_sb_reconfigure_test: ../utils.c
+$(OUTPUT)/root_readdir_test: LDLIBS += -pthread
diff --git a/tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c b/tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c
new file mode 100644
index 000000000000..cb645d1e5a9a
--- /dev/null
+++ b/tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c
@@ -0,0 +1,108 @@
+// SPDX-License-Identifier: GPL-2.0-or-later
+/*
+ * The root of an empty mount namespace is a nullfs mount. Its superblock is
+ * kernel-internal and shared by every mount namespace. It can't be
+ * reconfigured, neither through fspick() nor through mount(MS_REMOUNT) nor
+ * through umount() of the root which remounts it read-only.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/statfs.h>
+#include <sys/statvfs.h>
+#include <sys/syscall.h>
+#include <sys/vfs.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include "../utils.h"
+#include "../wrappers.h"
+#include "empty_mntns.h"
+#include "kselftest_harness.h"
+
+#ifndef __NR_fspick
+#define __NR_fspick 433
+#endif
+
+static int sys_fspick(int dfd, const char *path, unsigned int flags)
+{
+ return syscall(__NR_fspick, dfd, path, flags);
+}
+
+/* Child exit codes. */
+enum {
+ CHILD_OK,
+ CHILD_USERNS, /* could not create the user namespace */
+ CHILD_UNSHARE, /* could not create the empty mount namespace */
+ CHILD_FSPICK, /* fspick() of the root was not refused with EINVAL */
+ CHILD_REMOUNT, /* mount(MS_REMOUNT) was not refused with EINVAL */
+ CHILD_UMOUNT, /* umount() of the root succeeded */
+ CHILD_STATFS, /* statfs() of the root failed */
+ CHILD_RDONLY, /* the root ended up read-only */
+};
+
+static int empty_mntns_child(void)
+{
+ struct statfs st;
+
+ if (enter_userns())
+ return CHILD_USERNS;
+ if (unshare(UNSHARE_EMPTY_MNTNS))
+ return CHILD_UNSHARE;
+
+ if (sys_fspick(AT_FDCWD, "/", 0) >= 0 || errno != EINVAL)
+ return CHILD_FSPICK;
+ if (!mount(NULL, "/", NULL, MS_REMOUNT | MS_RDONLY, NULL) ||
+ errno != EINVAL)
+ return CHILD_REMOUNT;
+ if (!umount2("/", 0))
+ return CHILD_UMOUNT;
+ if (statfs("/", &st))
+ return CHILD_STATFS;
+ if (st.f_flags & ST_RDONLY)
+ return CHILD_RDONLY;
+ return CHILD_OK;
+}
+
+FIXTURE(internal_sb_reconfigure) {};
+
+FIXTURE_SETUP(internal_sb_reconfigure)
+{
+ pid_t pid;
+ int status;
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ if (enter_userns())
+ _exit(1);
+ if (unshare(UNSHARE_EMPTY_MNTNS))
+ _exit(1);
+ _exit(0);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ if (!WIFEXITED(status) || WEXITSTATUS(status))
+ SKIP(return, "UNSHARE_EMPTY_MNTNS not supported");
+}
+
+FIXTURE_TEARDOWN(internal_sb_reconfigure) {}
+
+TEST_F(internal_sb_reconfigure, nullfs_root)
+{
+ pid_t pid;
+ int status;
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(empty_mntns_child());
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), CHILD_OK);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c b/tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c
new file mode 100644
index 000000000000..51e34f3c4f54
--- /dev/null
+++ b/tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c
@@ -0,0 +1,129 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * The root of every empty mount namespace is the same nullfs inode. A read
+ * of it by one user must not change the access time another user sees.
+ */
+#define _GNU_SOURCE
+#include <dirent.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/stat.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef UNSHARE_EMPTY_MNTNS
+#define UNSHARE_EMPTY_MNTNS 0x00100000
+#endif
+
+enum {
+ CHILD_OK,
+ CHILD_UNSUPPORTED,
+ CHILD_SETUP,
+ CHILD_CHANGED,
+};
+
+static int wait_byte(int fd)
+{
+ char c;
+
+ return read(fd, &c, 1) == 1 ? 0 : -1;
+}
+
+static int send_byte(int fd)
+{
+ return write(fd, "x", 1) == 1 ? 0 : -1;
+}
+
+static int empty_mntns(void)
+{
+ if (!unshare(UNSHARE_EMPTY_MNTNS))
+ return 0;
+ return errno == EINVAL ? CHILD_UNSUPPORTED : CHILD_SETUP;
+}
+
+/* the watcher: stats its root before and after the reader read its own */
+static int watcher(int to_reader, int from_reader)
+{
+ struct stat before, after;
+ int ret;
+
+ ret = empty_mntns();
+ if (ret)
+ return ret;
+ if (stat("/", &before))
+ return CHILD_SETUP;
+ if (send_byte(to_reader) || wait_byte(from_reader))
+ return CHILD_SETUP;
+ if (stat("/", &after))
+ return CHILD_SETUP;
+ if (before.st_atim.tv_sec != after.st_atim.tv_sec ||
+ before.st_atim.tv_nsec != after.st_atim.tv_nsec)
+ return CHILD_CHANGED;
+ return CHILD_OK;
+}
+
+/* the reader: lists its own root, which is the same inode */
+static int reader(int to_watcher, int from_watcher)
+{
+ struct dirent *de;
+ DIR *d;
+ int ret;
+
+ ret = empty_mntns();
+ if (ret)
+ return ret;
+ if (wait_byte(from_watcher))
+ return CHILD_SETUP;
+ d = opendir("/");
+ if (!d)
+ return CHILD_SETUP;
+ while ((de = readdir(d)))
+ ;
+ closedir(d);
+ return send_byte(to_watcher) ? CHILD_SETUP : CHILD_OK;
+}
+
+static int wait_child(pid_t pid)
+{
+ int status;
+
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+TEST(empty_mntns_root_atime)
+{
+ int to_reader[2], to_watcher[2], w, r;
+ pid_t watcher_pid, reader_pid;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+ ASSERT_EQ(pipe(to_reader), 0);
+ ASSERT_EQ(pipe(to_watcher), 0);
+
+ watcher_pid = fork();
+ ASSERT_GE(watcher_pid, 0);
+ if (watcher_pid == 0)
+ _exit(watcher(to_reader[1], to_watcher[0]));
+ reader_pid = fork();
+ ASSERT_GE(reader_pid, 0);
+ if (reader_pid == 0)
+ _exit(reader(to_watcher[1], to_reader[0]));
+
+ r = wait_child(reader_pid);
+ w = wait_child(watcher_pid);
+ if (r == CHILD_UNSUPPORTED || w == CHILD_UNSUPPORTED)
+ SKIP(return, "UNSHARE_EMPTY_MNTNS not supported");
+ EXPECT_EQ(r, CHILD_OK);
+ EXPECT_EQ(w, CHILD_OK)
+ TH_LOG("the access time of the root changed while this namespace did nothing");
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c b/tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c
new file mode 100644
index 000000000000..0ff544247027
--- /dev/null
+++ b/tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c
@@ -0,0 +1,45 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Reading the root of an empty mount namespace holds nothing that anybody
+ * else waits for: the directory never has an entry, so a readdir stuck in
+ * the page fault of its buffer stalls neither a create nor a lookup in it.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <unistd.h>
+
+#include "../../kselftest_harness.h"
+#include "../readdir_hold.h"
+#include "empty_mntns.h"
+
+TEST(readdir_blocks_nobody)
+{
+ struct readdir_hold hold;
+ bool stalled;
+ int dfd;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+ if (readdir_hold_init(&hold))
+ SKIP(return, "test requires userfaultfd");
+ if (unshare(UNSHARE_EMPTY_MNTNS)) {
+ readdir_hold_destroy(&hold);
+ if (errno == EINVAL)
+ SKIP(return, "UNSHARE_EMPTY_MNTNS not supported");
+ ASSERT_TRUE(false)
+ TH_LOG("unshare(UNSHARE_EMPTY_MNTNS): %m");
+ }
+ dfd = open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(dfd, 0);
+ ASSERT_EQ(readdir_hold_check(&hold, dfd, &stalled), 0);
+ /* the lookup came back while the readdir was stuck in its fault */
+ EXPECT_FALSE(stalled);
+ close(dfd);
+ readdir_hold_destroy(&hold);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/.gitignore b/tools/testing/selftests/filesystems/mount_cycle/.gitignore
new file mode 100644
index 000000000000..30f7dd071bcb
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/.gitignore
@@ -0,0 +1,8 @@
+# SPDX-License-Identifier: GPL-2.0-only
+loop_cycle_test
+unmounted_tree_test
+overmount_reparent_test
+mount_cover_test
+locked_handle_test
+nsfs_rbind_loop_test
+overmount_ns_file_test
diff --git a/tools/testing/selftests/filesystems/mount_cycle/Makefile b/tools/testing/selftests/filesystems/mount_cycle/Makefile
new file mode 100644
index 000000000000..fbe8c5e19d42
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/Makefile
@@ -0,0 +1,12 @@
+# SPDX-License-Identifier: GPL-2.0
+TEST_GEN_PROGS := loop_cycle_test unmounted_tree_test overmount_reparent_test mount_cover_test locked_handle_test
+TEST_GEN_PROGS += nsfs_rbind_loop_test overmount_ns_file_test
+
+CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
+
+LOCAL_HDRS += ../readdir_hold.h
+
+include ../../lib.mk
+
+$(OUTPUT)/locked_handle_test: LDLIBS += -pthread
+$(OUTPUT)/mount_cover_test: LDLIBS += -pthread
diff --git a/tools/testing/selftests/filesystems/mount_cycle/config b/tools/testing/selftests/filesystems/mount_cycle/config
new file mode 100644
index 000000000000..3bd5ce46af74
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/config
@@ -0,0 +1,39 @@
+CONFIG_USER_NS=y
+CONFIG_TMPFS=y
+CONFIG_BLK_DEV_LOOP=y
+CONFIG_VFAT_FS=y
+CONFIG_MSDOS_FS=y
+CONFIG_NLS_CODEPAGE_437=y
+CONFIG_NLS_ISO8859_1=y
+CONFIG_MINIX_FS=y
+CONFIG_AUTOFS_FS=y
+CONFIG_ZRAM=y
+CONFIG_ZRAM_WRITEBACK=y
+CONFIG_KEYS=y
+CONFIG_ECRYPT_FS=y
+CONFIG_BINFMT_MISC=y
+CONFIG_FUSE_FS=y
+CONFIG_FUSE_PASSTHROUGH=y
+CONFIG_BLK_DEV_ZONED=y
+CONFIG_BLK_DEV_ZONED_LOOP=y
+CONFIG_CONFIGFS_FS=y
+CONFIG_USB_SUPPORT=y
+CONFIG_USB=y
+CONFIG_USB_STORAGE=y
+CONFIG_SCSI=y
+CONFIG_BLK_DEV_SD=y
+CONFIG_USB_GADGET=y
+CONFIG_USB_DUMMY_HCD=y
+CONFIG_USB_CONFIGFS=y
+CONFIG_USB_CONFIGFS_MASS_STORAGE=y
+CONFIG_MD=y
+CONFIG_BLK_DEV_MD=y
+CONFIG_MD_RAID1=y
+CONFIG_MD_BITMAP=y
+CONFIG_MD_BITMAP_FILE=y
+CONFIG_INOTIFY_USER=y
+CONFIG_DNOTIFY=y
+CONFIG_FANOTIFY=y
+CONFIG_FILE_LOCKING=y
+CONFIG_CRYPTO_AES=y
+CONFIG_USERFAULTFD=y
diff --git a/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c b/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c
new file mode 100644
index 000000000000..b21f43471e49
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c
@@ -0,0 +1,377 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * may_decode_fh() refuses a file handle below a mount with locked children
+ * to root in a user namespace. That has to hold while the mount is lazily
+ * unmounted and its children, the locked ones too, are taken off it.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <pthread.h>
+#include <sched.h>
+#include <stdatomic.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef OPEN_TREE_CLONE
+#define OPEN_TREE_CLONE 1
+#endif
+#ifndef OPEN_TREE_CLOEXEC
+#define OPEN_TREE_CLOEXEC O_CLOEXEC
+#endif
+#ifndef AT_RECURSIVE
+#define AT_RECURSIVE 0x8000
+#endif
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+#define ROUNDS 100 /* lazy umounts raced per test */
+#define RACERS 3 /* threads in open_by_handle_at() */
+#define EXTRA_MOUNTS 64 /* make umount_tree() hold mount_lock longer */
+#define MAX_DELAY_US 3000 /* before the umount */
+#define TAIL_US 2000 /* after it */
+
+struct handle {
+ struct file_handle fh;
+ unsigned char buf[MAX_HANDLE_SZ];
+};
+
+/* exit codes of the child */
+enum {
+ CHILD_OK,
+ CHILD_SETUP,
+ CHILD_DECODED, /* decoded past a locked child */
+ CHILD_MOUNTED, /* not refused while mounted */
+ CHILD_UNMOUNTED, /* not refused once unmounted */
+ CHILD_ALLOWED, /* refused where nothing is locked */
+ CHILD_NOUSERNS, /* no user namespace to be had */
+};
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ prctl(PR_SET_DUMPABLE, 1);
+ /* EINVAL: no USER_NS, ENOSPC: user.max_user_namespaces is 0, EPERM: an LSM */
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return errno == EINVAL || errno == ENOSPC || errno == EPERM ?
+ CHILD_NOUSERNS : CHILD_SETUP;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return CHILD_SETUP;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return CHILD_SETUP;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return CHILD_SETUP;
+ if (setgid(0) || setuid(0))
+ return CHILD_SETUP;
+ if (mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+ return CHILD_SETUP;
+ return CHILD_OK;
+}
+
+static int get_handle(const char *path, struct handle *h)
+{
+ int mntid;
+
+ h->fh.handle_bytes = MAX_HANDLE_SZ;
+ return name_to_handle_at(AT_FDCWD, path, &h->fh, &mntid, 0);
+}
+
+static int decode(int dfd, struct handle *h)
+{
+ return open_by_handle_at(dfd, &h->fh, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+}
+
+struct race {
+ int dfd;
+ struct handle *h;
+ atomic_int go;
+ atomic_int stop;
+ atomic_int won; /* the first decoded descriptor */
+};
+
+static void *racer(void *arg)
+{
+ struct race *r = arg;
+
+ while (!atomic_load(&r->go))
+ ;
+ while (!atomic_load(&r->stop)) {
+ int none = -1;
+ int fd;
+
+ fd = decode(r->dfd, r->h);
+ if (fd < 0)
+ continue;
+ if (!atomic_compare_exchange_strong(&r->won, &none, fd))
+ close(fd);
+ }
+ return NULL;
+}
+
+/* stop the @n racers started so far and wait for them, they spin on @r */
+static void stop_racers(struct race *r, pthread_t *th, int n)
+{
+ int i;
+
+ atomic_store(&r->stop, 1);
+ atomic_store(&r->go, 1); /* one still waiting for the start sees stop next */
+ for (i = 0; i < n; i++)
+ pthread_join(th[i], NULL);
+}
+
+/*
+ * Bind @h recursively at @w, or take a detached copy of it, and race
+ * open_by_handle_at() of @inner against the lazy umount. The decoded
+ * descriptor, if there was one, is left in @won.
+ */
+static int race_round(const char *h, const char *w, struct handle *inner,
+ bool dissolve, int *won)
+{
+ struct race r = { .h = inner, .won = -1 };
+ pthread_t th[RACERS];
+ int treefd = -1, fd, i;
+
+ if (dissolve) {
+ treefd = syscall(__NR_open_tree, AT_FDCWD, h,
+ OPEN_TREE_CLONE | AT_RECURSIVE | OPEN_TREE_CLOEXEC);
+ if (treefd < 0)
+ return CHILD_SETUP;
+ /* an ordinary descriptor on the copy, treefd's close dissolves it */
+ r.dfd = openat(treefd, ".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ } else {
+ if (mount(h, w, NULL, MS_BIND | MS_REC, NULL))
+ return CHILD_SETUP;
+ r.dfd = open(w, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ }
+ if (r.dfd < 0)
+ return CHILD_SETUP;
+
+ fd = decode(r.dfd, inner);
+ if (fd >= 0 || errno != EPERM) {
+ if (fd >= 0)
+ close(fd);
+ return CHILD_MOUNTED;
+ }
+
+ for (i = 0; i < RACERS; i++) {
+ if (pthread_create(&th[i], NULL, racer, &r)) {
+ stop_racers(&r, th, i);
+ return CHILD_SETUP;
+ }
+ }
+ atomic_store(&r.go, 1);
+ usleep(rand() % MAX_DELAY_US);
+ if (dissolve)
+ close(treefd);
+ else if (umount2(w, MNT_DETACH)) {
+ stop_racers(&r, th, RACERS);
+ return CHILD_SETUP;
+ }
+ usleep(TAIL_US);
+ stop_racers(&r, th, RACERS);
+
+ fd = decode(r.dfd, inner);
+ if (fd >= 0) {
+ close(fd);
+ return CHILD_UNMOUNTED;
+ }
+ close(r.dfd);
+ *won = atomic_load(&r.won);
+ return CHILD_OK;
+}
+
+static int race_child(const char *h, const char *w, struct handle *inner,
+ bool dissolve)
+{
+ int ret, won, i;
+
+ srand(getpid());
+ ret = enter_userns();
+ if (ret)
+ return ret;
+ for (i = 0; i < ROUNDS; i++) {
+ ret = race_round(h, w, inner, dissolve, &won);
+ if (ret)
+ return ret;
+ if (won >= 0)
+ return CHILD_DECODED;
+ }
+ return CHILD_OK;
+}
+
+/* a bind without locked children decodes while mounted and not after */
+static int allowed_child(const char *plain, const char *w, struct handle *h)
+{
+ int dfd, fd, ret;
+
+ ret = enter_userns();
+ if (ret)
+ return ret;
+ if (mount(plain, w, NULL, MS_BIND | MS_REC, NULL))
+ return CHILD_SETUP;
+ dfd = open(w, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (dfd < 0)
+ return CHILD_SETUP;
+ fd = decode(dfd, h);
+ if (fd < 0)
+ return CHILD_ALLOWED;
+ close(fd);
+ if (umount2(w, MNT_DETACH))
+ return CHILD_SETUP;
+ fd = decode(dfd, h);
+ if (fd >= 0) {
+ close(fd);
+ return CHILD_UNMOUNTED;
+ }
+ return errno == EPERM ? CHILD_OK : CHILD_UNMOUNTED;
+}
+
+FIXTURE(locked_handle) {
+ char base[DIR_LEN];
+ char h[PATH_LEN]; /* the tree with the covered directory */
+ char plain[PATH_LEN]; /* a tree with no mount in it */
+ char w[PATH_LEN]; /* where the child binds either */
+ struct handle inner; /* h/top/secret/inner, covered */
+ struct handle uncovered; /* plain/inner */
+};
+
+FIXTURE_SETUP(locked_handle)
+{
+ char p[PATH_LEN], e[PATH_LEN];
+ int i;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+
+ snprintf(self->base, sizeof(self->base), "/tmp/locked_handle.XXXXXX");
+ ASSERT_NE(mkdtemp(self->base), NULL);
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0);
+
+ snprintf(self->h, sizeof(self->h), "%s/h", self->base);
+ snprintf(self->plain, sizeof(self->plain), "%s/plain", self->base);
+ snprintf(self->w, sizeof(self->w), "%s/w", self->base);
+ ASSERT_EQ(mkdir(self->h, 0755), 0);
+ ASSERT_EQ(mkdir(self->plain, 0755), 0);
+ ASSERT_EQ(mkdir(self->w, 0755), 0);
+ snprintf(p, sizeof(p), "%s/plain/inner", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(get_handle(p, &self->uncovered), 0);
+
+ /* the handle is for this filesystem */
+ ASSERT_EQ(mount("tmpfs", self->h, "tmpfs", 0, NULL), 0);
+ snprintf(p, sizeof(p), "%s/h/top", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/h/top/secret", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/h/top/secret/inner", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(get_handle(p, &self->inner), 0);
+ snprintf(e, sizeof(e), "%s/h/empty", self->base);
+ ASSERT_EQ(mkdir(e, 0755), 0);
+
+ /* cover it, and some more so the umount takes longer */
+ snprintf(p, sizeof(p), "%s/h/top/secret", self->base);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, NULL), 0);
+ for (i = 0; i < EXTRA_MOUNTS; i++) {
+ snprintf(p, sizeof(p), "%s/h/c%d", self->base, i);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(mount(e, p, NULL, MS_BIND, NULL), 0);
+ }
+}
+
+FIXTURE_TEARDOWN(locked_handle)
+{
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+}
+
+static int wait_child(pid_t pid)
+{
+ int status;
+
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+TEST_F(locked_handle, lazy_umount)
+{
+ pid_t pid;
+ int ret;
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(race_child(self->h, self->w, &self->inner, false));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ if (ret == CHILD_NOUSERNS)
+ SKIP(return, "no user namespaces");
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_F(locked_handle, dissolve)
+{
+ pid_t pid;
+ int ret;
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(race_child(self->h, self->w, &self->inner, true));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ if (ret == CHILD_NOUSERNS)
+ SKIP(return, "no user namespaces");
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_F(locked_handle, allowed_use)
+{
+ pid_t pid;
+ int ret;
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(allowed_child(self->plain, self->w, &self->uncovered));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ if (ret == CHILD_NOUSERNS)
+ SKIP(return, "no user namespaces");
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c b/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c
new file mode 100644
index 000000000000..6b4f5304c2e4
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c
@@ -0,0 +1,1542 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A mount that another namespace's rmdir detached, or that went down with
+ * the detached tree it was in when the tree's last fd was closed, keeps
+ * its submounts connected, and a connected submount is put by its
+ * parent's final mntput(). A submount whose filesystem keeps a file open
+ * on the parent then holds the parent's count above zero for good:
+ * nothing in userspace refers to either mount any more and nothing can
+ * release them. A loop device is the simplest such filesystem, its
+ * backing file sits on the parent.
+ */
+#define _GNU_SOURCE
+#include <dirent.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/ioctl.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <stdbool.h>
+#include <sys/wait.h>
+#include <unistd.h>
+#include <linux/fuse.h>
+#include <linux/keyctl.h>
+#include <linux/major.h>
+#include <linux/raid/md_u.h>
+#include <linux/raid/md_p.h>
+#include <linux/loop.h>
+#include <linux/magic.h>
+#include <sys/syscall.h>
+#include <sys/sysmacros.h>
+#include <sys/uio.h>
+
+#include "../wrappers.h"
+#include "../../kselftest_harness.h"
+
+#define IMAGE_SIZE (1440 * 1024)
+#define SECTOR 512
+
+/*
+ * The tmpfs of the fixture. A directory of its own from mkdtemp(), so that
+ * the test needs no writable root directory and two instances don't take
+ * each other's mounts down.
+ */
+static char base[64];
+
+/* The configfs directory of the gadget, named after the test process. */
+static char gadget[96];
+
+/* A path below the base directory. Eight of them can be in use at a time. */
+static const char *at(const char *rel)
+{
+ static char buf[8][PATH_MAX];
+ static unsigned int next;
+ char *p = buf[next++ % 8];
+
+ snprintf(p, PATH_MAX, "%s/%s", base, rel);
+ return p;
+}
+
+/* A path below the gadget's configfs directory. */
+static const char *gat(const char *rel)
+{
+ static char buf[4][PATH_MAX];
+ static unsigned int next;
+ char *p = buf[next++ % 4];
+
+ snprintf(p, PATH_MAX, "%s%s", gadget, rel);
+ return p;
+}
+
+/* The loop devices this process has bound, to release them on every path. */
+#define MAX_LOOPS 4
+static int bound[MAX_LOOPS];
+static int nr_bound;
+
+/* What a child tells its parent: the loop devices it has bound, -1 for none. */
+struct report {
+ int n[2];
+};
+
+/* A blank FAT12 floppy image: boot sector, two FATs, an empty root directory. */
+static int write_fat12(int fd)
+{
+ struct stat st;
+ unsigned char sector[SECTOR] = {
+ 0xeb, 0x3c, 0x90, 'M', 'S', 'W', 'I', 'N', '4', '.', '1',
+ [11] = 0x00, 0x02, /* bytes per sector: 512 */
+ [13] = 1, /* sectors per cluster */
+ [14] = 1, 0, /* reserved sectors */
+ [16] = 2, /* FATs */
+ [17] = 0xe0, 0x00, /* root directory entries: 224 */
+ [19] = 0x40, 0x0b, /* total sectors: 2880 */
+ [21] = 0xf0, /* media descriptor */
+ [22] = 9, 0, /* sectors per FAT */
+ [24] = 18, 0, /* sectors per track */
+ [26] = 2, 0, /* heads */
+ [38] = 0x29, /* extended boot signature */
+ [39] = 0x12, 0x34, 0x56, 0x78,
+ [43] = 'N', 'O', ' ', 'N', 'A', 'M', 'E', ' ', ' ', ' ', ' ',
+ [54] = 'F', 'A', 'T', '1', '2', ' ', ' ', ' ',
+ [510] = 0x55, 0xaa,
+ };
+ unsigned char fat[SECTOR] = { 0xf0, 0xff, 0xff };
+
+ if (pwrite(fd, sector, SECTOR, 0) != SECTOR)
+ return -1;
+ /* the first FAT and the second one, one sector each is enough */
+ if (pwrite(fd, fat, SECTOR, 1 * SECTOR) != SECTOR ||
+ pwrite(fd, fat, SECTOR, 10 * SECTOR) != SECTOR)
+ return -1;
+ if (fstat(fd, &st) || S_ISBLK(st.st_mode))
+ return 0;
+ return ftruncate(fd, IMAGE_SIZE);
+}
+
+/*
+ * A blank FAT12 image with 4 KiB sectors and @sectors of them, for a device
+ * with 4 KiB logical blocks (zram) or for a bigger image than a floppy.
+ */
+static int write_fat12_4k(int fd, unsigned int sectors)
+{
+ unsigned int fat_sectors = (sectors * 3 / 2 + 4095) / 4096;
+ unsigned char sector[4096] = {
+ 0xeb, 0x3c, 0x90, 'M', 'S', 'W', 'I', 'N', '4', '.', '1',
+ [11] = 0x00, 0x10, /* bytes per sector: 4096 */
+ [13] = 1, /* sectors per cluster */
+ [14] = 1, 0, /* reserved sectors */
+ [16] = 2, /* FATs */
+ [17] = 128, 0, /* root directory entries: one sector */
+ [19] = sectors & 0xff, sectors >> 8,
+ [21] = 0xf8, /* media descriptor */
+ [22] = fat_sectors, 0,
+ [24] = 63, 0, /* sectors per track */
+ [26] = 255, 0, /* heads */
+ [38] = 0x29, /* extended boot signature */
+ [39] = 0x12, 0x34, 0x56, 0x78,
+ [43] = 'N', 'O', ' ', 'N', 'A', 'M', 'E', ' ', ' ', ' ', ' ',
+ [54] = 'F', 'A', 'T', '1', '2', ' ', ' ', ' ',
+ [510] = 0x55, 0xaa,
+ };
+ unsigned char fat[4096] = { 0xf8, 0xff, 0xff };
+ struct stat st;
+
+ if (pwrite(fd, sector, sizeof(sector), 0) != sizeof(sector))
+ return -1;
+ if (pwrite(fd, fat, sizeof(fat), 1 * 4096) != sizeof(fat) ||
+ pwrite(fd, fat, sizeof(fat), (1 + fat_sectors) * 4096) != sizeof(fat))
+ return -1;
+ if (fstat(fd, &st) || S_ISBLK(st.st_mode))
+ return 0;
+ return ftruncate(fd, (off_t)sectors * 4096);
+}
+
+#define MINIX_BLOCK 1024
+#define MINIX_BLOCKS 4096 /* a 4 MiB image */
+#define MINIX_INODES 512
+#define MINIX_ITABLE (MINIX_INODES * 32 / MINIX_BLOCK)
+#define MINIX_FIRSTDATA (2 + 1 + 1 + MINIX_ITABLE) /* boot, super, imap, zmap, inodes */
+
+/*
+ * A blank minix v1 image, for the holders that need a FIFO or a device
+ * node on the dying mount, which vfat can't hold. Superblock in block 1,
+ * one block each for the inode and zone bitmaps, the inode table, and
+ * the root directory in the first data zone.
+ */
+static int write_minix(int fd)
+{
+ struct {
+ __u16 s_ninodes, s_nzones, s_imap_blocks, s_zmap_blocks;
+ __u16 s_firstdatazone, s_log_zone_size;
+ __u32 s_max_size;
+ __u16 s_magic, s_state;
+ } sb = {
+ .s_ninodes = MINIX_INODES,
+ .s_nzones = MINIX_BLOCKS,
+ .s_imap_blocks = 1,
+ .s_zmap_blocks = 1,
+ .s_firstdatazone = MINIX_FIRSTDATA,
+ .s_max_size = (7 + 512 + 512 * 512) * MINIX_BLOCK,
+ .s_magic = MINIX_SUPER_MAGIC,
+ .s_state = 1, /* MINIX_VALID_FS */
+ };
+ struct {
+ __u16 i_mode, i_uid;
+ __u32 i_size, i_time;
+ __u8 i_gid, i_nlinks;
+ __u16 i_zone[9];
+ } root = {
+ .i_mode = S_IFDIR | 0755,
+ .i_size = 2 * 16,
+ .i_nlinks = 2,
+ .i_zone = { MINIX_FIRSTDATA },
+ };
+ unsigned char imap[MINIX_BLOCK], zmap[MINIX_BLOCK], dir[MINIX_BLOCK] = {};
+ int i;
+
+ /* bit 0 is reserved in both maps, the root inode and its zone are in use */
+ memset(imap, 0xff, sizeof(imap));
+ for (i = 2; i <= MINIX_INODES; i++)
+ imap[i / 8] &= ~(1 << (i % 8));
+ memset(zmap, 0xff, sizeof(zmap));
+ for (i = 2; i <= MINIX_BLOCKS - MINIX_FIRSTDATA; i++)
+ zmap[i / 8] &= ~(1 << (i % 8));
+ dir[0] = 1;
+ dir[2] = '.';
+ dir[16] = 1;
+ dir[18] = '.';
+ dir[19] = '.';
+
+ if (pwrite(fd, &sb, sizeof(sb), 1 * MINIX_BLOCK) != sizeof(sb) ||
+ pwrite(fd, imap, sizeof(imap), 2 * MINIX_BLOCK) != sizeof(imap) ||
+ pwrite(fd, zmap, sizeof(zmap), 3 * MINIX_BLOCK) != sizeof(zmap) ||
+ pwrite(fd, &root, sizeof(root), 4 * MINIX_BLOCK) != sizeof(root) ||
+ pwrite(fd, dir, sizeof(dir), MINIX_FIRSTDATA * MINIX_BLOCK) != sizeof(dir))
+ return -1;
+ return ftruncate(fd, (off_t)MINIX_BLOCKS * MINIX_BLOCK);
+}
+
+static int read_sysfs(const char *path, char *buf, size_t size)
+{
+ ssize_t n;
+ int fd;
+
+ fd = open(path, O_RDONLY);
+ if (fd < 0)
+ return -1;
+ n = read(fd, buf, size - 1);
+ close(fd);
+ if (n < 0)
+ return -1;
+ buf[n] = '\0';
+ return 0;
+}
+
+/* Is @dev the mount source of this line of mountinfo? loop1 is not loop10. */
+static bool line_has_source(const char *line, const char *dev)
+{
+ const char *sep, *src, *end;
+
+ sep = strstr(line, " - ");
+ if (!sep)
+ return false;
+ src = strchr(sep + 3, ' '); /* skip the filesystem type */
+ if (!src)
+ return false;
+ src++;
+ end = strchr(src, ' ');
+ if (!end)
+ return false;
+ return (size_t)(end - src) == strlen(dev) && !strncmp(src, dev, end - src);
+}
+
+/*
+ * Does a mount namespace of a process that /proc shows have a mount of @dev?
+ * The mounts under test are in no namespace at this point on any kernel, so
+ * this only checks that the test got as far as it thinks.
+ */
+static bool mounted_anywhere(const char *dev)
+{
+ char path[PATH_MAX], line[4096];
+ struct dirent *de;
+ bool found = false;
+ DIR *proc;
+ FILE *f;
+
+ proc = opendir("/proc");
+ if (!proc)
+ return false;
+ while (!found && (de = readdir(proc))) {
+ if (de->d_name[0] < '0' || de->d_name[0] > '9')
+ continue;
+ snprintf(path, sizeof(path), "/proc/%s/mountinfo", de->d_name);
+ f = fopen(path, "re");
+ if (!f)
+ continue;
+ while (fgets(line, sizeof(line), f)) {
+ if (line_has_source(line, dev)) {
+ found = true;
+ break;
+ }
+ }
+ fclose(f);
+ }
+ closedir(proc);
+ return found;
+}
+
+/* Wait up to @ms milliseconds for the loop device to give up its backing file. */
+static bool loop_released(const char *sysfs, int ms)
+{
+ char buf[PATH_MAX];
+
+ for (; ms > 0; ms -= 100) {
+ if (read_sysfs(sysfs, buf, sizeof(buf)) < 0)
+ return errno == ENOENT;
+ usleep(100000);
+ }
+ return read_sysfs(sysfs, buf, sizeof(buf)) < 0 && errno == ENOENT;
+}
+
+/* Tell the loop device to give its file up, now or when its last user is gone. */
+static void loop_clear(int n)
+{
+ char dev[32];
+ int lfd;
+
+ if (n < 0)
+ return;
+ snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+ lfd = open(dev, O_RDWR);
+ if (lfd < 0)
+ return;
+ ioctl(lfd, LOOP_CLR_FD);
+ close(lfd);
+}
+
+FIXTURE(loop_cycle) {
+ char dev[32]; /* the loop device the child set up */
+ char sysfs[64]; /* its backing_file attribute */
+ int loops[MAX_LOOPS]; /* every loop device the case has bound */
+ int nr_loops;
+ bool gadget; /* the gadget's configfs directory exists */
+};
+
+static void remember_loop(FIXTURE_DATA(loop_cycle) *self, int n)
+{
+ if (n >= 0 && self->nr_loops < MAX_LOOPS)
+ self->loops[self->nr_loops++] = n;
+}
+
+FIXTURE_SETUP(loop_cycle)
+{
+ if (geteuid() != 0)
+ SKIP(return, "test requires CAP_SYS_ADMIN");
+ if (access("/dev/loop-control", R_OK | W_OK))
+ SKIP(return, "test requires loop devices");
+
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+ snprintf(base, sizeof(base), "/tmp/loop_cycle.XXXXXX");
+ ASSERT_NE(mkdtemp(base), NULL);
+ /* a failed assertion in the setup does not run the teardown */
+ ASSERT_EQ(mount("tmpfs", base, "tmpfs", 0, NULL), 0)
+ rmdir(base);
+ snprintf(gadget, sizeof(gadget),
+ "/sys/kernel/config/usb_gadget/kselftest_cycle_%d", getpid());
+ self->dev[0] = '\0';
+ self->nr_loops = 0;
+ self->gadget = false;
+}
+
+static int mount_configfs(void);
+static void gadget_remove(void);
+
+FIXTURE_TEARDOWN(loop_cycle)
+{
+ if (self->gadget)
+ gadget_remove();
+ /* whatever a failed assertion has left bound */
+ for (int i = 0; i < self->nr_loops; i++)
+ loop_clear(self->loops[i]);
+ umount2(base, MNT_DETACH);
+ rmdir(base);
+}
+
+/* Child exit codes. */
+enum {
+ CHILD_OK,
+ CHILD_NS, /* could not set up the namespace or the tmpfs */
+ CHILD_IMAGE, /* could not write the image */
+ CHILD_LOOP, /* could not set up the loop device */
+ CHILD_MOUNT, /* could not mount it (vfat and msdos both refused) */
+ CHILD_PIPE, /* the parent went away */
+ CHILD_HOLDER, /* could not set the holder up below the mount */
+ CHILD_SKIP, /* the kernel lacks what the holder needs */
+ CHILD_NOFS, /* the kernel lacks the filesystem of the image */
+};
+
+/* Bind the loop device that LOOP_CTL_GET_FREE names to @ifd; the device number. */
+static int loop_bind_free(int ifd)
+{
+ int cfd, lfd, n;
+ char dev[32];
+
+ cfd = open("/dev/loop-control", O_RDWR);
+ if (cfd < 0)
+ return -1;
+ n = ioctl(cfd, LOOP_CTL_GET_FREE);
+ close(cfd);
+ if (n < 0)
+ return -1;
+ snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+ lfd = open(dev, O_RDWR);
+ if (lfd < 0)
+ return -1;
+ if (ioctl(lfd, LOOP_SET_FD, ifd))
+ n = -1;
+ close(lfd);
+ return n;
+}
+
+/*
+ * Bind a free loop device to the open image @ifd; the device number. The
+ * device is free when LOOP_CTL_GET_FREE names it and may be somebody else's
+ * a moment later, so try again when it is busy.
+ */
+static int loop_bind(int ifd)
+{
+ int n = -1;
+
+ for (int i = 0; i < 64 && n < 0; i++) {
+ n = loop_bind_free(ifd);
+ if (n < 0 && errno != EBUSY)
+ return -1;
+ }
+ if (n >= 0 && nr_bound < MAX_LOOPS)
+ bound[nr_bound++] = n;
+ return n;
+}
+
+/* Mount a FAT image, as vfat or as msdos; -1 with ENODEV if the kernel has neither. */
+static int mount_fat(const char *dev, const char *mp)
+{
+ int err;
+
+ if (!mount(dev, mp, "vfat", 0, NULL))
+ return 0;
+ err = errno;
+ if (!mount(dev, mp, "msdos", 0, NULL))
+ return 0;
+ if (err != ENODEV)
+ errno = err;
+ return -1;
+}
+
+/*
+ * A child that gives up has to take down what it has set up: nobody else
+ * knows about it. Unmount, then tell the loop devices to let go. The mounts
+ * are released after this process is gone and the devices follow them.
+ */
+static int child_fails(int ret)
+{
+ umount2(at("vol"), MNT_DETACH);
+ umount2(at("vol2"), MNT_DETACH);
+ umount2(at("p"), MNT_DETACH);
+ umount2(at("img"), MNT_DETACH);
+ for (int i = 0; i < nr_bound; i++)
+ loop_clear(bound[i]);
+ return ret;
+}
+
+/* Write an image to @img, bind a loop device to it and mount that at @mp; the device number. */
+static int loop_mount(const char *img, const char *mp)
+{
+ char dev[32];
+ int ifd, n;
+
+ ifd = open(img, O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (ifd < 0 || write_fat12(ifd))
+ return -CHILD_IMAGE;
+ n = loop_bind(ifd);
+ close(ifd); /* the loop device holds the file from now on */
+ if (n < 0)
+ return -CHILD_LOOP;
+ snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+ if (mkdir(mp, 0755))
+ return -CHILD_MOUNT;
+ if (mount_fat(dev, mp))
+ return errno == ENODEV ? -CHILD_NOFS : -CHILD_MOUNT;
+ return n;
+}
+
+/*
+ * Two tmpfs mounts, each carrying the image of the loop mount below the
+ * other: the loop mount below vol has its image on vol2 and the other way
+ * round.
+ */
+static int crossed_child(int to_parent, int from_parent)
+{
+ struct report r;
+ char c;
+
+ if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+ return CHILD_NS;
+ if (mkdir(at("vol"), 0755) || mount("tmpfs", at("vol"), "tmpfs", 0, NULL) ||
+ mkdir(at("vol2"), 0755) || mount("tmpfs", at("vol2"), "tmpfs", 0, NULL))
+ return child_fails(CHILD_NS);
+ r.n[0] = loop_mount(at("vol2/img"), at("vol/mnt"));
+ if (r.n[0] < 0)
+ return child_fails(-r.n[0]);
+ r.n[1] = loop_mount(at("vol/img"), at("vol2/mnt"));
+ if (r.n[1] < 0)
+ return child_fails(-r.n[1]);
+ if (write(to_parent, &r, sizeof(r)) != sizeof(r))
+ return child_fails(CHILD_PIPE);
+ if (read(from_parent, &c, 1) != 1)
+ return child_fails(CHILD_PIPE);
+ return CHILD_OK;
+}
+
+/*
+ * In its own mount namespace the child mounts a tmpfs on vol,
+ * puts a filesystem image on it, binds a loop device to the image and
+ * mounts that loop device below. The loop device's backing file is a
+ * reference on the mount the image is on, held by the loop device, held
+ * by the mounted filesystem, held by the mount below.
+ */
+static int loop_child(int to_parent, int from_parent)
+{
+ struct report r = { { -1, -1 } };
+ char c;
+
+ if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+ return CHILD_NS;
+ if (mkdir(at("vol"), 0755) || mount("tmpfs", at("vol"), "tmpfs", 0, NULL))
+ return child_fails(CHILD_NS);
+ r.n[0] = loop_mount(at("vol/img"), at("vol/mnt"));
+ if (r.n[0] < 0)
+ return child_fails(-r.n[0]);
+
+ if (write(to_parent, &r, sizeof(r)) != sizeof(r))
+ return child_fails(CHILD_PIPE);
+ /* keep the namespace alive while the parent removes the directory */
+ if (read(from_parent, &c, 1) != 1)
+ return child_fails(CHILD_PIPE);
+ return CHILD_OK;
+}
+
+/*
+ * The child did not report. Skip if the kernel lacks something, fail with
+ * what the child said otherwise. A child that a signal killed has failed.
+ */
+#define CHILD_GAVE_UP(pid) do { \
+ int __status; \
+ \
+ ASSERT_EQ(waitpid(pid, &__status, 0), pid); \
+ if (WIFEXITED(__status) && WEXITSTATUS(__status) == CHILD_NOFS) \
+ SKIP(return, "test requires the filesystem of the image (FAT or minix)"); \
+ if (WIFEXITED(__status) && WEXITSTATUS(__status) == CHILD_SKIP) \
+ SKIP(return, "the kernel lacks what this holder needs"); \
+ ASSERT_TRUE(false) \
+ TH_LOG("child failed to set up: %s %d", \
+ WIFEXITED(__status) ? "exit status" : "signal", \
+ WIFEXITED(__status) ? WEXITSTATUS(__status) : WTERMSIG(__status)); \
+} while (0)
+
+/* The child has to leave by itself and with nothing to complain about. */
+#define CHILD_LEFT(pid) do { \
+ int __status; \
+ \
+ ASSERT_EQ(waitpid(pid, &__status, 0), pid); \
+ ASSERT_TRUE(WIFEXITED(__status)); \
+ ASSERT_EQ(WEXITSTATUS(__status), CHILD_OK); \
+} while (0)
+
+/*
+ * An exclusive open of the device fails while a filesystem holds it, and
+ * the only filesystem that ever did is the one mounted below the dead
+ * mount. Give a release in flight a moment. Then the device must clear
+ * right away rather than only be marked for autoclear.
+ */
+static void assert_loop_released(struct __test_metadata *_metadata,
+ FIXTURE_DATA(loop_cycle) *self)
+{
+ int lfd;
+
+ for (int i = 0; i < 20; i++) {
+ lfd = open(self->dev, O_RDONLY | O_EXCL);
+ if (lfd >= 0)
+ break;
+ usleep(100000);
+ }
+ EXPECT_GE(lfd, 0)
+ TH_LOG("%s is still held by the loop mount below the dead mount: nothing refers to either mount and nothing can release them",
+ self->dev);
+ if (lfd >= 0)
+ close(lfd);
+
+ lfd = open(self->dev, O_RDWR);
+ ASSERT_GE(lfd, 0);
+ ASSERT_EQ(ioctl(lfd, LOOP_CLR_FD), 0);
+ close(lfd);
+ ASSERT_TRUE(loop_released(self->sysfs, 5000))
+ TH_LOG("%s kept its backing file after LOOP_CLR_FD: the filesystem on it is still mounted somewhere nobody can reach",
+ self->dev);
+ /* somebody else may bind it from now on, so the teardown leaves it alone */
+ for (int i = 0; i < self->nr_loops; i++) {
+ char dev[32];
+
+ snprintf(dev, sizeof(dev), "/dev/loop%d", self->loops[i]);
+ if (!strcmp(dev, self->dev))
+ self->loops[i] = -1;
+ }
+}
+
+/*
+ * rmdir of vol from here, where it is not a mountpoint, detaches
+ * the child's tmpfs with the loop mount connected below it. Once the child
+ * is gone nothing refers to either mount. The loop device must then be
+ * free to give up its backing file, which only happens when the mounted
+ * filesystem below the detached tmpfs has been released.
+ */
+TEST_F(loop_cycle, detached_loop_mount_released)
+{
+ int to_parent[2], to_child[2];
+ struct report r = { { -1, -1 } };
+ char buf[PATH_MAX];
+ pid_t pid;
+
+ ASSERT_EQ(pipe(to_parent), 0);
+ ASSERT_EQ(pipe(to_child), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ close(to_parent[0]);
+ close(to_child[1]);
+ _exit(loop_child(to_parent[1], to_child[0]));
+ }
+ close(to_parent[1]);
+ close(to_child[0]);
+
+ if (read(to_parent[0], &r, sizeof(r)) != sizeof(r))
+ CHILD_GAVE_UP(pid);
+ remember_loop(self, r.n[0]);
+ snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", r.n[0]);
+ snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", r.n[0]);
+ ASSERT_EQ(read_sysfs(self->sysfs, buf, sizeof(buf)), 0);
+ ASSERT_NE(strstr(buf, "/vol/img"), NULL);
+
+ /* not a mountpoint in this namespace, so the directory can go */
+ ASSERT_EQ(rmdir(at("vol")), 0);
+
+ /* the child leaves: its namespace and every reference it held are gone */
+ ASSERT_EQ(write(to_child[1], "", 1), 1);
+ CHILD_LEFT(pid);
+ close(to_parent[0]);
+ close(to_child[1]);
+
+ /* nothing can reach the two mounts any more */
+ ASSERT_EQ(access(at("vol"), F_OK), -1);
+ ASSERT_FALSE(mounted_anywhere(self->dev));
+
+ assert_loop_released(_metadata, self);
+}
+
+/*
+ * The same two mounts in a detached tree: a clone of vol from
+ * open_tree(), the image opened through the clone so that the loop device
+ * holds the clone, and the loop mount moved below the clone. The last
+ * close of the tree's fd dissolves the tree with the loop mount left
+ * connected below the dead clone, and nothing refers to either afterwards.
+ */
+TEST_F(loop_cycle, dissolved_tree_loop_mount_released)
+{
+ int tfd, ifd, fsfd, mfd, n;
+ char buf[PATH_MAX];
+
+ fsfd = sys_fsopen("vfat", 0);
+ if (fsfd < 0)
+ fsfd = sys_fsopen("msdos", 0);
+ if (fsfd < 0)
+ SKIP(return, "test requires a FAT filesystem");
+
+ ASSERT_EQ(mkdir(at("vol"), 0755), 0);
+ ASSERT_EQ(mount("tmpfs", at("vol"), "tmpfs", 0, NULL), 0);
+ tfd = sys_open_tree(AT_FDCWD, at("vol"), OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(tfd, 0);
+
+ /* the image, opened through the clone: the loop device holds the clone */
+ ifd = openat(tfd, "img", O_RDWR | O_CREAT | O_EXCL, 0600);
+ ASSERT_GE(ifd, 0);
+ ASSERT_EQ(write_fat12(ifd), 0);
+ n = loop_bind(ifd);
+ close(ifd);
+ ASSERT_GE(n, 0);
+ remember_loop(self, n);
+ snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", n);
+ snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", n);
+
+ /* the loop mount, moved below the clone */
+ ASSERT_EQ(mkdirat(tfd, "mnt", 0755), 0);
+ ASSERT_EQ(sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", self->dev, 0), 0);
+ ASSERT_EQ(sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0), 0);
+ mfd = sys_fsmount(fsfd, 0, 0);
+ ASSERT_GE(mfd, 0);
+ close(fsfd);
+ ASSERT_EQ(sys_move_mount(mfd, "", tfd, "mnt", MOVE_MOUNT_F_EMPTY_PATH), 0);
+ close(mfd);
+
+ ASSERT_EQ(read_sysfs(self->sysfs, buf, sizeof(buf)), 0);
+ ASSERT_NE(strstr(buf, "img"), NULL);
+
+ /* the last fd of the tree: both mounts die, the loop mount connected */
+ close(tfd);
+
+ /* nothing can reach the two mounts any more */
+ ASSERT_FALSE(mounted_anywhere(self->dev));
+
+ assert_loop_released(_metadata, self);
+}
+
+/*
+ * The cycle in two steps: rmdir of vol leaves the loop mount below it
+ * connected while its image's mount, vol2, is alive; then rmdir of vol2
+ * takes that one with the loop mount whose image is on the dead vol. Each
+ * dead mount now owns a loop mount whose filesystem pins the other.
+ */
+TEST_F(loop_cycle, crossed_images_released)
+{
+ int to_parent[2], to_child[2];
+ struct report r = { { -1, -1 } };
+ char sysfs[2][64], dev[2][32];
+ pid_t pid;
+
+ ASSERT_EQ(pipe(to_parent), 0);
+ ASSERT_EQ(pipe(to_child), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ close(to_parent[0]);
+ close(to_child[1]);
+ _exit(crossed_child(to_parent[1], to_child[0]));
+ }
+ close(to_parent[1]);
+ close(to_child[0]);
+
+ if (read(to_parent[0], &r, sizeof(r)) != sizeof(r))
+ CHILD_GAVE_UP(pid);
+ for (int i = 0; i < 2; i++) {
+ remember_loop(self, r.n[i]);
+ snprintf(dev[i], sizeof(dev[i]), "/dev/loop%d", r.n[i]);
+ snprintf(sysfs[i], sizeof(sysfs[i]), "/sys/block/loop%d/loop/backing_file", r.n[i]);
+ }
+
+ /* step one: the mount with the first loop mount below it goes */
+ ASSERT_EQ(rmdir(at("vol")), 0);
+
+ /* step two: the other one, with the loop mount whose image is on the first */
+ ASSERT_EQ(rmdir(at("vol2")), 0);
+
+ /* the child leaves: its namespace and every reference it held are gone */
+ ASSERT_EQ(write(to_child[1], "", 1), 1);
+ CHILD_LEFT(pid);
+ close(to_parent[0]);
+ close(to_child[1]);
+
+ /* nothing can reach the four mounts any more */
+ for (int i = 0; i < 2; i++) {
+ ASSERT_FALSE(mounted_anywhere(dev[i]));
+ strcpy(self->dev, dev[i]);
+ strcpy(self->sysfs, sysfs[i]);
+ assert_loop_released(_metadata, self);
+ }
+}
+
+/*
+ * The holders. Each keeps a file or a path on a mount P for as long as
+ * its own filesystem or device lives, and each has that filesystem or
+ * device mounted at C below P. Once another namespace's rmdir has
+ * detached P with C connected below it and the child is gone, P is owned
+ * by nobody, C by P, and P is kept by whatever the holder still holds.
+ *
+ * P is a loop mount so that its death can be observed: the loop device
+ * gives its backing file up when P's superblock goes. The image sits on
+ * a tmpfs next to P, not above it, so the loop device's own reference is
+ * not part of the picture.
+ */
+enum holder {
+ HOLDER_AUTOFS, /* a FIFO on P as the daemon's pipe */
+ HOLDER_ZRAM, /* a device node on P as the writeback device */
+ HOLDER_ECRYPTFS, /* a directory on P as the lower directory */
+ HOLDER_BINFMT_MISC, /* an executable on P as an 'F' interpreter */
+ HOLDER_FUSE, /* a file on P as a passthrough backing file */
+ HOLDER_ZLOOP, /* a directory on P for the zone files */
+ HOLDER_GADGET, /* a file on P as a mass storage LUN, over dummy_hcd */
+ HOLDER_MD, /* a file on P as an array's bitmap file */
+};
+
+#define HOLDER_IMG at("img/p.img")
+#define HOLDER_MNT at("p")
+#define HOLDER_BELOW at("p/c")
+
+static int write_file(const char *path, const char *s)
+{
+ int fd = open(path, O_WRONLY);
+ ssize_t n;
+
+ if (fd < 0)
+ return -1;
+ n = write(fd, s, strlen(s));
+ close(fd);
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Bind a free loop device to @img; the device number. */
+static int loop_attach(const char *img)
+{
+ int ifd, n;
+
+ ifd = open(img, O_RDWR);
+ if (ifd < 0)
+ return -1;
+ n = loop_bind(ifd);
+ close(ifd);
+ return n;
+}
+
+static int holder_autofs(void)
+{
+ char opts[64];
+ int pfd;
+
+ if (mkfifo(at("p/pipe"), 0600))
+ return CHILD_HOLDER;
+ pfd = open(at("p/pipe"), O_RDWR);
+ if (pfd < 0)
+ return CHILD_HOLDER;
+ snprintf(opts, sizeof(opts), "fd=%d,minproto=5,maxproto=5", pfd);
+ if (mount("autofs", HOLDER_BELOW, "autofs", 0, opts))
+ return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER;
+ close(pfd); /* the mount keeps its own */
+ return CHILD_OK;
+}
+
+/*
+ * zram's writeback device has to be a block device node, and that node has
+ * to be on P. Point it at a second loop device.
+ */
+static void zram_reset(void);
+
+static int holder_zram(void)
+{
+ char dev[32], buf[64];
+ struct stat st;
+ int fd, n;
+
+ if (access("/sys/block/zram0/backing_dev", W_OK))
+ return CHILD_SKIP;
+ /* somebody else's device, leave it alone */
+ if (read_sysfs("/sys/block/zram0/initstate", buf, sizeof(buf)) || buf[0] != '0')
+ return CHILD_SKIP;
+ fd = open(at("img/wb.img"), O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0 || ftruncate(fd, IMAGE_SIZE))
+ return CHILD_HOLDER;
+ close(fd);
+ n = loop_attach(at("img/wb.img"));
+ if (n < 0)
+ return CHILD_HOLDER;
+ /* the minor is not the number of the device when loop has partitions */
+ snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+ if (stat(dev, &st) || mknod(at("p/wbdev"), S_IFBLK | 0600, st.st_rdev))
+ return CHILD_HOLDER;
+ if (write_file("/sys/block/zram0/backing_dev", at("p/wbdev")))
+ return CHILD_HOLDER;
+ snprintf(buf, sizeof(buf), "%d", 4 * 1024 * 1024);
+ if (write_file("/sys/block/zram0/disksize", buf))
+ goto undo;
+ fd = open("/dev/zram0", O_RDWR);
+ if (fd < 0)
+ goto undo;
+ n = write_fat12_4k(fd, 1024); /* zram has 4 KiB blocks */
+ close(fd); /* the reset in undo is refused while the device is open */
+ if (n)
+ goto undo;
+ if (mount_fat("/dev/zram0", HOLDER_BELOW)) {
+ n = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+ zram_reset();
+ return n;
+ }
+ return CHILD_OK;
+undo:
+ zram_reset();
+ return CHILD_HOLDER;
+}
+
+/* Is @alg in /proc/crypto? A module is listed once a request has loaded it. */
+static bool crypto_has(const char *alg)
+{
+ char line[256], name[64];
+ bool found = false;
+ FILE *f;
+
+ f = fopen("/proc/crypto", "r");
+ if (!f)
+ return true; /* no way to tell, assume it is */
+ while (!found && fgets(line, sizeof(line), f))
+ found = sscanf(line, "name : %63s", name) == 1 && !strcmp(name, alg);
+ fclose(f);
+ return found;
+}
+
+/* The kernel's auth token layout, which userspace has to match byte for byte. */
+struct ecryptfs_auth_tok {
+ __u16 version;
+ __u16 token_type;
+ __u32 flags;
+ struct {
+ __u32 flags, encrypted_key_size, decrypted_key_size;
+ __u8 encrypted_key[512], decrypted_key[64];
+ } session_key;
+ __u8 reserved[32];
+ struct {
+ __u32 password_bytes;
+ __s32 hash_algo;
+ __u32 hash_iterations, session_key_encryption_key_bytes, flags;
+ __u8 session_key_encryption_key[64];
+ __u8 signature[17];
+ __u8 salt[8];
+ } password;
+} __attribute__((packed));
+
+#define ECRYPTFS_SIG "0123456789abcdef"
+
+static int holder_ecryptfs(void)
+{
+ struct ecryptfs_auth_tok tok = {
+ .version = 0x0004,
+ .token_type = 0, /* ECRYPTFS_PASSWORD */
+ .password.session_key_encryption_key_bytes = 16,
+ .password.flags = 0x02, /* ECRYPTFS_SESSION_KEY_ENCRYPTION_KEY_SET */
+ .password.signature = ECRYPTFS_SIG,
+ };
+
+ /* a session keyring of this child's own, so that the key goes with it */
+ if (syscall(__NR_keyctl, KEYCTL_JOIN_SESSION_KEYRING, NULL) < 0)
+ return errno == ENOSYS ? CHILD_SKIP : CHILD_HOLDER;
+ if (syscall(__NR_add_key, "user", ECRYPTFS_SIG, &tok, sizeof(tok),
+ KEY_SPEC_SESSION_KEYRING) < 0)
+ return CHILD_HOLDER;
+ if (mkdir(at("p/lower"), 0755))
+ return CHILD_HOLDER;
+ if (mount(at("p/lower"), HOLDER_BELOW, "ecryptfs", 0,
+ "ecryptfs_sig=" ECRYPTFS_SIG ",ecryptfs_cipher=aes,ecryptfs_key_bytes=16")) {
+ /* EINVAL without the cipher; a module is loaded by the attempt */
+ if (errno == ENODEV || (errno == EINVAL && !crypto_has("aes")))
+ return CHILD_SKIP;
+ return CHILD_HOLDER;
+ }
+ return CHILD_OK;
+}
+
+/*
+ * binfmt_misc instances are per user namespace, so the mount below P is
+ * made from a new one, which gets a copy of P.
+ */
+static int holder_binfmt_misc(void)
+{
+ static const char interp[] = "#!/bin/true\n";
+ char reg[PATH_MAX];
+ int out;
+
+ /* 'F' opens the interpreter at registration, nothing runs it here */
+ out = open(at("p/interp"), O_WRONLY | O_CREAT | O_EXCL, 0755);
+ if (out < 0 || write(out, interp, sizeof(interp) - 1) != sizeof(interp) - 1)
+ return CHILD_HOLDER;
+ close(out);
+
+ /* ENOSPC: user.max_user_namespaces is 0, EPERM: an LSM says no */
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) {
+ if (errno == EINVAL || errno == ENOSPC || errno == EPERM)
+ return CHILD_SKIP;
+ return CHILD_HOLDER;
+ }
+ if (write_file("/proc/self/setgroups", "deny") ||
+ write_file("/proc/self/uid_map", "0 0 1") ||
+ write_file("/proc/self/gid_map", "0 0 1"))
+ return CHILD_HOLDER;
+ if (mount("binfmt_misc", HOLDER_BELOW, "binfmt_misc", 0, NULL))
+ return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER;
+ snprintf(reg, sizeof(reg), ":cycle:E::cyc::%s:F", at("p/interp"));
+ if (write_file(at("p/c/register"), reg))
+ return CHILD_HOLDER;
+ return CHILD_OK;
+}
+
+/*
+ * A fuse server that only ever answers FUSE_INIT, with passthrough on, and
+ * then registers a file on P as a backing file. The registration alone
+ * makes the fuse superblock hold the file.
+ */
+static int holder_fuse(void)
+{
+ struct fuse_backing_map map = {};
+ struct fuse_in_header *ih;
+ struct fuse_init_out init = {
+ .major = FUSE_KERNEL_VERSION,
+ .minor = FUSE_KERNEL_MINOR_VERSION,
+ .flags = FUSE_INIT_EXT,
+ .flags2 = FUSE_PASSTHROUGH >> 32,
+ .max_write = 4096,
+ .max_stack_depth = 1,
+ };
+ struct fuse_out_header oh = { .len = sizeof(oh) + sizeof(init) };
+ struct iovec iov[2] = { { &oh, sizeof(oh) }, { &init, sizeof(init) } };
+ char opts[64], buf[8192];
+ int ffd, bfd;
+ ssize_t n;
+
+ bfd = open(at("p/backing"), O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (bfd < 0)
+ return CHILD_HOLDER;
+ ffd = open("/dev/fuse", O_RDWR);
+ if (ffd < 0)
+ return CHILD_SKIP;
+ snprintf(opts, sizeof(opts), "fd=%d,rootmode=40000,user_id=0,group_id=0", ffd);
+ if (mount("fuse", HOLDER_BELOW, "fuse", 0, opts))
+ return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER;
+
+ n = read(ffd, buf, sizeof(buf));
+ ih = (void *)buf;
+ if (n < (ssize_t)sizeof(*ih) || ih->opcode != FUSE_INIT)
+ return CHILD_HOLDER;
+ oh.unique = ih->unique;
+ if (writev(ffd, iov, 2) != (ssize_t)oh.len)
+ return CHILD_HOLDER;
+
+ map.fd = bfd;
+ if (ioctl(ffd, FUSE_DEV_IOC_BACKING_OPEN, &map) < 0) {
+ /* EOPNOTSUPP: no FUSE_PASSTHROUGH, ENOTTY: no such ioctl */
+ if (errno == EPERM || errno == EOPNOTSUPP || errno == ENOTTY)
+ return CHILD_SKIP;
+ return CHILD_HOLDER;
+ }
+ close(bfd); /* the connection keeps its own */
+ return CHILD_OK; /* ffd stays open until the child exits */
+}
+
+/* Wait for a device node the kernel is about to create. */
+static int open_when_there(const char *dev, int flags, int ms)
+{
+ int fd;
+
+ for (; ms > 0; ms -= 100) {
+ fd = open(dev, flags);
+ if (fd >= 0)
+ return fd;
+ usleep(100000);
+ }
+ return -1;
+}
+
+/*
+ * zloop keeps every zone file open. One conventional zone is enough for a
+ * FAT image, the sequential one stays empty.
+ */
+static int holder_zloop(void)
+{
+ char cmd[PATH_MAX];
+ int fd, ret = CHILD_HOLDER;
+
+ if (access("/dev/zloop-control", W_OK))
+ return CHILD_SKIP;
+ /* somebody else's device, leave it alone */
+ if (!access("/sys/block/zloop0", F_OK))
+ return CHILD_SKIP;
+ if (mkdir(at("p/zl"), 0755) || mkdir(at("p/zl/0"), 0755))
+ return CHILD_HOLDER;
+ snprintf(cmd, sizeof(cmd),
+ "add id=0,capacity_mb=8,zone_size_mb=4,conv_zones=1,base_dir=%s",
+ at("p/zl"));
+ if (write_file("/dev/zloop-control", cmd))
+ return CHILD_HOLDER;
+ fd = open_when_there("/dev/zloop0", O_RDWR, 5000);
+ if (fd < 0 || write_fat12_4k(fd, 1024)) /* 4 KiB blocks, like P */
+ goto undo;
+ close(fd);
+ fd = -1;
+ if (mount_fat("/dev/zloop0", HOLDER_BELOW)) {
+ ret = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+ goto undo;
+ }
+ return CHILD_OK;
+undo:
+ if (fd >= 0)
+ close(fd);
+ write_file("/dev/zloop-control", "remove id=0");
+ return ret;
+}
+
+static int mount_configfs(void)
+{
+ if (access("/sys/kernel/config", F_OK))
+ return -1;
+ if (mount("configfs", "/sys/kernel/config", "configfs", 0, NULL) && errno != EBUSY)
+ return -1;
+ return 0;
+}
+
+/* Take the gadget out of configfs again, in the reverse order of its creation. */
+static void gadget_remove(void)
+{
+ if (mount_configfs())
+ return;
+ write_file(gat("/functions/mass_storage.0/lun.0/file"), "\n");
+ write_file(gat("/UDC"), "\n");
+ unlink(gat("/configs/c.1/mass_storage.0"));
+ rmdir(gat("/configs/c.1/strings/0x409"));
+ rmdir(gat("/configs/c.1"));
+ rmdir(gat("/functions/mass_storage.0"));
+ rmdir(gat("/strings/0x409"));
+ rmdir(gat(""));
+}
+
+/* The disk usb-storage created for the gadget, by the LUN's inquiry string. */
+static int find_gadget_disk(char *dev, size_t len, int ms)
+{
+ char path[PATH_MAX], model[64];
+ struct dirent *de;
+ DIR *d;
+
+ for (; ms > 0; ms -= 100, usleep(100000)) {
+ d = opendir("/sys/block");
+ if (!d)
+ return -1;
+ while ((de = readdir(d))) {
+ if (strncmp(de->d_name, "sd", 2))
+ continue;
+ snprintf(path, sizeof(path), "/sys/block/%s/device/model", de->d_name);
+ if (read_sysfs(path, model, sizeof(model)) ||
+ strncmp(model, "File-Stor Gadget", 16))
+ continue;
+ snprintf(dev, len, "/dev/%s", de->d_name);
+ closedir(d);
+ return 0;
+ }
+ closedir(d);
+ }
+ return -1;
+}
+
+/*
+ * A mass storage gadget bound to the dummy UDC, so that this kernel is
+ * also the USB host that sees the LUN as a SCSI disk. sd locks the
+ * medium on open, which is what keeps the LUN's file from being ejected.
+ */
+#define GADGET_STEP(x) do { \
+ if (x) { \
+ fprintf(stderr, "gadget: %s failed: %s\n", #x, strerror(errno)); \
+ gadget_remove(); \
+ return CHILD_HOLDER; \
+ } \
+} while (0)
+
+static int holder_gadget(void)
+{
+ char dev[PATH_MAX];
+ int fd;
+
+ if (mount_configfs())
+ return CHILD_SKIP;
+ if (access("/sys/kernel/config/usb_gadget", F_OK) ||
+ access("/sys/class/udc/dummy_udc.0", F_OK))
+ return CHILD_SKIP;
+
+ fd = open(at("p/lun.img"), O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0 || write_fat12(fd))
+ return CHILD_HOLDER;
+ close(fd);
+
+ GADGET_STEP(mkdir(gat(""), 0755));
+ GADGET_STEP(write_file(gat("/idVendor"), "0x1d6b"));
+ GADGET_STEP(write_file(gat("/idProduct"), "0x0104"));
+ GADGET_STEP(mkdir(gat("/strings/0x409"), 0755));
+ GADGET_STEP(write_file(gat("/strings/0x409/serialnumber"), "1"));
+ GADGET_STEP(write_file(gat("/strings/0x409/manufacturer"), "kselftest"));
+ GADGET_STEP(write_file(gat("/strings/0x409/product"), "cycle"));
+ GADGET_STEP(mkdir(gat("/configs/c.1"), 0755));
+ GADGET_STEP(mkdir(gat("/configs/c.1/strings/0x409"), 0755));
+ GADGET_STEP(write_file(gat("/configs/c.1/strings/0x409/configuration"), "c"));
+ GADGET_STEP(mkdir(gat("/functions/mass_storage.0"), 0755));
+ GADGET_STEP(write_file(gat("/functions/mass_storage.0/lun.0/removable"), "1"));
+ GADGET_STEP(write_file(gat("/functions/mass_storage.0/lun.0/file"), at("p/lun.img")));
+ GADGET_STEP(symlink(gat("/functions/mass_storage.0"), gat("/configs/c.1/mass_storage.0")));
+ /* EBUSY: the controller is somebody else's, leave it alone */
+ if (write_file(gat("/UDC"), "dummy_udc.0")) {
+ fd = errno == EBUSY ? CHILD_SKIP : CHILD_HOLDER;
+ gadget_remove();
+ return fd;
+ }
+
+ /* usb-storage waits a second before it scans the device */
+ if (find_gadget_disk(dev, sizeof(dev), 15000)) {
+ /* without usb-storage and sd the LUN never shows up as a disk */
+ fd = (access("/sys/bus/usb/drivers/usb-storage", F_OK) ||
+ access("/sys/bus/scsi/drivers/sd", F_OK)) ? CHILD_SKIP : CHILD_HOLDER;
+ gadget_remove();
+ return fd;
+ }
+ fd = open_when_there(dev, O_RDONLY, 5000);
+ GADGET_STEP(fd < 0);
+ close(fd);
+ if (mount_fat(dev, HOLDER_BELOW)) {
+ fd = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+ gadget_remove();
+ return fd;
+ }
+ return CHILD_OK;
+}
+
+#define BITMAP_MAGIC 0x6d746962
+
+/*
+ * A RAID1 of one loop device, not persistent, with its bitmap in a file
+ * on P. The bitmap file needs a superblock the kernel accepts; sync_size,
+ * uuid and events are not looked at for a non-persistent array.
+ */
+static int holder_md(void)
+{
+ struct {
+ __u32 magic, version;
+ __u8 uuid[16];
+ __u64 events, events_cleared, sync_size;
+ __u32 state, chunksize, daemon_sleep, write_behind;
+ } bsb = {
+ .magic = BITMAP_MAGIC,
+ .version = 4,
+ .chunksize = 64 * 1024,
+ .daemon_sleep = 5,
+ };
+ mdu_array_info_t info = {
+ .level = 1,
+ .raid_disks = 1,
+ .size = 8 * 1024, /* KiB */
+ .not_persistent = 1,
+ };
+ mdu_disk_info_t disk = {
+ .major = 7,
+ .state = (1 << MD_DISK_ACTIVE) | (1 << MD_DISK_SYNC),
+ };
+ int fd, mdfd, bfd, n, ret = CHILD_HOLDER;
+ char dev[32], buf[4096];
+ struct stat st;
+
+ fd = open(at("img/md.img"), O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0 || ftruncate(fd, 8 * 1024 * 1024))
+ return CHILD_HOLDER;
+ close(fd);
+ n = loop_attach(at("img/md.img"));
+ if (n < 0)
+ return CHILD_HOLDER;
+ /* the minor is not the number of the device when loop has partitions */
+ snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+ if (stat(dev, &st))
+ return CHILD_HOLDER;
+ disk.major = major(st.st_rdev);
+ disk.minor = minor(st.st_rdev);
+
+ bfd = open(at("p/bitmap"), O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (bfd < 0 || write(bfd, &bsb, sizeof(bsb)) != sizeof(bsb) || ftruncate(bfd, 4096))
+ return CHILD_HOLDER;
+
+ if (read_sysfs("/proc/mdstat", buf, sizeof(buf))) {
+ fprintf(stderr, "md: no /proc/mdstat\n");
+ return CHILD_SKIP;
+ }
+ if (!strstr(buf, "[raid1]")) {
+ fprintf(stderr, "md: no raid1 personality: %s\n", buf);
+ return CHILD_SKIP;
+ }
+ /* a node of our own: /dev may not have one and is not ours to change */
+ if (mknod(at("img/md0"), S_IFBLK | 0600, makedev(MD_MAJOR, 0)))
+ return CHILD_HOLDER;
+ mdfd = open(at("img/md0"), O_RDWR);
+ if (mdfd < 0)
+ return CHILD_SKIP;
+ /* somebody else's array: SET_ARRAY_INFO would say EINVAL, not EBUSY */
+ if (!read_sysfs("/sys/block/md0/md/array_state", buf, sizeof(buf)) &&
+ strncmp(buf, "clear", 5)) {
+ fprintf(stderr, "md: md0 is in use: %s", buf);
+ close(mdfd);
+ return CHILD_SKIP;
+ }
+ /* the bitmap ops are only installed once a bitmap type is chosen */
+ write_file("/sys/block/md0/md/bitmap_type", "bitmap");
+ /* EBUSY: somebody else's array, leave it alone */
+ if (ioctl(mdfd, SET_ARRAY_INFO, &info))
+ return errno == EBUSY ? CHILD_SKIP : CHILD_HOLDER;
+ /* the array is ours from here on and has to be stopped on every path */
+ if (ioctl(mdfd, ADD_NEW_DISK, &disk))
+ goto undo;
+ /* attach the bitmap file before the array runs, the way mdadm does */
+ if (ioctl(mdfd, SET_BITMAP_FILE, bfd)) {
+ fprintf(stderr, "md: SET_BITMAP_FILE: %s\n", strerror(errno));
+ if (errno == EINVAL)
+ ret = CHILD_SKIP;
+ goto undo;
+ }
+ close(bfd); /* the array keeps its own */
+ if (ioctl(mdfd, RUN_ARRAY, NULL)) {
+ fprintf(stderr, "md: RUN_ARRAY: %s\n", strerror(errno));
+ goto undo;
+ }
+ if (write_fat12(mdfd))
+ goto undo;
+ close(mdfd);
+ if (mount_fat(at("img/md0"), HOLDER_BELOW)) {
+ ret = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER;
+ mdfd = open(at("img/md0"), O_RDWR);
+ goto undo;
+ }
+ return CHILD_OK;
+undo:
+ if (mdfd >= 0) {
+ ioctl(mdfd, STOP_ARRAY);
+ close(mdfd);
+ }
+ return ret;
+}
+
+/*
+ * A device keeps its file for as long as it is configured, so once nothing
+ * below the dead mount is left the device has to be told to let go. With
+ * the cycle unbroken C's filesystem still holds the device and every one
+ * of these refuses.
+ */
+static int holder_let_go_once(enum holder holder)
+{
+ int fd, ret;
+
+ switch (holder) {
+ case HOLDER_ZRAM:
+ return write_file("/sys/block/zram0/reset", "1");
+ case HOLDER_ZLOOP:
+ return write_file("/dev/zloop-control", "remove id=0");
+ case HOLDER_GADGET:
+ if (mount_configfs())
+ return -1;
+ /* a zero-length write is a no-op for configfs; a newline ejects */
+ return write_file(gat("/functions/mass_storage.0/lun.0/file"), "\n");
+ case HOLDER_MD:
+ fd = open(at("img/md0"), O_RDONLY);
+ if (fd < 0) {
+ /* the child's node went with its tmpfs */
+ mknod(at("md0"), S_IFBLK | 0600, makedev(MD_MAJOR, 0));
+ fd = open(at("md0"), O_RDONLY);
+ }
+ if (fd < 0)
+ return -1;
+ ret = ioctl(fd, STOP_ARRAY);
+ close(fd);
+ return ret;
+ default:
+ return 0;
+ }
+}
+
+static void zram_reset(void)
+{
+ holder_let_go_once(HOLDER_ZRAM);
+}
+
+/*
+ * In its own mount namespace the child puts the image on a tmpfs next to
+ * P, mounts P from a loop device, and sets the holder up with its own
+ * filesystem or device mounted at C below P.
+ */
+static int holder_child(int to_parent, int from_parent, enum holder holder)
+{
+ bool fifo = holder == HOLDER_AUTOFS || holder == HOLDER_ZRAM;
+ /* room for a 4 MiB zone file or a floppy image on P */
+ bool big = holder == HOLDER_ZLOOP || holder == HOLDER_GADGET;
+ struct report r = { { -1, -1 } };
+ char dev[32], c;
+ int ifd, n, ret;
+
+ if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+ return CHILD_NS;
+ if (mkdir(at("img"), 0755) || mount("tmpfs", at("img"), "tmpfs", 0, NULL))
+ return child_fails(CHILD_NS);
+
+ ifd = open(HOLDER_IMG, O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (ifd < 0 || (fifo ? write_minix(ifd) :
+ big ? write_fat12_4k(ifd, 3072) : write_fat12(ifd)))
+ return child_fails(CHILD_IMAGE);
+ close(ifd);
+ n = loop_attach(HOLDER_IMG);
+ if (n < 0)
+ return child_fails(CHILD_LOOP);
+ snprintf(dev, sizeof(dev), "/dev/loop%d", n);
+ if (mkdir(HOLDER_MNT, 0755))
+ return child_fails(CHILD_MOUNT);
+ if (fifo ? mount(dev, HOLDER_MNT, "minix", 0, NULL) : mount_fat(dev, HOLDER_MNT))
+ return child_fails(errno == ENODEV ? CHILD_NOFS : CHILD_MOUNT);
+ if (mkdir(HOLDER_BELOW, 0755))
+ return child_fails(CHILD_MOUNT);
+
+ switch (holder) {
+ case HOLDER_AUTOFS:
+ ret = holder_autofs();
+ break;
+ case HOLDER_ZRAM:
+ ret = holder_zram();
+ break;
+ case HOLDER_ECRYPTFS:
+ ret = holder_ecryptfs();
+ break;
+ case HOLDER_BINFMT_MISC:
+ ret = holder_binfmt_misc();
+ break;
+ case HOLDER_FUSE:
+ ret = holder_fuse();
+ break;
+ case HOLDER_ZLOOP:
+ ret = holder_zloop();
+ break;
+ case HOLDER_GADGET:
+ ret = holder_gadget();
+ break;
+ case HOLDER_MD:
+ ret = holder_md();
+ break;
+ default:
+ ret = CHILD_HOLDER;
+ }
+ if (ret != CHILD_OK)
+ return child_fails(ret);
+
+ /* P's device first, then the one the holder has bound for itself */
+ r.n[0] = n;
+ if (nr_bound > 1)
+ r.n[1] = bound[1];
+ if (write(to_parent, &r, sizeof(r)) != sizeof(r) ||
+ read(from_parent, &c, 1) != 1) {
+ /* the parent went away and will not tell the devices to let go */
+ umount2(HOLDER_BELOW, MNT_DETACH);
+ for (int i = 0; i < 50 && holder_let_go_once(holder); i++)
+ usleep(100000);
+ if (holder == HOLDER_GADGET)
+ gadget_remove();
+ return child_fails(CHILD_PIPE);
+ }
+ return CHILD_OK;
+}
+
+/*
+ * The release of C's filesystem may still be in flight when the child is
+ * gone, and a device that is still held refuses to let go, so try for a
+ * while. With the cycle unbroken it refuses for good.
+ */
+static void holder_let_go(enum holder holder)
+{
+ for (int i = 0; i < 50; i++) {
+ if (!holder_let_go_once(holder))
+ return;
+ usleep(100000);
+ }
+}
+
+/*
+ * rmdir of p from here, where it is a plain directory, detaches
+ * P in the child's namespace with C connected below it. Once the child is
+ * gone the holder's file on P is the only thing left that refers to P,
+ * and it is dropped only when C's filesystem dies, which waits for P.
+ * The loop device backing P tells whether that resolved.
+ */
+static void holder_cycle(struct __test_metadata *_metadata,
+ FIXTURE_DATA(loop_cycle) *self, enum holder holder)
+{
+ struct report r = { { -1, -1 } };
+ int to_parent[2], from_parent[2];
+ pid_t pid;
+
+ ASSERT_EQ(pipe(to_parent), 0);
+ ASSERT_EQ(pipe(from_parent), 0);
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ close(to_parent[0]);
+ close(from_parent[1]);
+ _exit(holder_child(to_parent[1], from_parent[0], holder));
+ }
+ close(to_parent[1]);
+ close(from_parent[0]);
+
+ if (read(to_parent[0], &r, sizeof(r)) != sizeof(r))
+ CHILD_GAVE_UP(pid);
+ self->gadget = holder == HOLDER_GADGET;
+ remember_loop(self, r.n[0]);
+ remember_loop(self, r.n[1]);
+ snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", r.n[0]);
+ snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", r.n[0]);
+
+ ASSERT_EQ(rmdir(HOLDER_MNT), 0);
+ ASSERT_EQ(write(from_parent[1], "x", 1), 1);
+ CHILD_LEFT(pid);
+ close(to_parent[0]);
+ close(from_parent[1]);
+
+ holder_let_go(holder);
+ assert_loop_released(_metadata, self);
+ /* the teardown takes the gadget and the holder's own loop device away */
+}
+
+TEST_F(loop_cycle, autofs_pipe_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_AUTOFS);
+}
+
+TEST_F(loop_cycle, zram_writeback_node_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_ZRAM);
+}
+
+TEST_F(loop_cycle, ecryptfs_lower_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_ECRYPTFS);
+}
+
+TEST_F(loop_cycle, binfmt_misc_interpreter_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_BINFMT_MISC);
+}
+
+TEST_F(loop_cycle, fuse_backing_file_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_FUSE);
+}
+
+TEST_F(loop_cycle, zloop_zone_files_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_ZLOOP);
+}
+
+/* up to 15 s for the disk to show up and 12 s for a device that is held */
+TEST_F_TIMEOUT(loop_cycle, mass_storage_lun_on_dead_mount_released, 120)
+{
+ holder_cycle(_metadata, self, HOLDER_GADGET);
+}
+
+TEST_F(loop_cycle, md_bitmap_file_on_dead_mount_released)
+{
+ holder_cycle(_metadata, self, HOLDER_MD);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c b/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c
new file mode 100644
index 000000000000..e3092454ae68
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c
@@ -0,0 +1,565 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An unmounted mount that would have stayed attached to its unmounted parent
+ * leaves a cover behind instead. A lookup on the parent at the mountpoint
+ * finds knullfs: an empty read-only directory that is shared by every cover
+ * and every kernel thread, so it can't be watched or locked and nothing can
+ * be mounted on it. The mount itself is a root from then on. Where a file
+ * was mounted, the stand-in is an empty regular file of the same instance.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/fanotify.h>
+#include <sys/file.h>
+#include <sys/inotify.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/statvfs.h>
+#include <sys/syscall.h>
+#include <sys/vfs.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+#include "../readdir_hold.h"
+
+#ifndef NULL_FS_MAGIC
+#define NULL_FS_MAGIC 0x4E554C4C
+#endif
+
+#ifndef TMPFS_MAGIC
+#define TMPFS_MAGIC 0x01021994
+#endif
+
+#ifndef F_SETDELEG
+#define F_SETDELEG (F_SETLEASE + 16)
+#endif
+
+/* struct delegation of <linux/fcntl.h>, which doesn't mix with <fcntl.h> */
+struct delegation_req {
+ uint32_t d_flags;
+ uint16_t d_type;
+ uint16_t __pad;
+};
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+/* what the parent asks the child to do */
+#define CMD_RMDIR 'r'
+#define CMD_CLOSE_T 't'
+#define CMD_QUIT 'q'
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int touch(const char *path)
+{
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644);
+ if (fd < 0)
+ return -1;
+ close(fd);
+ return 0;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ prctl(PR_SET_DUMPABLE, 1);
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ if (setgid(0) || setuid(0))
+ return -1;
+ return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+/*
+ * The child mounts P on @base/p, C on P/covered and the file P/src on P/file
+ * in a mount namespace of its own, binds P a second time at @base/q and hands
+ * out descriptors on P and on C. rmdir() of @base/p from here unmounts P
+ * together with C and the file bind. P and C are held by the descriptors and
+ * the unmounted children leave their covers behind. On request the child
+ * removes C's mountpoint through the bind.
+ *
+ * It also mounts T on @base/t with a child on T/covered, binds T at @base/u
+ * with a second child on the same dentry and hands out descriptors on T and
+ * U. rmdir() of both from here leaves two covers on one mountpoint. On request
+ * the child lets go of T.
+ */
+static int cover_child(const char *base, int to_parent, int from_parent)
+{
+ char p[PATH_LEN], c[PATH_LEN], q[PATH_LEN], qc[PATH_LEN];
+ char f[PATH_LEN], src[PATH_LEN], t[PATH_LEN], u[PATH_LEN], tc[PATH_LEN];
+ int fds[4], ret;
+ char cmd;
+
+ if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL))
+ return 1;
+ snprintf(p, sizeof(p), "%s/p", base);
+ if (mkdir(p, 0755) || mount("tmpfs", p, "tmpfs", 0, NULL))
+ return 2;
+ snprintf(c, sizeof(c), "%s/p/covered", base);
+ if (mkdir(c, 0755) || mount("tmpfs", c, "tmpfs", 0, NULL))
+ return 3;
+ snprintf(f, sizeof(f), "%s/p/file", base);
+ snprintf(src, sizeof(src), "%s/p/src", base);
+ if (touch(src) || touch(f) || mount(src, f, NULL, MS_BIND, NULL))
+ return 4;
+ snprintf(q, sizeof(q), "%s/q", base);
+ if (mkdir(q, 0755) || mount(p, q, NULL, MS_BIND, NULL))
+ return 5;
+ snprintf(t, sizeof(t), "%s/t", base);
+ if (mkdir(t, 0755) || mount("tmpfs", t, "tmpfs", 0, NULL))
+ return 6;
+ snprintf(tc, sizeof(tc), "%s/t/covered", base);
+ if (mkdir(tc, 0755) || mount("tmpfs", tc, "tmpfs", 0, NULL))
+ return 7;
+ snprintf(u, sizeof(u), "%s/u", base);
+ if (mkdir(u, 0755) || mount(t, u, NULL, MS_BIND, NULL))
+ return 8;
+ snprintf(tc, sizeof(tc), "%s/u/covered", base);
+ if (mount("tmpfs", tc, "tmpfs", 0, NULL))
+ return 9;
+ fds[0] = open(p, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ fds[1] = open(c, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ fds[2] = open(t, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ fds[3] = open(u, O_PATH | O_DIRECTORY | O_CLOEXEC);
+ if (fds[0] < 0 || fds[1] < 0 || fds[2] < 0 || fds[3] < 0)
+ return 10;
+ if (write(to_parent, fds, sizeof(fds)) != sizeof(fds))
+ return 11;
+
+ snprintf(qc, sizeof(qc), "%s/q/covered", base);
+ for (;;) {
+ if (read(from_parent, &cmd, 1) != 1)
+ return 12;
+ switch (cmd) {
+ case CMD_RMDIR:
+ ret = rmdir(qc) ? errno : 0;
+ if (write(to_parent, &ret, sizeof(ret)) != sizeof(ret))
+ return 13;
+ break;
+ case CMD_CLOSE_T:
+ ret = close(fds[2]) ? errno : 0;
+ if (write(to_parent, &ret, sizeof(ret)) != sizeof(ret))
+ return 13;
+ break;
+ case CMD_QUIT:
+ return 0;
+ default:
+ return 14;
+ }
+ }
+}
+
+FIXTURE(mount_cover) {
+ char base[DIR_LEN];
+ pid_t child;
+ int to_child;
+ int from_child;
+ int dfd; /* P, unmounted, held */
+ int cfd; /* C, unmounted, held */
+ int fd; /* what a lookup on P finds at C's mountpoint */
+ int tfd; /* T, unmounted, held */
+ int ufd; /* U, a bind of T, unmounted, held */
+ int fan; /* fanotify group from before the user namespace, or -1 */
+ struct readdir_hold hold; /* likewise from before, uffd -1 without */
+};
+
+/*
+ * Everything the setup has made. The harness does not run the teardown
+ * when an assertion of the setup fails, so the setup calls this itself.
+ */
+static void cover_cleanup(FIXTURE_DATA(mount_cover) *self)
+{
+ char cmd = CMD_QUIT;
+ int status;
+
+ if (self->fd >= 0)
+ close(self->fd);
+ if (self->cfd >= 0)
+ close(self->cfd);
+ if (self->dfd >= 0)
+ close(self->dfd);
+ if (self->tfd >= 0)
+ close(self->tfd);
+ if (self->ufd >= 0)
+ close(self->ufd);
+ if (self->fan >= 0)
+ close(self->fan);
+ readdir_hold_destroy(&self->hold);
+ if (self->child > 0) {
+ if (write(self->to_child, &cmd, 1) != 1)
+ kill(self->child, SIGKILL);
+ waitpid(self->child, &status, 0);
+ }
+ if (self->to_child >= 0)
+ close(self->to_child);
+ if (self->from_child >= 0)
+ close(self->from_child);
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+}
+
+static int same_file(int fd1, int fd2)
+{
+ struct stat st1, st2;
+
+ if (fstat(fd1, &st1) || fstat(fd2, &st2))
+ return 0;
+ return st1.st_dev == st2.st_dev && st1.st_ino == st2.st_ino;
+}
+
+FIXTURE_SETUP(mount_cover)
+{
+ int to_parent[2], to_child[2], fds[4], pidfd, status;
+ char dir[PATH_LEN];
+ struct statfs sf;
+
+ self->child = 0;
+ self->to_child = self->from_child = -1;
+ self->dfd = self->cfd = self->fd = self->tfd = self->ufd = -1;
+
+ /*
+ * A group for plain events takes CAP_SYS_ADMIN in the initial user
+ * namespace, so get one before that is gone. An inode mark can be
+ * added to it from anywhere.
+ */
+ self->fan = fanotify_init(FAN_CLASS_NOTIF | FAN_CLOEXEC, O_RDONLY);
+ /* same for the userfaultfd that holds a readdir in its fault */
+ readdir_hold_init(&self->hold);
+
+ snprintf(self->base, sizeof(self->base), "/tmp/mount_cover.XXXXXX");
+ ASSERT_NE(mkdtemp(self->base), NULL);
+ if (enter_userns()) {
+ cover_cleanup(self);
+ SKIP(return, "test requires user namespaces");
+ }
+ ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0)
+ cover_cleanup(self);
+
+ snprintf(dir, sizeof(dir), "%s/p", self->base);
+ ASSERT_EQ(pipe(to_parent), 0)
+ cover_cleanup(self);
+ ASSERT_EQ(pipe(to_child), 0)
+ cover_cleanup(self);
+ self->child = fork();
+ ASSERT_GE(self->child, 0)
+ cover_cleanup(self);
+ if (self->child == 0) {
+ close(to_parent[0]);
+ close(to_child[1]);
+ _exit(cover_child(self->base, to_parent[1], to_child[0]));
+ }
+ close(to_parent[1]);
+ close(to_child[0]);
+ self->to_child = to_child[1];
+ self->from_child = to_parent[0];
+ if (read(self->from_child, fds, sizeof(fds)) != sizeof(fds)) {
+ pid_t pid = self->child;
+
+ waitpid(pid, &status, 0);
+ self->child = 0;
+ cover_cleanup(self);
+ ASSERT_TRUE(false)
+ TH_LOG("child failed to set up: %s %d",
+ WIFEXITED(status) ? "step" : "signal",
+ WIFEXITED(status) ? WEXITSTATUS(status) : WTERMSIG(status));
+ }
+
+ pidfd = syscall(__NR_pidfd_open, self->child, 0);
+ ASSERT_GE(pidfd, 0)
+ cover_cleanup(self);
+ self->dfd = syscall(__NR_pidfd_getfd, pidfd, fds[0], 0);
+ self->cfd = syscall(__NR_pidfd_getfd, pidfd, fds[1], 0);
+ self->tfd = syscall(__NR_pidfd_getfd, pidfd, fds[2], 0);
+ self->ufd = syscall(__NR_pidfd_getfd, pidfd, fds[3], 0);
+ close(pidfd);
+ ASSERT_GE(self->dfd, 0)
+ cover_cleanup(self);
+ ASSERT_GE(self->cfd, 0)
+ cover_cleanup(self);
+ ASSERT_GE(self->tfd, 0)
+ cover_cleanup(self);
+ ASSERT_GE(self->ufd, 0)
+ cover_cleanup(self);
+
+ /* unmounts P and C, C leaves its cover behind */
+ ASSERT_EQ(rmdir(dir), 0)
+ cover_cleanup(self);
+ /* unmounts T and U with their children, two covers on one mountpoint */
+ snprintf(dir, sizeof(dir), "%s/t", self->base);
+ ASSERT_EQ(rmdir(dir), 0)
+ cover_cleanup(self);
+ snprintf(dir, sizeof(dir), "%s/u", self->base);
+ ASSERT_EQ(rmdir(dir), 0)
+ cover_cleanup(self);
+
+ self->fd = openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(self->fd, 0)
+ cover_cleanup(self);
+ ASSERT_EQ(fstatfs(self->fd, &sf), 0)
+ cover_cleanup(self);
+ ASSERT_EQ(sf.f_type, NULL_FS_MAGIC)
+ cover_cleanup(self);
+}
+
+FIXTURE_TEARDOWN(mount_cover)
+{
+ cover_cleanup(self);
+}
+
+TEST_F(mount_cover, not_watchable)
+{
+ char p[PATH_LEN];
+ int ifd;
+
+ snprintf(p, sizeof(p), "/proc/self/fd/%d", self->fd);
+
+ ifd = inotify_init1(IN_CLOEXEC);
+ if (ifd < 0 && errno == ENOSYS) {
+ TH_LOG("no inotify in this kernel, skipping that part");
+ } else {
+ ASSERT_GE(ifd, 0);
+ EXPECT_EQ(inotify_add_watch(ifd, p, IN_OPEN), -1);
+ EXPECT_EQ(errno, EINVAL);
+ close(ifd);
+ }
+
+ /* dnotify ends up at the same place */
+ EXPECT_EQ(fcntl(self->fd, F_NOTIFY, DN_ACCESS), -1);
+ EXPECT_EQ(errno, EINVAL);
+
+ if (self->fan < 0) {
+ TH_LOG("no fanotify group without CAP_SYS_ADMIN in the initial user namespace, skipping the fanotify part");
+ return;
+ }
+ EXPECT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD, FAN_OPEN, self->fd, NULL), -1);
+ EXPECT_EQ(errno, EINVAL);
+}
+
+TEST_F(mount_cover, not_lockable)
+{
+ struct flock fl = {
+ .l_type = F_RDLCK,
+ .l_whence = SEEK_SET,
+ };
+
+ EXPECT_EQ(flock(self->fd, LOCK_EX | LOCK_NB), -1);
+ EXPECT_EQ(errno, ENOLCK);
+ EXPECT_EQ(fcntl(self->fd, F_SETLK, &fl), -1);
+ EXPECT_EQ(errno, ENOLCK);
+ EXPECT_EQ(fcntl(self->fd, F_GETLK, &fl), -1);
+ EXPECT_EQ(errno, ENOLCK);
+}
+
+/* a lease is refused for the owner and for everybody else alike */
+TEST_F(mount_cover, not_leasable)
+{
+ struct delegation_req deleg = {
+ .d_type = F_RDLCK,
+ };
+
+ EXPECT_EQ(fcntl(self->fd, F_SETLEASE, F_RDLCK), -1);
+ EXPECT_TRUE(errno == EINVAL || errno == EACCES);
+ EXPECT_EQ(fcntl(self->fd, F_SETDELEG, &deleg), -1);
+ EXPECT_TRUE(errno == EINVAL || errno == EACCES);
+}
+
+TEST_F(mount_cover, not_mountable)
+{
+ char p[PATH_LEN];
+
+ snprintf(p, sizeof(p), "/proc/self/fd/%d", self->fd);
+ EXPECT_EQ(mount("tmpfs", p, "tmpfs", 0, NULL), -1);
+ EXPECT_EQ(errno, ENOENT);
+}
+
+TEST_F(mount_cover, read_only)
+{
+ struct statvfs sv;
+
+ EXPECT_EQ(mkdirat(self->fd, "x", 0755), -1);
+ EXPECT_EQ(errno, ENOENT);
+ EXPECT_EQ(fchmod(self->fd, 0777), -1);
+ EXPECT_EQ(errno, EROFS);
+ /* the immutable inode is checked before the read-only mount */
+ EXPECT_EQ(faccessat(self->fd, ".", W_OK, 0), -1);
+ EXPECT_EQ(errno, EPERM);
+ ASSERT_EQ(fstatvfs(self->fd, &sv), 0);
+ EXPECT_TRUE(sv.f_flag & ST_RDONLY);
+}
+
+/* the stand-in is a root of its own, ".." stays put */
+TEST_F(mount_cover, island)
+{
+ int fd;
+
+ fd = openat(self->fd, "..", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ EXPECT_TRUE(same_file(fd, self->fd));
+ close(fd);
+}
+
+/*
+ * C is alive for as long as the child holds it, but it can't be reached
+ * through P anymore and it's a root of its own as well.
+ */
+TEST_F(mount_cover, held_child_detached)
+{
+ struct statfs sf;
+ int fd;
+
+ ASSERT_EQ(fstatfs(self->cfd, &sf), 0);
+ EXPECT_EQ(sf.f_type, TMPFS_MAGIC);
+
+ fd = openat(self->cfd, "..", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ EXPECT_TRUE(same_file(fd, self->cfd));
+ close(fd);
+
+ fd = openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstatfs(fd, &sf), 0);
+ EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+ close(fd);
+}
+
+/*
+ * The cover goes with its mountpoint: once the child has removed C's
+ * mountpoint through the bind of P, the name is gone from P as well.
+ * What was opened through the cover before stays open.
+ */
+TEST_F(mount_cover, cover_goes_with_mountpoint)
+{
+ char cmd = CMD_RMDIR;
+ struct statfs sf;
+ int ret;
+
+ ASSERT_EQ(write(self->to_child, &cmd, 1), 1);
+ ASSERT_EQ(read(self->from_child, &ret, sizeof(ret)), sizeof(ret));
+ ASSERT_EQ(ret, 0);
+
+ EXPECT_EQ(openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC), -1);
+ EXPECT_EQ(errno, ENOENT);
+ ASSERT_EQ(fstatfs(self->fd, &sf), 0);
+ EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+}
+
+/*
+ * A readdir of the stand-in holds nothing that others wait for: one holder
+ * sticks in the page fault of its buffer, and a create and a lookup of
+ * another come back meanwhile.
+ */
+TEST_F(mount_cover, readdir_blocks_nobody)
+{
+ bool stalled;
+
+ if (self->hold.uffd < 0)
+ SKIP(return, "test requires userfaultfd");
+ ASSERT_EQ(readdir_hold_check(&self->hold, self->fd, &stalled), 0);
+ EXPECT_FALSE(stalled);
+}
+
+/* where a file was mounted, the stand-in is an empty regular file */
+TEST_F(mount_cover, file_stand_in)
+{
+ char src[PATH_LEN], p[PATH_LEN];
+ struct statfs sf;
+ struct stat st;
+ char c;
+ int fd;
+
+ fd = openat(self->dfd, "file", O_RDONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstat(fd, &st), 0);
+ EXPECT_TRUE(S_ISREG(st.st_mode));
+ ASSERT_EQ(fstatfs(fd, &sf), 0);
+ EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+ /* the two stand-ins are two inodes */
+ EXPECT_FALSE(same_file(fd, self->fd));
+ EXPECT_EQ(read(fd, &c, 1), 0);
+ EXPECT_EQ(flock(fd, LOCK_EX | LOCK_NB), -1);
+ EXPECT_EQ(errno, ENOLCK);
+
+ snprintf(src, sizeof(src), "%s/src", self->base);
+ ASSERT_EQ(touch(src), 0);
+ snprintf(p, sizeof(p), "/proc/self/fd/%d", fd);
+ EXPECT_EQ(mount(src, p, NULL, MS_BIND, NULL), -1);
+ EXPECT_EQ(errno, ENOENT);
+ close(fd);
+
+ EXPECT_EQ(openat(self->dfd, "file", O_WRONLY | O_CLOEXEC), -1);
+ EXPECT_EQ(errno, EPERM);
+ EXPECT_EQ(openat(self->dfd, "file", O_RDONLY | O_DIRECTORY | O_CLOEXEC), -1);
+ EXPECT_EQ(errno, ENOTDIR);
+}
+
+/*
+ * Two unmounted parents left covers on the same dentry. A lookup on
+ * either finds a stand-in, and U's cover stays when T goes.
+ */
+TEST_F(mount_cover, shared_mountpoint)
+{
+ char cmd = CMD_CLOSE_T;
+ struct statfs sf;
+ int fd, ret;
+
+ fd = openat(self->tfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstatfs(fd, &sf), 0);
+ EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+ close(fd);
+
+ fd = openat(self->ufd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstatfs(fd, &sf), 0);
+ EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+ close(fd);
+
+ /* the last references to T go, with T its cover */
+ ASSERT_EQ(write(self->to_child, &cmd, 1), 1);
+ ASSERT_EQ(read(self->from_child, &ret, sizeof(ret)), sizeof(ret));
+ ASSERT_EQ(ret, 0);
+ close(self->tfd);
+ self->tfd = -1;
+
+ fd = openat(self->ufd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(fstatfs(fd, &sf), 0);
+ EXPECT_EQ(sf.f_type, NULL_FS_MAGIC);
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c b/tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c
new file mode 100644
index 000000000000..0928a584eddc
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c
@@ -0,0 +1,193 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A recursive bind mount of a namespace file that lives in another mount
+ * namespace copies whatever is stacked on top of it there. If that includes
+ * the file of the caller's own mount namespace, or of an older one, the copy
+ * would pin the namespace it is put in forever. The bind mount has to be
+ * refused, a plain bind mount of the file itself still works.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+/* Child exit codes. */
+enum {
+ CHILD_OK,
+ CHILD_UNSHARE, /* could not create the newer mount namespace */
+ CHILD_PIPE, /* the parent went away */
+ CHILD_TMPFS, /* could not mount the tmpfs in the new namespace */
+ CHILD_REC_ALLOWED, /* the recursive bind mount was not refused */
+ CHILD_REC_ERRNO, /* it was refused with the wrong error */
+ CHILD_PLAIN_REFUSED, /* the plain bind mount of the file was refused */
+};
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int create_file(const char *path)
+{
+ int fd = open(path, O_WRONLY | O_CREAT | O_CLOEXEC, 0644);
+
+ if (fd < 0)
+ return -1;
+ close(fd);
+ return 0;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ if (setgid(0) || setuid(0))
+ return -1;
+ return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+static int send_msg(int fd, char msg)
+{
+ return write(fd, &msg, 1) == 1 ? 0 : -1;
+}
+
+static char recv_msg(int fd)
+{
+ char msg;
+
+ if (read(fd, &msg, 1) != 1)
+ return 0;
+ return msg;
+}
+
+FIXTURE(nsfs_rbind_loop) {
+ char dir[DIR_LEN];
+ char x[PATH_LEN];
+};
+
+FIXTURE_SETUP(nsfs_rbind_loop)
+{
+ snprintf(self->dir, sizeof(self->dir), "/tmp/nsfs_rbind_loop.XXXXXX");
+ ASSERT_NE(mkdtemp(self->dir), NULL);
+ if (enter_userns()) {
+ rmdir(self->dir);
+ SKIP(return, "test requires user namespaces");
+ }
+ ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0);
+ snprintf(self->x, sizeof(self->x), "%s/x", self->dir);
+ ASSERT_EQ(create_file(self->x), 0);
+}
+
+FIXTURE_TEARDOWN(nsfs_rbind_loop)
+{
+ umount2(self->dir, MNT_DETACH);
+ rmdir(self->dir);
+}
+
+/*
+ * The child in the newer mount namespace binds the network namespace file
+ * mount of the parent through @fd. Recursively that would copy the mount of
+ * its own mount namespace file that the parent stacked on top.
+ */
+static int newer_ns_child(const char *dir, int fd, int to_parent, int from_parent)
+{
+ char src[32], y[PATH_LEN];
+
+ if (unshare(CLONE_NEWNS))
+ return CHILD_UNSHARE;
+ if (send_msg(to_parent, 'r') || recv_msg(from_parent) != 'g')
+ return CHILD_PIPE;
+
+ snprintf(y, sizeof(y), "%s/y", dir);
+ if (mount("tmpfs", y, "tmpfs", 0, NULL))
+ return CHILD_TMPFS;
+ snprintf(src, sizeof(src), "/proc/self/fd/%d", fd);
+ snprintf(y, sizeof(y), "%s/y/f", dir);
+ if (create_file(y))
+ return CHILD_TMPFS;
+
+ if (!mount(src, y, NULL, MS_BIND | MS_REC, NULL))
+ return CHILD_REC_ALLOWED;
+ if (errno != EINVAL)
+ return CHILD_REC_ERRNO;
+ if (mount(src, y, NULL, MS_BIND, NULL))
+ return CHILD_PLAIN_REFUSED;
+ umount2(y, MNT_DETACH);
+ return CHILD_OK;
+}
+
+TEST_F(nsfs_rbind_loop, own_ns_file_below_foreign_source)
+{
+ int to_child[2], to_parent[2], fd, status;
+ char p[PATH_LEN];
+ pid_t pid;
+
+ snprintf(p, sizeof(p), "%s/y", self->dir);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+
+ /* M, a mount of our network namespace file, held by a descriptor */
+ ASSERT_EQ(mount("/proc/self/ns/net", self->x, NULL, MS_BIND, NULL), 0);
+ fd = open(self->x, O_PATH | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+
+ ASSERT_EQ(pipe(to_child), 0);
+ ASSERT_EQ(pipe(to_parent), 0);
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ close(to_child[1]);
+ close(to_parent[0]);
+ _exit(newer_ns_child(self->dir, fd, to_parent[1], to_child[0]));
+ }
+ close(to_child[0]);
+ close(to_parent[1]);
+ ASSERT_EQ(recv_msg(to_parent[0]), 'r');
+
+ /* the child's mount namespace file on top of M */
+ snprintf(p, sizeof(p), "/proc/%d/ns/mnt", pid);
+ ASSERT_EQ(mount(p, self->x, NULL, MS_BIND, NULL), 0);
+
+ ASSERT_EQ(send_msg(to_child[1], 'g'), 0);
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), CHILD_OK);
+
+ close(fd);
+ ASSERT_EQ(umount2(self->x, MNT_DETACH), 0);
+ ASSERT_EQ(umount2(self->x, MNT_DETACH), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c b/tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c
new file mode 100644
index 000000000000..ea1b33fa0bf8
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c
@@ -0,0 +1,190 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A mount namespace file bind-mounted on top of the mount that is moved
+ * onto a shared mount isn't copied to the peers and slaves. The mount that
+ * already sits at the destination in a slave has to end up on top of the
+ * propagated copy, not below the root of the mount namespace file where no
+ * path walk ever finds it.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <signal.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "../wrappers.h"
+#include "../../kselftest_harness.h"
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int create_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0644);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* the first bytes of the file at @path, "" if it can't be read */
+static const char *read_file(const char *path, char *buf, size_t len)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_RDONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = read(fd, buf, len - 1);
+ close(fd);
+ }
+ buf[n > 0 ? n : 0] = '\0';
+ return buf;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ if (setgid(0) || setuid(0))
+ return -1;
+ return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+FIXTURE(overmount_ns_file) {
+ char dir[DIR_LEN];
+ pid_t child;
+};
+
+FIXTURE_SETUP(overmount_ns_file)
+{
+ self->child = -1;
+ snprintf(self->dir, sizeof(self->dir), "/tmp/overmount_ns_file.XXXXXX");
+ ASSERT_NE(mkdtemp(self->dir), NULL);
+ if (enter_userns()) {
+ rmdir(self->dir);
+ SKIP(return, "test requires user namespaces");
+ }
+ ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0);
+}
+
+FIXTURE_TEARDOWN(overmount_ns_file)
+{
+ if (self->child > 0) {
+ kill(self->child, SIGKILL);
+ waitpid(self->child, NULL, 0);
+ }
+ umount2(self->dir, MNT_DETACH);
+ rmdir(self->dir);
+}
+
+/*
+ * A is a shared tmpfs and B its slave with Q, a bind mount of a file, on
+ * B/file. S is a bind mount of a file with N, a bind mount of a newer mount
+ * namespace's file, on top of it. S is moved onto A/file. Its copy S' lands
+ * on B/file below Q, without N. B/file keeps reading Q and once Q is
+ * unmounted it reads S'.
+ */
+TEST_F(overmount_ns_file, existing_mount_stays_on_top)
+{
+ char a[PATH_LEN], b[PATH_LEN], s[PATH_LEN], p[PATH_LEN], buf[16];
+ int fd, pfd[2];
+ char c;
+
+ snprintf(a, sizeof(a), "%s/A", self->dir);
+ snprintf(b, sizeof(b), "%s/B", self->dir);
+ snprintf(s, sizeof(s), "%s/s", self->dir);
+ ASSERT_EQ(mkdir(a, 0755), 0);
+ ASSERT_EQ(mkdir(b, 0755), 0);
+ ASSERT_EQ(mkdir(s, 0755), 0);
+
+ /* A shared, B its slave, Q on B/file */
+ ASSERT_EQ(mount("tmpfs", a, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mount(NULL, a, NULL, MS_SHARED, NULL), 0);
+ snprintf(p, sizeof(p), "%s/A/file", self->dir);
+ ASSERT_EQ(create_file(p, "A"), 0);
+ ASSERT_EQ(mount(a, b, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(mount(NULL, b, NULL, MS_SLAVE, NULL), 0);
+ snprintf(p, sizeof(p), "%s/Q", self->dir);
+ ASSERT_EQ(create_file(p, "Q"), 0);
+ snprintf(b, sizeof(b), "%s/B/file", self->dir);
+ ASSERT_EQ(mount(p, b, NULL, MS_BIND, NULL), 0);
+
+ /* S on s/f, pinned by a file descriptor before N goes on top */
+ ASSERT_EQ(mount("tmpfs", s, "tmpfs", 0, NULL), 0);
+ snprintf(p, sizeof(p), "%s/s/f", self->dir);
+ ASSERT_EQ(create_file(p, "f"), 0);
+ snprintf(s, sizeof(s), "%s/s/S", self->dir);
+ ASSERT_EQ(create_file(s, "S"), 0);
+ ASSERT_EQ(mount(s, p, NULL, MS_BIND, NULL), 0);
+ fd = open(p, O_PATH | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+
+ /* a newer mount namespace whose file can be bound */
+ ASSERT_EQ(pipe(pfd), 0);
+ self->child = fork();
+ ASSERT_GE(self->child, 0);
+ if (self->child == 0) {
+ close(pfd[0]);
+ if (unshare(CLONE_NEWNS) || write(pfd[1], "r", 1) != 1)
+ _exit(1);
+ pause();
+ _exit(0);
+ }
+ close(pfd[1]);
+ ASSERT_EQ(read(pfd[0], &c, 1), 1);
+ close(pfd[0]);
+ snprintf(s, sizeof(s), "/proc/%d/ns/mnt", self->child);
+ ASSERT_EQ(mount(s, p, NULL, MS_BIND, NULL), 0);
+
+ ASSERT_STREQ(read_file(b, buf, sizeof(buf)), "Q");
+
+ snprintf(a, sizeof(a), "%s/A/file", self->dir);
+ ASSERT_EQ(sys_move_mount(fd, "", AT_FDCWD, a, MOVE_MOUNT_F_EMPTY_PATH), 0);
+ close(fd);
+
+ /* Q is still on top of the copy in B and can be unmounted */
+ EXPECT_STREQ(read_file(b, buf, sizeof(buf)), "Q");
+ EXPECT_EQ(umount2(b, 0), 0);
+ EXPECT_STREQ(read_file(b, buf, sizeof(buf)), "S");
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c b/tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c
new file mode 100644
index 000000000000..46a56b722f3e
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c
@@ -0,0 +1,181 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * propagate_umount() slides a surviving overmount off a mount that is being
+ * unmounted and unmounts that mount right after. If a file descriptor keeps
+ * the unmounted mount alive its ->overmount is left pointing at the moved
+ * mount, which is freed on its own schedule. move_mount(MOVE_MOUNT_BENEATH)
+ * with such a file descriptor as the target walks ->overmount in
+ * topmost_overmount() before it checks that the target is still mounted and
+ * must not step into the freed mount.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+
+#include "../wrappers.h"
+#include "../../kselftest_harness.h"
+
+#ifndef MOVE_MOUNT_BENEATH
+#define MOVE_MOUNT_BENEATH 0x00000200
+#endif
+
+#ifndef FSCONFIG_CMD_CREATE
+#define FSCONFIG_CMD_CREATE 6
+#endif
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ if (setgid(0) || setuid(0))
+ return -1;
+ return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+/* A detached tmpfs mount to serve as the source of a move. */
+static int detached_tmpfs(void)
+{
+ int sfd, mfd;
+
+ sfd = sys_fsopen("tmpfs", 0);
+ if (sfd < 0)
+ return -1;
+ if (sys_fsconfig(sfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0)) {
+ close(sfd);
+ return -1;
+ }
+ mfd = sys_fsmount(sfd, 0, 0);
+ close(sfd);
+ return mfd;
+}
+
+FIXTURE(overmount_reparent) {
+ char dir[DIR_LEN];
+};
+
+FIXTURE_SETUP(overmount_reparent)
+{
+ snprintf(self->dir, sizeof(self->dir), "/tmp/overmount_reparent.XXXXXX");
+ ASSERT_NE(mkdtemp(self->dir), NULL);
+ if (enter_userns()) {
+ rmdir(self->dir);
+ SKIP(return, "test requires user namespaces");
+ }
+}
+
+FIXTURE_TEARDOWN(overmount_reparent)
+{
+ char p[PATH_LEN];
+
+ snprintf(p, sizeof(p), "%s/b", self->dir);
+ umount2(p, MNT_DETACH);
+ snprintf(p, sizeof(p), "%s/a", self->dir);
+ umount2(p, MNT_DETACH);
+}
+
+/*
+ * A shared mount base_a is bind-mounted to base_b as its peer. A mount X on
+ * base_a/mp propagates a copy X_b onto base_b/mp. A file descriptor pins X_b,
+ * then X_b is made private and an overmount is stacked on its root.
+ *
+ * A lazy unmount of X propagates to X_b: X_b is committed to the unmount while
+ * its overmount survives, so propagate_umount() reparents the overmount onto
+ * base_b and unmounts X_b, which the file descriptor keeps alive. Unmounting
+ * the reparented overmount frees it. X_b->overmount now dangles unless
+ * mnt_change_mountpoint() reset it.
+ *
+ * move_mount(MOVE_MOUNT_BENEATH) through the file descriptor makes
+ * do_lock_mount() follow X_b->overmount in topmost_overmount() before it
+ * notices that X_b is no longer mounted. With the pointer reset the move fails
+ * cleanly with ENOENT; otherwise it reads the freed overmount.
+ */
+TEST_F(overmount_reparent, dead_overmount_holder_not_followed)
+{
+ char a[PATH_LEN], b[PATH_LEN], amp[PATH_LEN], bmp[PATH_LEN];
+ int fd, mfd, ret;
+
+ snprintf(a, sizeof(a), "%s/a", self->dir);
+ snprintf(b, sizeof(b), "%s/b", self->dir);
+ ASSERT_EQ(mkdir(a, 0755), 0);
+ ASSERT_EQ(mkdir(b, 0755), 0);
+
+ ASSERT_EQ(mount("tmpfs", a, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mount(NULL, a, NULL, MS_SHARED, NULL), 0);
+ ASSERT_EQ(mount(a, b, NULL, MS_BIND, NULL), 0);
+
+ snprintf(amp, sizeof(amp), "%s/a/mp", self->dir);
+ snprintf(bmp, sizeof(bmp), "%s/b/mp", self->dir);
+ /* base_a and base_b share one superblock, so this dir is in both. */
+ ASSERT_EQ(mkdir(amp, 0755), 0);
+
+ /* X on base_a/mp propagates a copy X_b onto base_b/mp. */
+ ASSERT_EQ(mount("tmpfs", amp, "tmpfs", 0, NULL), 0);
+
+ /* Pin X_b before anything is stacked on top of it. */
+ fd = open(bmp, O_PATH | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+
+ /* Keep the overmount local to X_b. */
+ ASSERT_EQ(mount(NULL, bmp, NULL, MS_PRIVATE, NULL), 0);
+
+ /* The overmount on X_b's root: X_b->overmount points at it. */
+ ASSERT_EQ(mount("tmpfs", bmp, "tmpfs", 0, NULL), 0);
+
+ /* Reparents the overmount onto base_b and unmounts X_b. */
+ ASSERT_EQ(umount2(amp, MNT_DETACH), 0);
+
+ /* Free the reparented overmount. */
+ ASSERT_EQ(umount2(bmp, MNT_DETACH), 0);
+ usleep(100000);
+
+ mfd = detached_tmpfs();
+ ASSERT_GE(mfd, 0);
+
+ ret = sys_move_mount(mfd, "", fd, "",
+ MOVE_MOUNT_F_EMPTY_PATH | MOVE_MOUNT_T_EMPTY_PATH |
+ MOVE_MOUNT_BENEATH);
+ EXPECT_EQ(ret, -1);
+ EXPECT_EQ(errno, ENOENT);
+
+ close(mfd);
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/mount_cycle/settings b/tools/testing/selftests/filesystems/mount_cycle/settings
new file mode 100644
index 000000000000..694d70710ff0
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/settings
@@ -0,0 +1 @@
+timeout=300
diff --git a/tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c b/tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c
new file mode 100644
index 000000000000..233f8ea92efc
--- /dev/null
+++ b/tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c
@@ -0,0 +1,484 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * An unmounted mount tree that a file descriptor keeps alive is put and
+ * vacated behind the root's back, so nothing may walk it under
+ * namespace_sem: it can't be copied recursively and its propagation can't
+ * be changed. The mount at its root alone can still be copied and that copy
+ * is an ordinary mount.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+#include <unistd.h>
+#include <linux/mount.h>
+
+#include "../wrappers.h"
+#include "../../kselftest_harness.h"
+
+#ifndef __NR_mount_setattr
+#define __NR_mount_setattr 442
+#endif
+
+#ifndef __NR_pidfd_open
+#define __NR_pidfd_open 434
+#endif
+
+#define PATH_LEN 64
+
+static inline int sys_mount_setattr(int dfd, const char *path, unsigned int flags,
+ struct mount_attr *attr, size_t size)
+{
+ return syscall(__NR_mount_setattr, dfd, path, flags, attr, size);
+}
+
+static inline int sys_pidfd_open(pid_t pid, unsigned int flags)
+{
+ return syscall(__NR_pidfd_open, pid, flags);
+}
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int touch(const char *path)
+{
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644);
+ if (fd < 0)
+ return -1;
+ close(fd);
+ return 0;
+}
+
+/* Become root in a new user namespace with a private mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ if (setgid(0) || setuid(0))
+ return -1;
+ return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+/* A private mount namespace of our own, all mounts private. */
+static int own_mntns(void)
+{
+ if (unshare(CLONE_NEWNS))
+ return -1;
+ return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+FIXTURE(unmounted_tree) {
+ char dir[PATH_LEN];
+};
+
+FIXTURE_SETUP(unmounted_tree)
+{
+ snprintf(self->dir, sizeof(self->dir), "/tmp/unmounted_tree.XXXXXX");
+ ASSERT_NE(mkdtemp(self->dir), NULL);
+ if (enter_userns()) {
+ rmdir(self->dir);
+ SKIP(return, "test requires user namespaces");
+ }
+ ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0);
+}
+
+FIXTURE_TEARDOWN(unmounted_tree)
+{
+ umount2(self->dir, MNT_DETACH);
+ rmdir(self->dir);
+}
+
+/* A recursive copy of the mount @fd refers to must fail with EINVAL. */
+static void assert_not_walked(struct __test_metadata *_metadata, int fd,
+ const char *target)
+{
+ char link[PATH_LEN];
+ int tfd;
+
+ snprintf(link, sizeof(link), "/proc/self/fd/%d", fd);
+ EXPECT_EQ(mount(link, target, NULL, MS_BIND | MS_REC, NULL), -1);
+ EXPECT_EQ(errno, EINVAL);
+
+ tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | AT_RECURSIVE | OPEN_TREE_CLONE |
+ OPEN_TREE_CLOEXEC);
+ EXPECT_LT(tfd, 0);
+ EXPECT_EQ(errno, EINVAL);
+ if (tfd >= 0)
+ close(tfd);
+}
+
+/* The mount @fd refers to can be copied on its own and mounted on @target. */
+static void assert_root_copied(struct __test_metadata *_metadata, int fd,
+ const char *target)
+{
+ int tfd;
+
+ tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(tfd, 0);
+ ASSERT_EQ(sys_move_mount(tfd, "", AT_FDCWD, target, MOVE_MOUNT_F_EMPTY_PATH), 0);
+ close(tfd);
+ ASSERT_EQ(umount2(target, MNT_DETACH), 0);
+}
+
+/*
+ * A lazily unmounted bind mount of an nsfs file, kept alive by an fd, can't
+ * be copied recursively anymore. Copying the mount itself still works and
+ * so does copying a mounted bind mount of the same file.
+ */
+TEST_F(unmounted_tree, detached_nsfs_bind_not_walked)
+{
+ char x[PATH_LEN], y[PATH_LEN], z[PATH_LEN], link[PATH_LEN];
+ int fd, tfd;
+
+ snprintf(x, sizeof(x), "%s/x", self->dir);
+ snprintf(y, sizeof(y), "%s/y", self->dir);
+ snprintf(z, sizeof(z), "%s/z", self->dir);
+ ASSERT_EQ(touch(x), 0);
+ ASSERT_EQ(touch(y), 0);
+ ASSERT_EQ(touch(z), 0);
+
+ ASSERT_EQ(mount("/proc/self/ns/net", x, NULL, MS_BIND, NULL), 0);
+ fd = open(x, O_PATH | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(umount2(x, MNT_DETACH), 0);
+
+ assert_not_walked(_metadata, fd, y);
+
+ snprintf(link, sizeof(link), "/proc/self/fd/%d", fd);
+ ASSERT_EQ(mount(link, y, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(umount2(y, MNT_DETACH), 0);
+ assert_root_copied(_metadata, fd, y);
+ /* opening it is fine too */
+ tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(tfd, 0);
+ close(tfd);
+ close(fd);
+
+ ASSERT_EQ(mount("/proc/self/ns/net", y, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(mount(y, z, NULL, MS_BIND | MS_REC, NULL), 0);
+ tfd = sys_open_tree(AT_FDCWD, y, AT_RECURSIVE | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(tfd, 0);
+ close(tfd);
+}
+
+/* Bind-mount the pidfd @pidfd on @target the way pidfd_bind_mount does. */
+static int bind_pidfd(int pidfd, const char *target)
+{
+ int tfd, ret;
+
+ tfd = sys_open_tree(pidfd, "", AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ if (tfd < 0)
+ return -1;
+ ret = sys_move_mount(tfd, "", AT_FDCWD, target, MOVE_MOUNT_F_EMPTY_PATH);
+ close(tfd);
+ return ret;
+}
+
+/* The same for a bind mount of a pidfd. */
+TEST_F(unmounted_tree, detached_pidfs_bind_not_walked)
+{
+ char x[PATH_LEN], y[PATH_LEN];
+ int pidfd, fd, tfd;
+
+ snprintf(x, sizeof(x), "%s/x", self->dir);
+ snprintf(y, sizeof(y), "%s/y", self->dir);
+ ASSERT_EQ(touch(x), 0);
+ ASSERT_EQ(touch(y), 0);
+
+ pidfd = sys_pidfd_open(getpid(), 0);
+ ASSERT_GE(pidfd, 0);
+ ASSERT_EQ(bind_pidfd(pidfd, x), 0);
+ fd = open(x, O_PATH | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ ASSERT_EQ(umount2(x, MNT_DETACH), 0);
+
+ assert_not_walked(_metadata, fd, y);
+ assert_root_copied(_metadata, fd, y);
+ close(fd);
+
+ ASSERT_EQ(bind_pidfd(pidfd, y), 0);
+ tfd = sys_open_tree(AT_FDCWD, y, AT_RECURSIVE | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(tfd, 0);
+ close(tfd);
+ close(pidfd);
+}
+
+/* What a child in its own mount namespace reports back. */
+enum {
+ CHILD_OK,
+ CHILD_NS, /* could not set up the mount namespace */
+ CHILD_MOUNT, /* could not set up the mounts */
+ CHILD_PIPE, /* the parent went away */
+ CHILD_WALKED, /* the detached tree was copied recursively */
+ CHILD_ROOT_REFUSED, /* the mount at its root wasn't copied */
+ CHILD_STILL_MOUNTED, /* the copy survived the unlink of its mountpoint */
+ CHILD_ERRNO, /* refused, but not with EINVAL */
+};
+
+static const char *child_reason(int code)
+{
+ static const char *const reasons[] = {
+ [CHILD_OK] = "ok",
+ [CHILD_NS] = "could not set up the mount namespace",
+ [CHILD_MOUNT] = "could not set up the mounts",
+ [CHILD_PIPE] = "the parent went away",
+ [CHILD_WALKED] = "the detached tree was copied recursively",
+ [CHILD_ROOT_REFUSED] = "the mount at its root wasn't copied",
+ [CHILD_STILL_MOUNTED] = "the copy survived the unlink of its mountpoint",
+ [CHILD_ERRNO] = "refused, but not with EINVAL",
+ };
+
+ if (code < 0 || code >= (int)(sizeof(reasons) / sizeof(reasons[0])))
+ return "child died";
+ return reasons[code];
+}
+
+struct child_args {
+ const char *x; /* the nsfs bind mount goes here */
+ const char *s; /* a file to bind on top of it */
+ const char *y; /* a target to copy to */
+};
+
+/*
+ * Run @fn in a child in a mount namespace of its own. Once the child
+ * reports that its mounts are in place, unlink @victim here, where it is a
+ * plain file, and let the child carry on. Returns what the child reported.
+ */
+static int run_child(struct __test_metadata *_metadata,
+ int (*fn)(const struct child_args *, int, int),
+ const struct child_args *a, const char *victim)
+{
+ int to_parent[2], to_child[2], status;
+ pid_t pid;
+ char c;
+
+ if (pipe(to_parent) || pipe(to_child))
+ return -1;
+ pid = fork();
+ if (pid < 0)
+ return -1;
+ if (pid == 0) {
+ close(to_parent[0]);
+ close(to_child[1]);
+ _exit(fn(a, to_parent[1], to_child[0]));
+ }
+ close(to_parent[1]);
+ close(to_child[0]);
+ if (read(to_parent[0], &c, 1) == 1) {
+ /* not a mountpoint in this namespace, so the file can go */
+ EXPECT_EQ(unlink(victim), 0);
+ EXPECT_EQ(write(to_child[1], "", 1), 1);
+ }
+ close(to_parent[0]);
+ close(to_child[1]);
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+/*
+ * An nsfs bind mount on @x, an fd on it and a bind mount of @s on top. Once
+ * the parent has unlinked @x underneath, the first mount is detached and the
+ * second one, which stayed attached to it, has been vacated. The tree may
+ * not be walked, the mount at its root may still be copied.
+ */
+static int vacant_child(const struct child_args *a, int to_parent, int from_parent)
+{
+ char link[PATH_LEN];
+ int fd, tfd;
+ char c;
+
+ if (own_mntns())
+ return CHILD_NS;
+ if (mount("/proc/self/ns/net", a->x, NULL, MS_BIND, NULL))
+ return CHILD_MOUNT;
+ fd = open(a->x, O_PATH | O_CLOEXEC);
+ if (fd < 0 || mount(a->s, a->x, NULL, MS_BIND, NULL))
+ return CHILD_MOUNT;
+
+ if (write(to_parent, "", 1) != 1 || read(from_parent, &c, 1) != 1)
+ return CHILD_PIPE;
+
+ tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | AT_RECURSIVE | OPEN_TREE_CLONE |
+ OPEN_TREE_CLOEXEC);
+ if (tfd >= 0) {
+ close(tfd);
+ return CHILD_WALKED;
+ }
+ if (errno != EINVAL)
+ return CHILD_ERRNO;
+ snprintf(link, sizeof(link), "/proc/self/fd/%d", fd);
+ if (!mount(link, a->y, NULL, MS_BIND | MS_REC, NULL))
+ return CHILD_WALKED;
+ if (errno != EINVAL)
+ return CHILD_ERRNO;
+
+ tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ if (tfd < 0)
+ return CHILD_ROOT_REFUSED;
+ close(tfd);
+ /* the last reference on the detached mount collects the vacant one */
+ close(fd);
+ return CHILD_OK;
+}
+
+/*
+ * unlink() of the child's mountpoint from here, where it is a plain file,
+ * detaches the child's nsfs bind mount with the mount on top left attached
+ * to it. The mount on top loses its last reference right there and is
+ * vacated. The child must not be able to walk that tree.
+ */
+TEST_F(unmounted_tree, vacant_child_not_copied)
+{
+ char x[PATH_LEN], s[PATH_LEN], y[PATH_LEN];
+ struct child_args a = { .x = x, .s = s, .y = y };
+ int ret;
+
+ snprintf(x, sizeof(x), "%s/x", self->dir);
+ snprintf(s, sizeof(s), "%s/s", self->dir);
+ snprintf(y, sizeof(y), "%s/y", self->dir);
+ ASSERT_EQ(touch(x), 0);
+ ASSERT_EQ(touch(s), 0);
+ ASSERT_EQ(touch(y), 0);
+
+ ret = run_child(_metadata, vacant_child, &a, x);
+ ASSERT_EQ(ret, CHILD_OK)
+ TH_LOG("child: %s", child_reason(ret));
+}
+
+/*
+ * An nsfs bind mount on @x, lazily unmounted but kept alive by an fd, and a
+ * copy of it on @y. That copy is a mount like any other: once the parent has
+ * unlinked @y underneath, it is unmounted and its attributes can't be changed
+ * anymore. A copy that took the flag along is only unhooked, stays in the
+ * namespace and still takes the change.
+ */
+static int ordinary_copy(const struct child_args *a, int to_parent, int from_parent)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_NOSUID,
+ };
+ char link[PATH_LEN];
+ int fd, fdy;
+ char c;
+
+ if (own_mntns())
+ return CHILD_NS;
+ if (mount("/proc/self/ns/net", a->x, NULL, MS_BIND, NULL))
+ return CHILD_MOUNT;
+ fd = open(a->x, O_PATH | O_CLOEXEC);
+ if (fd < 0 || umount2(a->x, MNT_DETACH))
+ return CHILD_MOUNT;
+ snprintf(link, sizeof(link), "/proc/self/fd/%d", fd);
+ if (mount(link, a->y, NULL, MS_BIND, NULL))
+ return CHILD_MOUNT;
+ fdy = open(a->y, O_PATH | O_CLOEXEC);
+ if (fdy < 0)
+ return CHILD_MOUNT;
+
+ if (write(to_parent, "", 1) != 1 || read(from_parent, &c, 1) != 1)
+ return CHILD_PIPE;
+
+ if (!sys_mount_setattr(fdy, "", AT_EMPTY_PATH, &attr, sizeof(attr)))
+ return CHILD_STILL_MOUNTED;
+ if (errno != EINVAL)
+ return CHILD_ERRNO;
+ close(fdy);
+ close(fd);
+ return CHILD_OK;
+}
+
+/*
+ * A copy of a detached nsfs bind mount must not take after its source and
+ * read as unmounted. unlink() of its mountpoint from here unmounts it like
+ * any other mount, so the child can't change its attributes anymore.
+ */
+TEST_F(unmounted_tree, copy_of_detached_bind_is_ordinary)
+{
+ char x[PATH_LEN], y[PATH_LEN];
+ struct child_args a = { .x = x, .y = y };
+ int ret;
+
+ snprintf(x, sizeof(x), "%s/x", self->dir);
+ snprintf(y, sizeof(y), "%s/y", self->dir);
+ ASSERT_EQ(touch(x), 0);
+ ASSERT_EQ(touch(y), 0);
+
+ ret = run_child(_metadata, ordinary_copy, &a, y);
+ ASSERT_EQ(ret, CHILD_OK)
+ TH_LOG("child: %s", child_reason(ret));
+}
+
+/*
+ * mount_setattr() may change the propagation of a detached tree while that
+ * tree is still the root of an anonymous mount namespace, but not once the
+ * tree has been dissolved and only a second fd keeps its root alive.
+ */
+TEST_F(unmounted_tree, detached_tree_setattr_refused)
+{
+ struct mount_attr attr = {
+ .propagation = MS_SHARED,
+ };
+ char a[PATH_LEN], b[PATH_LEN];
+ int tfd, fd;
+
+ snprintf(a, sizeof(a), "%s/a", self->dir);
+ snprintf(b, sizeof(b), "%s/a/b", self->dir);
+ ASSERT_EQ(mkdir(a, 0755), 0);
+ ASSERT_EQ(mount("tmpfs", a, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mkdir(b, 0755), 0);
+ ASSERT_EQ(mount("tmpfs", b, "tmpfs", 0, NULL), 0);
+
+ tfd = sys_open_tree(AT_FDCWD, self->dir, AT_RECURSIVE | OPEN_TREE_CLONE |
+ OPEN_TREE_CLOEXEC);
+ ASSERT_GE(tfd, 0);
+ ASSERT_EQ(sys_mount_setattr(tfd, "", AT_EMPTY_PATH | AT_RECURSIVE, &attr, sizeof(attr)), 0);
+ attr.propagation = MS_PRIVATE;
+ ASSERT_EQ(sys_mount_setattr(tfd, "", AT_EMPTY_PATH | AT_RECURSIVE, &attr, sizeof(attr)), 0);
+
+ /* a second reference on the root, then dissolve the tree */
+ fd = sys_open_tree(tfd, "", AT_EMPTY_PATH | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ close(tfd);
+
+ attr.propagation = MS_SHARED;
+ ASSERT_EQ(sys_mount_setattr(fd, "", AT_EMPTY_PATH | AT_RECURSIVE, &attr, sizeof(attr)), -1);
+ ASSERT_EQ(errno, EINVAL);
+ attr.propagation = MS_PRIVATE;
+ ASSERT_EQ(sys_mount_setattr(fd, "", AT_EMPTY_PATH, &attr, sizeof(attr)), -1);
+ ASSERT_EQ(errno, EINVAL);
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/open_tree_ns/.gitignore b/tools/testing/selftests/filesystems/open_tree_ns/.gitignore
index fb12b93fbcaa..76f95c0ae5ef 100644
--- a/tools/testing/selftests/filesystems/open_tree_ns/.gitignore
+++ b/tools/testing/selftests/filesystems/open_tree_ns/.gitignore
@@ -1 +1,2 @@
open_tree_ns_test
+open_tree_ns_covered_test
diff --git a/tools/testing/selftests/filesystems/open_tree_ns/Makefile b/tools/testing/selftests/filesystems/open_tree_ns/Makefile
index 4976ed1d7d4a..fb2aa77b6edb 100644
--- a/tools/testing/selftests/filesystems/open_tree_ns/Makefile
+++ b/tools/testing/selftests/filesystems/open_tree_ns/Makefile
@@ -1,5 +1,5 @@
# SPDX-License-Identifier: GPL-2.0
-TEST_GEN_PROGS := open_tree_ns_test
+TEST_GEN_PROGS := open_tree_ns_test open_tree_ns_covered_test
CFLAGS += -Wall -O0 -g $(KHDR_INCLUDES) $(TOOLS_INCLUDES)
LDLIBS := -lcap
diff --git a/tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c b/tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c
new file mode 100644
index 000000000000..27f5cddfb9ef
--- /dev/null
+++ b/tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c
@@ -0,0 +1,195 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * open_tree(OPEN_TREE_NAMESPACE) by a caller that isn't privileged over the
+ * mount namespace it copies from must not reveal what the mounts below the
+ * copied mount cover. Without AT_RECURSIVE the copy is refused when there's
+ * anything mounted below the requested directory. With AT_RECURSIVE
+ * unbindable mounts are copied as well.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/wait.h>
+
+#include "../wrappers.h"
+#include "../../kselftest_harness.h"
+
+#ifndef OPEN_TREE_NAMESPACE
+#define OPEN_TREE_NAMESPACE (1 << 1)
+#endif
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+/* Child exit codes. */
+enum {
+ CHILD_OK,
+ CHILD_USERNS, /* could not create the child user namespace */
+ CHILD_NONREC_ALLOWED, /* the non-recursive copy was not refused */
+ CHILD_NONREC_ERRNO, /* it was refused with the wrong error */
+ CHILD_REC_REFUSED, /* the recursive copy failed */
+ CHILD_SETNS, /* could not enter the new mount namespace */
+ CHILD_NO_COVER, /* the covering mount is missing in the copy */
+ CHILD_REVEALED, /* the covered file is visible in the copy */
+};
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int create_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0644);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Become root in a new user namespace, the uid @uid is mapped to 0. */
+static int enter_userns(uid_t uid, gid_t gid)
+{
+ char map[32];
+
+ if (unshare(CLONE_NEWUSER))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ return setgid(0) || setuid(0) ? -1 : 0;
+}
+
+FIXTURE(open_tree_ns_covered) {
+ char dir[DIR_LEN];
+ char cover[PATH_LEN];
+};
+
+FIXTURE_VARIANT(open_tree_ns_covered) {
+ int propagation;
+};
+
+FIXTURE_VARIANT_ADD(open_tree_ns_covered, private_cover) {
+ .propagation = MS_PRIVATE,
+};
+
+FIXTURE_VARIANT_ADD(open_tree_ns_covered, unbindable_cover) {
+ .propagation = MS_UNBINDABLE,
+};
+
+/*
+ * Root in a user namespace owns a private mount namespace with a tmpfs
+ * on @dir and a second tmpfs covering @dir/covered/under.txt.
+ */
+FIXTURE_SETUP(open_tree_ns_covered)
+{
+ char p[PATH_LEN];
+ int fd;
+
+ snprintf(self->dir, sizeof(self->dir), "/tmp/open_tree_ns_covered.XXXXXX");
+ ASSERT_NE(mkdtemp(self->dir), NULL);
+ if (enter_userns(getuid(), getgid()) || unshare(CLONE_NEWNS)) {
+ rmdir(self->dir);
+ SKIP(return, "test requires user namespaces");
+ }
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0);
+
+ /* the flag has to exist before its rules can be checked */
+ fd = sys_open_tree(AT_FDCWD, self->dir,
+ OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC);
+ if (fd < 0 && (errno == EINVAL || errno == ENOSYS)) {
+ umount2(self->dir, MNT_DETACH);
+ rmdir(self->dir);
+ SKIP(return, "OPEN_TREE_NAMESPACE not supported");
+ }
+ ASSERT_GE(fd, 0);
+ close(fd);
+
+ snprintf(self->cover, sizeof(self->cover), "%s/covered", self->dir);
+ ASSERT_EQ(mkdir(self->cover, 0755), 0);
+ snprintf(p, sizeof(p), "%s/covered/under.txt", self->dir);
+ ASSERT_EQ(create_file(p, "hidden"), 0);
+ ASSERT_EQ(mount("tmpfs", self->cover, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mount(NULL, self->cover, NULL, variant->propagation, NULL), 0);
+}
+
+FIXTURE_TEARDOWN(open_tree_ns_covered)
+{
+ umount2(self->dir, MNT_DETACH);
+ rmdir(self->dir);
+}
+
+/* A caller in a new user namespace that doesn't own the mount namespace. */
+static int foreign_child(const char *dir)
+{
+ struct stat st;
+ int fd;
+
+ if (enter_userns(0, 0))
+ return CHILD_USERNS;
+
+ fd = sys_open_tree(AT_FDCWD, dir, OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC);
+ if (fd >= 0)
+ return CHILD_NONREC_ALLOWED;
+ if (errno != EINVAL)
+ return CHILD_NONREC_ERRNO;
+
+ fd = sys_open_tree(AT_FDCWD, dir,
+ OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC | AT_RECURSIVE);
+ if (fd < 0)
+ return CHILD_REC_REFUSED;
+ if (setns(fd, CLONE_NEWNS))
+ return CHILD_SETNS;
+ if (stat("/covered", &st))
+ return CHILD_NO_COVER;
+ if (!access("/covered/under.txt", F_OK) || errno != ENOENT)
+ return CHILD_REVEALED;
+ return CHILD_OK;
+}
+
+TEST_F(open_tree_ns_covered, foreign_user_namespace)
+{
+ int status, fd;
+ pid_t pid;
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(foreign_child(self->dir));
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), CHILD_OK);
+
+ /* the owner of the mount namespace keeps bind mount semantics */
+ fd = sys_open_tree(AT_FDCWD, self->dir,
+ OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/overlayfs/.gitignore b/tools/testing/selftests/filesystems/overlayfs/.gitignore
index 077f7a128168..b343cc430051 100644
--- a/tools/testing/selftests/filesystems/overlayfs/.gitignore
+++ b/tools/testing/selftests/filesystems/overlayfs/.gitignore
@@ -2,3 +2,4 @@
dev_in_maps
set_layers_via_fds
idmapped_mounts
+automount_in_layer
diff --git a/tools/testing/selftests/filesystems/overlayfs/Makefile b/tools/testing/selftests/filesystems/overlayfs/Makefile
index b3185f684add..382a59bda5e7 100644
--- a/tools/testing/selftests/filesystems/overlayfs/Makefile
+++ b/tools/testing/selftests/filesystems/overlayfs/Makefile
@@ -9,6 +9,7 @@ LOCAL_HDRS += ../wrappers.h log.h
TEST_GEN_PROGS := dev_in_maps
TEST_GEN_PROGS += set_layers_via_fds
TEST_GEN_PROGS += idmapped_mounts
+TEST_GEN_PROGS += automount_in_layer
include ../../lib.mk
diff --git a/tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c b/tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c
new file mode 100644
index 000000000000..0476c4582bdf
--- /dev/null
+++ b/tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c
@@ -0,0 +1,174 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A layer of an overlay is a private clone of the mount it was given and
+ * belongs to no mount namespace. fanotify hands out descriptors on it. An
+ * automount triggered through one has no namespace to go into: the open has
+ * to fail, not oops with namespace_sem held.
+ */
+#define _GNU_SOURCE
+#include <dirent.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <poll.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <linux/magic.h>
+#include <sys/fanotify.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/vfs.h>
+
+#include "../../kselftest_harness.h"
+
+#define DIR_LEN 64
+#define PATH_LEN 192
+
+static bool have_fs(const char *name)
+{
+ char line[128];
+ bool found = false;
+ FILE *f;
+
+ f = fopen("/proc/filesystems", "re");
+ if (!f)
+ return false;
+ while (fgets(line, sizeof(line), f)) {
+ char *nl = strchr(line, '\n');
+ char *tab = strchr(line, '\t');
+
+ if (nl)
+ *nl = 0;
+ if (tab && !strcmp(tab + 1, name))
+ found = true;
+ }
+ fclose(f);
+ return found;
+}
+
+static int mnt_id_of(int fd)
+{
+ char path[64], buf[4096], *p;
+ ssize_t n;
+ int info;
+
+ snprintf(path, sizeof(path), "/proc/self/fdinfo/%d", fd);
+ info = open(path, O_RDONLY | O_CLOEXEC);
+ if (info < 0)
+ return -1;
+ n = read(info, buf, sizeof(buf) - 1);
+ close(info);
+ if (n <= 0)
+ return -1;
+ buf[n] = 0;
+ p = strstr(buf, "mnt_id:");
+ return p ? atoi(p + strlen("mnt_id:")) : -1;
+}
+
+static bool on_debugfs(int fd)
+{
+ struct statfs sf;
+
+ return !fstatfs(fd, &sf) && sf.f_type == DEBUGFS_MAGIC;
+}
+
+FIXTURE(layer) {
+ char base[DIR_LEN];
+ int fan;
+ int evfd; /* on the layer clone of the lower debugfs */
+};
+
+FIXTURE_SETUP(layer)
+{
+ char lower[PATH_LEN], other[PATH_LEN], ovl[PATH_LEN], opts[2 * PATH_LEN + 16];
+ struct fanotify_event_metadata *ev;
+ struct pollfd pfd;
+ char buf[4096];
+ int lfd, lower_id;
+ ssize_t n;
+ DIR *d;
+
+ self->fan = -1;
+ self->evfd = -1;
+ if (geteuid())
+ SKIP(return, "test requires root");
+ if (!have_fs("debugfs") || !have_fs("tracefs"))
+ SKIP(return, "test requires debugfs with the tracefs automount");
+ if (!have_fs("overlay"))
+ SKIP(return, "test requires overlayfs");
+
+ snprintf(self->base, sizeof(self->base), "/tmp/layer.XXXXXX");
+ ASSERT_NE(mkdtemp(self->base), NULL);
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, "mode=0755"), 0);
+ snprintf(lower, sizeof(lower), "%s/lower", self->base);
+ snprintf(other, sizeof(other), "%s/other", self->base);
+ snprintf(ovl, sizeof(ovl), "%s/ovl", self->base);
+ ASSERT_EQ(mkdir(lower, 0755), 0);
+ ASSERT_EQ(mkdir(other, 0755), 0);
+ ASSERT_EQ(mkdir(ovl, 0755), 0);
+ ASSERT_EQ(mount("debugfs", lower, "debugfs", 0, NULL), 0);
+ snprintf(opts, sizeof(opts), "lowerdir=%s:%s", lower, other);
+ ASSERT_EQ(mount("overlay", ovl, "overlay", MS_RDONLY, opts), 0);
+
+ self->fan = fanotify_init(FAN_CLASS_NOTIF | FAN_NONBLOCK | FAN_CLOEXEC,
+ O_RDONLY | O_CLOEXEC);
+ ASSERT_GE(self->fan, 0);
+ ASSERT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD | FAN_MARK_FILESYSTEM,
+ FAN_OPEN | FAN_ONDIR, AT_FDCWD, lower), 0);
+ lfd = open(lower, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(lfd, 0);
+ lower_id = mnt_id_of(lfd);
+ close(lfd);
+
+ /* the overlay opens its lower directory through the layer clone */
+ d = opendir(ovl);
+ ASSERT_NE(d, NULL);
+ closedir(d);
+ pfd.fd = self->fan;
+ pfd.events = POLLIN;
+ ASSERT_EQ(poll(&pfd, 1, 5000), 1);
+ n = read(self->fan, buf, sizeof(buf));
+ ASSERT_GT(n, 0);
+ for (ev = (void *)buf; FAN_EVENT_OK(ev, n); ev = FAN_EVENT_NEXT(ev, n)) {
+ if (ev->fd < 0)
+ continue;
+ if (self->evfd < 0 && on_debugfs(ev->fd) &&
+ mnt_id_of(ev->fd) != lower_id)
+ self->evfd = ev->fd;
+ else
+ close(ev->fd);
+ }
+ ASSERT_GE(self->evfd, 0)
+ TH_LOG("no event on the layer clone");
+}
+
+FIXTURE_TEARDOWN(layer)
+{
+ if (self->evfd >= 0)
+ close(self->evfd);
+ if (self->fan >= 0)
+ close(self->fan);
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+}
+
+TEST_F(layer, automount_below_the_clone_is_refused)
+{
+ struct stat st;
+ int fd;
+
+ /* the automount point is there */
+ ASSERT_EQ(fstatat(self->evfd, "tracing", &st, AT_NO_AUTOMOUNT), 0);
+ /* the clone is in no namespace, so the automount has nowhere to go */
+ fd = openat(self->evfd, "tracing", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ EXPECT_LT(fd, 0);
+ EXPECT_EQ(errno, EINVAL);
+ if (fd >= 0)
+ close(fd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/readdir_hold.h b/tools/testing/selftests/filesystems/readdir_hold.h
new file mode 100644
index 000000000000..57eee1bef470
--- /dev/null
+++ b/tools/testing/selftests/filesystems/readdir_hold.h
@@ -0,0 +1,224 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Hold a readdir of a directory in the page fault of its buffer and see
+ * whether a create and a lookup in that directory wait for it. For a
+ * directory that is permanently empty they must not.
+ */
+#ifndef __SELFTESTS_READDIR_HOLD_H
+#define __SELFTESTS_READDIR_HOLD_H
+
+#include <errno.h>
+#include <fcntl.h>
+#include <poll.h>
+#include <pthread.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <linux/userfaultfd.h>
+#include <sys/ioctl.h>
+#include <sys/mman.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+
+#define HOLD_FAULT_MS 5000 /* for the reader to reach the fault */
+#define HOLD_QUEUE_MS 2000 /* for the create to queue up behind it */
+#define HOLD_LOOKUP_MS 5000 /* for the lookup to come back */
+
+struct readdir_hold {
+ int uffd;
+ int taskdir; /* /proc/self/task, from before any namespace change */
+ long page_size;
+ char *page; /* faults until released */
+ int dfd;
+ pid_t creator_tid;
+ int created;
+ int done[2]; /* the finder writes a byte when it is back */
+};
+
+/*
+ * The fault happens in the kernel, so the userfaultfd needs CAP_SYS_PTRACE
+ * in the initial user namespace or vm.unprivileged_userfaultfd. Call this
+ * before entering a user namespace.
+ */
+static inline int readdir_hold_init(struct readdir_hold *h)
+{
+ struct uffdio_api api = { .api = UFFD_API };
+
+ h->page = MAP_FAILED;
+ h->page_size = sysconf(_SC_PAGESIZE);
+ h->taskdir = open("/proc/self/task", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ h->uffd = syscall(__NR_userfaultfd, O_CLOEXEC | O_NONBLOCK);
+ if (h->uffd < 0 || h->taskdir < 0 || ioctl(h->uffd, UFFDIO_API, &api)) {
+ if (h->uffd >= 0)
+ close(h->uffd);
+ if (h->taskdir >= 0)
+ close(h->taskdir);
+ h->uffd = h->taskdir = -1;
+ return -1;
+ }
+ return 0;
+}
+
+static inline void readdir_hold_destroy(struct readdir_hold *h)
+{
+ if (h->uffd >= 0)
+ close(h->uffd);
+ if (h->taskdir >= 0)
+ close(h->taskdir);
+ h->uffd = h->taskdir = -1;
+}
+
+static inline void *readdir_hold_reader(void *arg)
+{
+ struct readdir_hold *h = arg;
+
+ /* the first byte written to the buffer faults until released */
+ syscall(__NR_getdents64, h->dfd, h->page, h->page_size);
+ return NULL;
+}
+
+static inline void *readdir_hold_creator(void *arg)
+{
+ struct readdir_hold *h = arg;
+
+ h->creator_tid = syscall(__NR_gettid);
+ /* takes the directory lock exclusive before it fails */
+ mkdirat(h->dfd, "x", 0755);
+ __atomic_store_n(&h->created, 1, __ATOMIC_RELEASE);
+ return NULL;
+}
+
+static inline void *readdir_hold_finder(void *arg)
+{
+ struct readdir_hold *h = arg;
+ int fd;
+
+ /* a lookup that misses the dcache takes the lock shared */
+ fd = openat(h->dfd, "no_such_name", O_RDONLY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+ if (write(h->done[1], "x", 1) != 1)
+ perror("readdir_hold: finder");
+ return NULL;
+}
+
+/* the creator is back, or waits in the kernel for the lock */
+static inline bool readdir_hold_creator_settled(struct readdir_hold *h)
+{
+ char path[32], buf[256], *p;
+ ssize_t n;
+ int fd;
+
+ if (__atomic_load_n(&h->created, __ATOMIC_ACQUIRE))
+ return true;
+ if (!h->creator_tid)
+ return false;
+ snprintf(path, sizeof(path), "%d/stat", h->creator_tid);
+ fd = openat(h->taskdir, path, O_RDONLY | O_CLOEXEC);
+ if (fd < 0)
+ return false;
+ n = read(fd, buf, sizeof(buf) - 1);
+ close(fd);
+ if (n <= 0)
+ return false;
+ buf[n] = 0;
+ /* "pid (comm) state ..." */
+ p = strrchr(buf, ')');
+ return p && p[1] == ' ' && p[2] == 'D';
+}
+
+static inline bool readdir_hold_faulted(struct readdir_hold *h)
+{
+ struct pollfd pfd = { .fd = h->uffd, .events = POLLIN };
+ struct uffd_msg msg;
+
+ if (poll(&pfd, 1, HOLD_FAULT_MS) != 1)
+ return false;
+ if (read(h->uffd, &msg, sizeof(msg)) != sizeof(msg))
+ return false;
+ return msg.event == UFFD_EVENT_PAGEFAULT;
+}
+
+/* let the reader go on */
+static inline void readdir_hold_release(struct readdir_hold *h)
+{
+ struct uffdio_copy cp = {
+ .dst = (unsigned long)h->page,
+ .len = h->page_size,
+ };
+ void *zero;
+
+ zero = calloc(1, h->page_size);
+ if (!zero)
+ return;
+ cp.src = (unsigned long)zero;
+ if (ioctl(h->uffd, UFFDIO_COPY, &cp) && errno != EEXIST)
+ perror("readdir_hold: UFFDIO_COPY");
+ free(zero);
+}
+
+/*
+ * A readdir of @dfd that sticks in the fault of its buffer, then a create
+ * and a lookup in @dfd. Whether the lookup came back while the readdir was
+ * still stuck goes to @stalled. Returns -1 when that could not be found
+ * out.
+ */
+static inline int readdir_hold_check(struct readdir_hold *h, int dfd,
+ bool *stalled)
+{
+ pthread_t reader, creator, finder;
+ struct uffdio_register reg = {};
+ struct pollfd pfd;
+ int ret = -1, i;
+
+ h->dfd = dfd;
+ h->created = 0;
+ h->creator_tid = 0;
+ h->page = mmap(NULL, h->page_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (h->page == MAP_FAILED)
+ return -1;
+ reg.range.start = (unsigned long)h->page;
+ reg.range.len = h->page_size;
+ reg.mode = UFFDIO_REGISTER_MODE_MISSING;
+ if (ioctl(h->uffd, UFFDIO_REGISTER, &reg) || pipe2(h->done, O_CLOEXEC))
+ goto out_page;
+
+ if (pthread_create(&reader, NULL, readdir_hold_reader, h))
+ goto out_pipe;
+ if (!readdir_hold_faulted(h))
+ goto out_reader;
+ if (pthread_create(&creator, NULL, readdir_hold_creator, h))
+ goto out_reader;
+ for (i = 0; i < HOLD_QUEUE_MS / 10 && !readdir_hold_creator_settled(h); i++)
+ usleep(10000);
+ if (pthread_create(&finder, NULL, readdir_hold_finder, h))
+ goto out_creator;
+
+ pfd.fd = h->done[0];
+ pfd.events = POLLIN;
+ *stalled = poll(&pfd, 1, HOLD_LOOKUP_MS) != 1;
+ ret = 0;
+
+ readdir_hold_release(h);
+ pthread_join(finder, NULL);
+out_creator:
+ if (ret)
+ readdir_hold_release(h);
+ pthread_join(creator, NULL);
+out_reader:
+ if (ret)
+ readdir_hold_release(h);
+ pthread_join(reader, NULL);
+out_pipe:
+ close(h->done[0]);
+ close(h->done[1]);
+out_page:
+ munmap(h->page, h->page_size);
+ h->page = MAP_FAILED;
+ return ret;
+}
+
+#endif /* __SELFTESTS_READDIR_HOLD_H */
diff --git a/tools/testing/selftests/filesystems/rw_hint_test.c b/tools/testing/selftests/filesystems/rw_hint_test.c
new file mode 100644
index 000000000000..d1930f82f63b
--- /dev/null
+++ b/tools/testing/selftests/filesystems/rw_hint_test.c
@@ -0,0 +1,129 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * F_SET_RW_HINT is refused on an immutable inode. Nothing is ever written
+ * to it and it may be shared with everybody, like a namespace file or the
+ * root of an empty mount namespace.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/ioctl.h>
+#include <sys/stat.h>
+#include <sys/wait.h>
+
+#include "../kselftest_harness.h"
+
+/* <linux/fs.h> and <linux/fcntl.h> don't mix with the libc headers */
+#ifndef FS_IOC_GETFLAGS
+#define FS_IOC_GETFLAGS _IOR('f', 1, long)
+#define FS_IOC_SETFLAGS _IOW('f', 2, long)
+#endif
+#ifndef FS_IMMUTABLE_FL
+#define FS_IMMUTABLE_FL 0x00000010
+#endif
+#ifndef F_LINUX_SPECIFIC_BASE
+#define F_LINUX_SPECIFIC_BASE 1024
+#endif
+#ifndef F_GET_RW_HINT
+#define F_GET_RW_HINT (F_LINUX_SPECIFIC_BASE + 11)
+#define F_SET_RW_HINT (F_LINUX_SPECIFIC_BASE + 12)
+#endif
+#ifndef RWH_WRITE_LIFE_SHORT
+#define RWH_WRITE_LIFE_SHORT 2
+#endif
+#ifndef UNSHARE_EMPTY_MNTNS
+#define UNSHARE_EMPTY_MNTNS 0x00100000
+#endif
+
+static int set_hint(int fd, uint64_t hint)
+{
+ return fcntl(fd, F_SET_RW_HINT, &hint);
+}
+
+static long get_hint(int fd)
+{
+ uint64_t hint;
+
+ if (fcntl(fd, F_GET_RW_HINT, &hint))
+ return -1;
+ return hint;
+}
+
+TEST(immutable_file)
+{
+ char path[] = "/tmp/rw_hint.XXXXXX";
+ int fd, flags;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+
+ fd = mkstemp(path);
+ ASSERT_GE(fd, 0);
+ unlink(path);
+ ASSERT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), 0);
+ EXPECT_EQ(get_hint(fd), RWH_WRITE_LIFE_SHORT);
+
+ if (ioctl(fd, FS_IOC_GETFLAGS, &flags)) {
+ close(fd);
+ SKIP(return, "no file attributes on this filesystem");
+ }
+ flags |= FS_IMMUTABLE_FL;
+ ASSERT_EQ(ioctl(fd, FS_IOC_SETFLAGS, &flags), 0);
+ EXPECT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), -1);
+ EXPECT_EQ(errno, EPERM);
+ flags &= ~FS_IMMUTABLE_FL;
+ ASSERT_EQ(ioctl(fd, FS_IOC_SETFLAGS, &flags), 0);
+ EXPECT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), 0);
+ close(fd);
+}
+
+TEST(namespace_file)
+{
+ int fd;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+
+ fd = open("/proc/self/ns/mnt", O_RDONLY | O_CLOEXEC);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), -1);
+ EXPECT_EQ(errno, EPERM);
+ close(fd);
+}
+
+TEST(empty_mntns_root)
+{
+ int status;
+ pid_t pid;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ int fd;
+
+ if (unshare(UNSHARE_EMPTY_MNTNS))
+ _exit(errno == EINVAL ? 100 : 1);
+ fd = open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ if (fd < 0)
+ _exit(2);
+ if (set_hint(fd, RWH_WRITE_LIFE_SHORT) == 0)
+ _exit(3);
+ _exit(errno == EPERM ? 0 : 4);
+ }
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ ASSERT_TRUE(WIFEXITED(status));
+ if (WEXITSTATUS(status) == 100)
+ SKIP(return, "UNSHARE_EMPTY_MNTNS not supported");
+ EXPECT_EQ(WEXITSTATUS(status), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/umount_propagation/Makefile b/tools/testing/selftests/filesystems/umount_propagation/Makefile
index fc0a0783018b..31f783a92b23 100644
--- a/tools/testing/selftests/filesystems/umount_propagation/Makefile
+++ b/tools/testing/selftests/filesystems/umount_propagation/Makefile
@@ -1,5 +1,5 @@
# SPDX-License-Identifier: GPL-2.0
-TEST_GEN_PROGS := umount_propagation_test
+TEST_GEN_PROGS := umount_propagation_test shrink_submounts_test locked_mount_test
CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
diff --git a/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c b/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c
new file mode 100644
index 000000000000..0799b2ba545d
--- /dev/null
+++ b/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c
@@ -0,0 +1,432 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * MNT_LOCKED keeps the owner of a user namespace from revealing what a
+ * mount covers. The lock has to be set on the copy in that namespace and
+ * only there, it has to survive the expiry of a mount placed beneath the
+ * locked one, and it has to survive the propagated umount of such a copy.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef MOVE_MOUNT_F_EMPTY_PATH
+#define MOVE_MOUNT_F_EMPTY_PATH 0x00000004
+#endif
+#ifndef MOVE_MOUNT_BENEATH
+#define MOVE_MOUNT_BENEATH 0x00000200
+#endif
+
+#define DIR_LEN 64
+#define PATH_LEN 192
+#define FLAGS (MS_NOSUID | MS_NODEV | MS_NOEXEC)
+
+/* exit codes of the children, each test says what they mean */
+enum {
+ CHILD_OK,
+ CHILD_SETUP,
+ CHILD_STEP1,
+ CHILD_STEP2,
+ CHILD_STEP3,
+ CHILD_STEP4,
+};
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int create_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Root in a new user namespace with a copy of the mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ prctl(PR_SET_DUMPABLE, 1);
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ return setgid(0) || setuid(0) ? -1 : 0;
+}
+
+static int wait_child(pid_t pid)
+{
+ int status;
+
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+static int wait_byte(int fd)
+{
+ char c;
+
+ return read(fd, &c, 1) == 1 ? 0 : -1;
+}
+
+static int send_byte(int fd)
+{
+ return write(fd, "x", 1) == 1 ? 0 : -1;
+}
+
+/* A read of @path fails: the file is covered. */
+static bool covered(const char *path)
+{
+ int fd = open(path, O_RDONLY | O_CLOEXEC);
+
+ if (fd >= 0)
+ close(fd);
+ return fd < 0;
+}
+
+FIXTURE(locked_mount) {
+ char base[DIR_LEN];
+ bool tracing; /* debugfs with the tracefs automount is there */
+};
+
+FIXTURE_SETUP(locked_mount)
+{
+ char p[PATH_LEN];
+ struct stat st;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+
+ snprintf(self->base, sizeof(self->base), "/tmp/locked_mount.XXXXXX");
+ ASSERT_NE(mkdtemp(self->base), NULL);
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, "mode=0755"), 0);
+
+ /* an automount that a user can name: the tracefs below debugfs */
+ snprintf(p, sizeof(p), "%s/dbg", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ self->tracing = !mount("debugfs", p, "debugfs", FLAGS, NULL);
+ if (self->tracing) {
+ /* the cases trigger it themselves */
+ snprintf(p, sizeof(p), "%s/dbg/tracing", self->base);
+ self->tracing = !fstatat(AT_FDCWD, p, &st, AT_NO_AUTOMOUNT) &&
+ S_ISDIR(st.st_mode);
+ }
+}
+
+FIXTURE_TEARDOWN(locked_mount)
+{
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+}
+
+/*
+ * The child keeps a directory descriptor on the host's debugfs mount, moves
+ * to a user namespace of its own and triggers the automount through the
+ * descriptor. The mount goes below the host's mount and propagates into the
+ * child's copy. The child's copy has to be locked, the host's not.
+ */
+static int automount_child(const char *base, int dfd, int to_host, int from_host)
+{
+ char p[PATH_LEN];
+ struct stat st;
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ if (fstatat(dfd, "tracing/.", &st, 0))
+ return CHILD_STEP1;
+ if (send_byte(to_host) || wait_byte(from_host))
+ return CHILD_SETUP;
+ /* the flags of the copy are locked: EPERM */
+ snprintf(p, sizeof(p), "%s/dbg/tracing", base);
+ if (!mount(NULL, p, NULL, MS_REMOUNT | MS_BIND, NULL) || errno != EPERM)
+ return CHILD_STEP2;
+ return CHILD_OK;
+}
+
+TEST_F(locked_mount, automount_locked_in_the_triggering_namespace)
+{
+ int to_host[2], from_host[2], dfd, ret;
+ char p[PATH_LEN];
+ pid_t pid;
+
+ if (!self->tracing)
+ SKIP(return, "test requires debugfs with the tracefs automount");
+
+ snprintf(p, sizeof(p), "%s/dbg", self->base);
+ ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0);
+ dfd = open(p, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(dfd, 0);
+ ASSERT_EQ(pipe(to_host), 0);
+ ASSERT_EQ(pipe(from_host), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(automount_child(self->base, dfd, to_host[1], from_host[0]));
+ close(dfd);
+ ASSERT_EQ(wait_byte(to_host[0]), 0);
+
+ /* the host's own mount isn't locked: the flags can go */
+ snprintf(p, sizeof(p), "%s/dbg/tracing", self->base);
+ EXPECT_EQ(mount(NULL, p, NULL, MS_REMOUNT | MS_BIND, NULL), 0);
+
+ ASSERT_EQ(send_byte(from_host[1]), 0);
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+/*
+ * A copy of a tree with a locked cover. The child puts a shrinkable mount
+ * beneath the cover, which hands the lock down, unmounts the cover and then
+ * asks for the umount of an unlocked ancestor. That expires the shrinkable
+ * mount on a kernel that doesn't look at the lock, and the covered
+ * directory is bare.
+ */
+static int expiry_child(const char *base)
+{
+ char srv[PATH_LEN], shr[PATH_LEN], x[PATH_LEN], c[PATH_LEN];
+ char hidden[PATH_LEN], secret[PATH_LEN];
+
+ snprintf(srv, sizeof(srv), "%s/srv", base);
+ snprintf(shr, sizeof(shr), "%s/shr", base);
+ snprintf(x, sizeof(x), "%s/x", base);
+ snprintf(c, sizeof(c), "%s/c", base);
+ snprintf(hidden, sizeof(hidden), "%s/x/hidden", base);
+ snprintf(secret, sizeof(secret), "%s/x/hidden/secret", base);
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ if (mount(srv, x, NULL, MS_BIND | MS_REC, NULL) ||
+ mount(shr, c, NULL, MS_BIND, NULL))
+ return CHILD_SETUP;
+ /* the cover is locked */
+ if (!umount2(hidden, 0) || errno != EINVAL)
+ return CHILD_STEP1;
+ if (syscall(__NR_move_mount, AT_FDCWD, c, AT_FDCWD, hidden, MOVE_MOUNT_BENEATH))
+ return CHILD_STEP2;
+ /* the lock moved down, the cover may go */
+ if (umount2(hidden, 0))
+ return CHILD_STEP3;
+ /* the holder of the lock may not, in any way */
+ if (!umount2(hidden, 0) || errno != EINVAL)
+ return CHILD_STEP4;
+ if (chdir(x))
+ return CHILD_SETUP;
+ umount2(x, 0);
+ return covered(secret) ? CHILD_OK : CHILD_STEP4;
+}
+
+TEST_F(locked_mount, expiry_leaves_a_locked_mount_alone)
+{
+ char p[PATH_LEN], q[PATH_LEN];
+ pid_t pid;
+ int ret;
+
+ if (!self->tracing)
+ SKIP(return, "test requires debugfs with the tracefs automount");
+
+ snprintf(p, sizeof(p), "%s/dbg/tracing", self->base);
+ snprintf(q, sizeof(q), "%s/shr", self->base);
+ ASSERT_EQ(mkdir(q, 0755), 0);
+ /* a bind of an automount is shrinkable as well */
+ ASSERT_EQ(mount(p, q, NULL, MS_BIND, NULL), 0);
+ snprintf(p, sizeof(p), "%s/srv", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ snprintf(p, sizeof(p), "%s/srv/hidden", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/srv/hidden/secret", self->base);
+ ASSERT_EQ(create_file(p, "covered-by-root\n"), 0);
+ snprintf(p, sizeof(p), "%s/srv/hidden", self->base);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ snprintf(p, sizeof(p), "%s/x", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/c", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(expiry_child(self->base));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+/*
+ * The copy of the mount tree that a user namespace gets at its creation has
+ * the automounts in it locked unless they are on an expiry list. A bind of
+ * such a tree inside the namespace copies the lock to the automount but not
+ * to the root of the bind, so the child may ask for the umount of that root.
+ * That must not expire the locked automount below it: the root is busy, the
+ * umount fails and the automount has to be there afterwards.
+ */
+static int copied_tree_child(const char *base)
+{
+ char dbg[PATH_LEN], x[PATH_LEN], tracing[PATH_LEN];
+ struct stat root, st;
+
+ snprintf(dbg, sizeof(dbg), "%s/dbg", base);
+ snprintf(x, sizeof(x), "%s/x", base);
+ snprintf(tracing, sizeof(tracing), "%s/x/tracing", base);
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ if (mount(dbg, x, NULL, MS_BIND | MS_REC, NULL))
+ return CHILD_SETUP;
+ /* the copy of the automount carries the lock */
+ if (!umount2(tracing, 0) || errno != EINVAL)
+ return CHILD_STEP1;
+ /* the root of the bind doesn't; keep it busy */
+ if (chdir(x))
+ return CHILD_SETUP;
+ if (!umount2(x, 0) || errno != EBUSY)
+ return CHILD_STEP2;
+ /* the automount below it must not have gone */
+ if (stat(x, &root) || fstatat(AT_FDCWD, tracing, &st, AT_NO_AUTOMOUNT))
+ return CHILD_SETUP;
+ return st.st_dev != root.st_dev ? CHILD_OK : CHILD_STEP3;
+}
+
+TEST_F(locked_mount, umount_of_a_bind_leaves_a_locked_automount_alone)
+{
+ char p[PATH_LEN];
+ struct stat st;
+ pid_t pid;
+ int ret;
+
+ if (!self->tracing)
+ SKIP(return, "test requires debugfs with the tracefs automount");
+
+ /* the copy the child gets has to contain the automount */
+ snprintf(p, sizeof(p), "%s/dbg/tracing/.", self->base);
+ ASSERT_EQ(stat(p, &st), 0);
+ snprintf(p, sizeof(p), "%s/x", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(copied_tree_child(self->base));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+/*
+ * Z is a user namespace with a copy of a tree in which P/d is covered by a
+ * locked mount. The host mounts X on P/d, which propagates beneath Z's
+ * cover, and unmounts it again. Z's cover has to be locked afterwards as
+ * it was before.
+ */
+static int propagation_child(const char *base, int to_host, int from_host)
+{
+ char d[PATH_LEN], secret[PATH_LEN];
+
+ snprintf(d, sizeof(d), "%s/P/d", base);
+ snprintf(secret, sizeof(secret), "%s/P/d/secret", base);
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ /* the cover is locked */
+ if (!umount2(d, 0) || errno != EINVAL)
+ return CHILD_STEP1;
+ if (send_byte(to_host) || wait_byte(from_host))
+ return CHILD_SETUP;
+ /* X came and went beneath it: still locked */
+ if (!umount2(d, 0) || errno != EINVAL)
+ return CHILD_STEP2;
+ return covered(secret) ? CHILD_OK : CHILD_STEP3;
+}
+
+TEST_F(locked_mount, propagated_copy_keeps_the_cover_locked)
+{
+ int to_host[2], from_host[2], ret;
+ char p[PATH_LEN];
+ pid_t pid;
+
+ snprintf(p, sizeof(p), "%s/P", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0);
+ snprintf(p, sizeof(p), "%s/P/d", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/P/d/secret", self->base);
+ ASSERT_EQ(create_file(p, "covered-by-root\n"), 0);
+ ASSERT_EQ(pipe(to_host), 0);
+ ASSERT_EQ(pipe(from_host), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ /* M: a manager's namespace that covers P/d for Z */
+ pid_t z;
+
+ if (unshare(CLONE_NEWNS))
+ _exit(CHILD_SETUP);
+ snprintf(p, sizeof(p), "%s/P", self->base);
+ if (mount(NULL, p, NULL, MS_SLAVE, NULL))
+ _exit(CHILD_SETUP);
+ snprintf(p, sizeof(p), "%s/P/d", self->base);
+ if (mount("tmpfs", p, "tmpfs", 0, "mode=0755"))
+ _exit(CHILD_SETUP);
+ z = fork();
+ if (z < 0)
+ _exit(CHILD_SETUP);
+ if (z == 0)
+ _exit(propagation_child(self->base, to_host[1], from_host[0]));
+ _exit(wait_child(z));
+ }
+ ASSERT_EQ(wait_byte(to_host[0]), 0);
+
+ /* the host mounts on P/d and unmounts again; both propagate */
+ snprintf(p, sizeof(p), "%s/P/d", self->base);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ ASSERT_EQ(umount2(p, 0), 0);
+
+ ASSERT_EQ(send_byte(from_host[1]), 0);
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c b/tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c
new file mode 100644
index 000000000000..b024ae3417ce
--- /dev/null
+++ b/tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c
@@ -0,0 +1,211 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A synchronous umount first unmounts the shrinkable submounts of the
+ * victim that aren't busy. Every one of them has to be checked right
+ * before it is unmounted: unmounting one can slide a busy mount to where
+ * the propagated copy of the next one is looked up.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/fanotify.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/statfs.h>
+#include <sys/vfs.h>
+#include <linux/magic.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef FAN_REPORT_MNT
+#define FAN_REPORT_MNT 0x00004000
+#endif
+#ifndef FAN_MARK_MNTNS
+#define FAN_MARK_MNTNS 0x00000110
+#endif
+#ifndef FAN_MNT_ATTACH
+#define FAN_MNT_ATTACH 0x01000000
+#endif
+#ifndef FAN_MNT_DETACH
+#define FAN_MNT_DETACH 0x02000000
+#endif
+
+#define DIR_LEN 64
+#define PATH_LEN 128
+
+FIXTURE(shrink_submounts) {
+ char base[DIR_LEN];
+ char automount[PATH_LEN];
+ bool mounted;
+ int fan;
+};
+
+/*
+ * Shrinkable mounts come from an automount. The tracefs mount below debugfs
+ * is one and bind mounts inherit the flag.
+ */
+FIXTURE_SETUP(shrink_submounts)
+{
+ struct stat st;
+ char p[PATH_LEN];
+
+ self->mounted = false;
+ self->fan = -1;
+
+ if (geteuid() != 0)
+ SKIP(return, "test requires CAP_SYS_ADMIN");
+
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+ snprintf(self->base, sizeof(self->base), "/tmp/shrink_submounts.XXXXXX");
+ ASSERT_NE(mkdtemp(self->base), NULL);
+ ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0);
+ self->mounted = true;
+ ASSERT_EQ(mount(NULL, self->base, NULL, MS_PRIVATE, NULL), 0);
+
+ snprintf(p, sizeof(p), "%s/dbg", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ if (mount("debugfs", p, "debugfs", 0, NULL)) {
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+ SKIP(return, "test requires debugfs");
+ }
+ snprintf(self->automount, sizeof(self->automount), "%s/dbg/tracing",
+ self->base);
+ snprintf(p, sizeof(p), "%s/dbg/tracing/.", self->base);
+ if (stat(p, &st)) {
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+ SKIP(return, "test requires the tracefs automount");
+ }
+}
+
+FIXTURE_TEARDOWN(shrink_submounts)
+{
+ if (self->fan >= 0)
+ close(self->fan);
+ chdir("/");
+ if (self->mounted)
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+}
+
+static bool mounted_tmpfs(const char *path)
+{
+ struct statfs st;
+
+ return !statfs(path, &st) && st.f_type == TMPFS_MAGIC;
+}
+
+/*
+ * P is a shared bind mount of the automount, P1 a slave of P that is shared
+ * in turn and P2 its peer. B, another bind mount of the automount, goes on
+ * P/options and propagates copies Bc1 and Bc2 onto P1/options and
+ * P2/options. R, a tmpfs and our working directory, sits on top of Bc2 which
+ * is made private first. P, with B on it, is moved to P1/instances.
+ *
+ * A synchronous umount of P1 unmounts the shrinkable submounts Bc1 and B
+ * first. Unmounting Bc1 takes Bc2 along and slides R to P2/options where
+ * the propagated copy of B is looked up next. R is busy, so B has to stay
+ * and the umount fails with EBUSY.
+ */
+TEST_F(shrink_submounts, busy_mount_moved_into_reach)
+{
+ char p[PATH_LEN], p1[PATH_LEN], p2[PATH_LEN], r[PATH_LEN], cwd[PATH_LEN];
+ int nsfd;
+
+ snprintf(p, sizeof(p), "%s/p", self->base);
+ snprintf(p1, sizeof(p1), "%s/p1", self->base);
+ snprintf(p2, sizeof(p2), "%s/p2", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(mkdir(p1, 0755), 0);
+ ASSERT_EQ(mkdir(p2, 0755), 0);
+
+ /* watch the mount namespace so that the detached mounts get queued */
+ self->fan = fanotify_init(FAN_REPORT_MNT, O_RDONLY);
+ if (self->fan >= 0) {
+ nsfd = open("/proc/self/ns/mnt", O_RDONLY | O_CLOEXEC);
+ ASSERT_GE(nsfd, 0);
+ EXPECT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD | FAN_MARK_MNTNS,
+ FAN_MNT_ATTACH | FAN_MNT_DETACH, nsfd, NULL), 0);
+ close(nsfd);
+ }
+
+ ASSERT_EQ(mount(self->automount, p, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0);
+ ASSERT_EQ(mount(p, p1, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(mount(NULL, p1, NULL, MS_SLAVE, NULL), 0);
+ ASSERT_EQ(mount(NULL, p1, NULL, MS_SHARED, NULL), 0);
+ ASSERT_EQ(mount(p1, p2, NULL, MS_BIND, NULL), 0);
+
+ /* B on P/options, copies on P1/options and P2/options */
+ snprintf(r, sizeof(r), "%s/p/options", self->base);
+ ASSERT_EQ(mount(self->automount, r, NULL, MS_BIND, NULL), 0);
+
+ /* R on top of Bc2 */
+ snprintf(r, sizeof(r), "%s/p2/options", self->base);
+ ASSERT_EQ(mount(NULL, r, NULL, MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(mount("R", r, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(chdir(r), 0);
+
+ /* P, with B on it, below the victim */
+ snprintf(cwd, sizeof(cwd), "%s/p1/instances", self->base);
+ ASSERT_EQ(mount(p, cwd, NULL, MS_MOVE, NULL), 0);
+
+ ASSERT_TRUE(mounted_tmpfs(r));
+ ASSERT_EQ(umount2(p1, 0), -1);
+ EXPECT_EQ(errno, EBUSY);
+
+ /* R is still mounted and still our working directory */
+ EXPECT_TRUE(mounted_tmpfs(r));
+ ASSERT_NE(getcwd(cwd, sizeof(cwd)), NULL);
+ EXPECT_STREQ(cwd, r);
+}
+
+/*
+ * T is shared and Q, a slave of T, has T moved into it, so Q receives
+ * propagation from its own child. M, another bind mount of the automount, is
+ * on T/options and that is the dentry T sits on in Q. Unmounting M makes T
+ * the propagated victim at that dentry in Q and takes T along. The shrink
+ * walk of V has just unmounted M and continues in the children of T.
+ */
+TEST_F(shrink_submounts, parent_goes_with_child)
+{
+ char v[PATH_LEN], t[PATH_LEN], q[PATH_LEN], p[PATH_LEN];
+ struct stat before, after;
+
+ snprintf(v, sizeof(v), "%s/v", self->base);
+ snprintf(t, sizeof(t), "%s/v/t", self->base);
+ snprintf(q, sizeof(q), "%s/v/q", self->base);
+ ASSERT_EQ(mkdir(v, 0755), 0);
+ ASSERT_EQ(mount("V", v, "tmpfs", 0, NULL), 0);
+ ASSERT_EQ(mount(NULL, v, NULL, MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(stat(v, &before), 0);
+ ASSERT_EQ(mkdir(t, 0755), 0);
+ ASSERT_EQ(mkdir(q, 0755), 0);
+
+ /* T shared, M on T/options before anything receives from T */
+ ASSERT_EQ(mount(self->automount, t, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(mount(NULL, t, NULL, MS_SHARED, NULL), 0);
+ snprintf(p, sizeof(p), "%s/v/t/options", self->base);
+ ASSERT_EQ(mount(self->automount, p, NULL, MS_BIND, NULL), 0);
+
+ /* Q, a slave of T, and T moved into Q at the dentry M sits on */
+ ASSERT_EQ(mount(t, q, NULL, MS_BIND, NULL), 0);
+ ASSERT_EQ(mount(NULL, q, NULL, MS_SLAVE, NULL), 0);
+ snprintf(p, sizeof(p), "%s/v/q/options", self->base);
+ ASSERT_EQ(mount(t, p, NULL, MS_MOVE, NULL), 0);
+
+ /* M, T and then Q go, V is empty and can be unmounted */
+ ASSERT_EQ(umount2(v, 0), 0);
+ ASSERT_EQ(stat(v, &after), 0);
+ EXPECT_NE(before.st_dev, after.st_dev);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/kselftest/runner.sh b/tools/testing/selftests/kselftest/runner.sh
index 311811dc55a0..ee2c1f0403d9 100644
--- a/tools/testing/selftests/kselftest/runner.sh
+++ b/tools/testing/selftests/kselftest/runner.sh
@@ -38,8 +38,12 @@ tap_prefix()
tap_timeout()
{
+ # nommu doesn't support timeout command (missing fork(2))
+ if [ "$NOMMU" = "1" ] ; then
+ echo "timeout isn't supported for NOMMU"
+ $1
# Make sure tests will time out if utility is available.
- if [ -x /usr/bin/timeout ] ; then
+ elif [ -x /usr/bin/timeout ] ; then
/usr/bin/timeout --foreground "$kselftest_timeout" \
/usr/bin/timeout "$kselftest_timeout" $1
else
diff --git a/tools/testing/selftests/kvm/arm64/vgic_init.c b/tools/testing/selftests/kvm/arm64/vgic_init.c
index 47e34b43afb2..5a30f3cb039b 100644
--- a/tools/testing/selftests/kvm/arm64/vgic_init.c
+++ b/tools/testing/selftests/kvm/arm64/vgic_init.c
@@ -5,6 +5,7 @@
* Copyright (C) 2020, Red Hat, Inc.
*/
#include <linux/kernel.h>
+#include <linux/sizes.h>
#include <sys/syscall.h>
#include <asm/kvm.h>
#include <asm/kvm_para.h>
@@ -13,12 +14,21 @@
#include "test_util.h"
#include "kvm_util.h"
+#include "gic.h"
#include "processor.h"
#include "vgic.h"
#include "gic_v3.h"
#define NR_VCPUS 4
+#define REDIST_RETRY_REGION0_BASE GICR_BASE_GPA
+#define REDIST_RETRY_REGION1_BASE \
+ (REDIST_RETRY_REGION0_BASE + 2 * KVM_VGIC_V3_REDIST_SIZE)
+#define REDIST_RETRY_DIST_BASE \
+ (REDIST_RETRY_REGION1_BASE + KVM_VGIC_V3_REDIST_SIZE)
+#define REDIST_RETRY_REGION2_BASE \
+ (REDIST_RETRY_DIST_BASE + KVM_VGIC_V3_DIST_SIZE)
+
#define REG_OFFSET(vcpu, offset) (((u64)vcpu << 32) | offset)
#define VGIC_DEV_IS_V2(_d) ((_d) == KVM_DEV_TYPE_ARM_VGIC_V2)
@@ -65,6 +75,23 @@ static void guest_code(void)
GUEST_DONE();
}
+static void guest_check_redist_retry(void)
+{
+ unsigned int i;
+
+ /* The first three redistributors span adjacent regions 0 and 1. */
+ for (i = 0; i < NR_VCPUS; i++) {
+ u64 base = i < 3 ? REDIST_RETRY_REGION0_BASE +
+ i * KVM_VGIC_V3_REDIST_SIZE :
+ REDIST_RETRY_REGION2_BASE;
+ u64 typer = readq((void *)(unsigned long)(base + GICR_TYPER));
+
+ GUEST_ASSERT_EQ(GICR_TYPER_CPU_NUMBER(typer), i);
+ }
+
+ GUEST_DONE();
+}
+
/* we don't want to assert on run execution, hence that helper */
static int run_vcpu(struct kvm_vcpu *vcpu)
{
@@ -73,6 +100,7 @@ static int run_vcpu(struct kvm_vcpu *vcpu)
static struct vm_gic vm_gic_create_with_vcpus(u32 gic_dev_type,
u32 nr_vcpus,
+ void *guest_code,
struct kvm_vcpu *vcpus[])
{
struct vm_gic v;
@@ -338,7 +366,7 @@ static void test_vgic_then_vcpus(u32 gic_dev_type)
struct vm_gic v;
int ret, i;
- v = vm_gic_create_with_vcpus(gic_dev_type, 1, vcpus);
+ v = vm_gic_create_with_vcpus(gic_dev_type, 1, guest_code, vcpus);
subtest_dist_rdist(&v);
@@ -359,7 +387,8 @@ static void test_vcpus_then_vgic(u32 gic_dev_type)
struct vm_gic v;
int ret;
- v = vm_gic_create_with_vcpus(gic_dev_type, NR_VCPUS, vcpus);
+ v = vm_gic_create_with_vcpus(gic_dev_type, NR_VCPUS, guest_code,
+ vcpus);
subtest_dist_rdist(&v);
@@ -411,7 +440,8 @@ static void test_v3_new_redist_regions(void)
u64 addr;
int ret;
- v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus);
+ v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS,
+ guest_code, vcpus);
subtest_v3_redist_regions(&v);
kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_CTRL,
KVM_DEV_ARM_VGIC_CTRL_INIT, NULL);
@@ -422,7 +452,8 @@ static void test_v3_new_redist_regions(void)
/* step2 */
- v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus);
+ v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS,
+ guest_code, vcpus);
subtest_v3_redist_regions(&v);
addr = REDIST_REGION_ATTR_ADDR(1, 0x280000, 0, 2);
@@ -436,7 +467,8 @@ static void test_v3_new_redist_regions(void)
/* step 3 */
- v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus);
+ v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS,
+ guest_code, vcpus);
subtest_v3_redist_regions(&v);
ret = __kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR,
@@ -457,6 +489,70 @@ static void test_v3_new_redist_regions(void)
vm_gic_destroy(&v);
}
+static void test_v3_redist_region_retry(void)
+{
+ struct kvm_vcpu *vcpus[NR_VCPUS];
+ struct vm_gic v;
+ struct ucall uc;
+ u64 addr;
+ int ret;
+
+ v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS,
+ guest_check_redist_retry, vcpus);
+
+ addr = REDIST_REGION_ATTR_ADDR(2, REDIST_RETRY_REGION0_BASE, 0, 0);
+ kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR,
+ KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, &addr);
+
+ addr = REDIST_REGION_ATTR_ADDR(1, REDIST_RETRY_REGION1_BASE, 0, 1);
+ kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR,
+ KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, &addr);
+
+ addr = REDIST_RETRY_DIST_BASE;
+ kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR,
+ KVM_VGIC_V3_ADDR_TYPE_DIST, &addr);
+
+ addr = REDIST_REGION_ATTR_ADDR(1, REDIST_RETRY_DIST_BASE, 0, 2);
+ ret = __kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR,
+ KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION,
+ &addr);
+ TEST_ASSERT(ret && errno == EINVAL,
+ "register redist region colliding with dist");
+
+ addr = REDIST_REGION_ATTR_ADDR(1, REDIST_RETRY_REGION2_BASE, 0, 2);
+ kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR,
+ KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, &addr);
+
+ virt_map(v.vm, REDIST_RETRY_REGION0_BASE, REDIST_RETRY_REGION0_BASE,
+ vm_calc_num_guest_pages(v.vm->mode,
+ 3 * KVM_VGIC_V3_REDIST_SIZE));
+ virt_map(v.vm, REDIST_RETRY_REGION2_BASE, REDIST_RETRY_REGION2_BASE,
+ vm_calc_num_guest_pages(v.vm->mode,
+ KVM_VGIC_V3_REDIST_SIZE));
+
+ kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_CTRL,
+ KVM_DEV_ARM_VGIC_CTRL_INIT, NULL);
+
+ vcpu_run(vcpus[0]);
+ switch (get_ucall(vcpus[0], &uc)) {
+ case UCALL_DONE:
+ break;
+ case UCALL_ABORT:
+ REPORT_GUEST_ASSERT(uc);
+ break;
+ case UCALL_NONE:
+ if (vcpus[0]->run->exit_reason == KVM_EXIT_MMIO)
+ TEST_FAIL("Unexpected MMIO exit at 0x%llx",
+ vcpus[0]->run->mmio.phys_addr);
+ fallthrough;
+ default:
+ TEST_FAIL("Unexpected ucall %lu, exit_reason %u",
+ uc.cmd, vcpus[0]->run->exit_reason);
+ }
+
+ vm_gic_destroy(&v);
+}
+
static void test_v3_typer_accesses(void)
{
struct vm_gic v;
@@ -608,7 +704,8 @@ static void test_v3_redist_ipa_range_check_at_vcpu_run(void)
int ret, i;
u64 addr;
- v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, 1, vcpus);
+ v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, 1, guest_code,
+ vcpus);
/* Set space for 3 redists, we have 1 vcpu, so this succeeds. */
addr = max_phys_size - (3 * 2 * 0x10000);
@@ -641,7 +738,8 @@ static void test_v3_its_region(void)
u64 addr;
int its_fd, ret;
- v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus);
+ v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS,
+ guest_code, vcpus);
its_fd = kvm_create_device(v.vm, KVM_DEV_TYPE_ARM_VGIC_ITS);
addr = 0x401000;
@@ -684,7 +782,8 @@ static void test_v3_nassgicap(void)
u32 typer2;
int ret;
- vm = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus);
+ vm = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS,
+ guest_code, vcpus);
kvm_device_attr_get(vm.gic_fd, KVM_DEV_ARM_VGIC_GRP_DIST_REGS,
GICD_TYPER2, &typer2);
has_nassgicap = typer2 & GICD_TYPER2_nASSGIcap;
@@ -978,6 +1077,7 @@ void run_tests(u32 gic_dev_type)
if (VGIC_DEV_IS_V3(gic_dev_type)) {
test_v3_new_redist_regions();
+ test_v3_redist_region_retry();
test_v3_typer_accesses();
test_v3_last_bit_redist_regions();
test_v3_last_bit_single_rdist();
diff --git a/tools/testing/selftests/mm/uffd-unit-tests.c b/tools/testing/selftests/mm/uffd-unit-tests.c
index 580178630ede..97484c82f24e 100644
--- a/tools/testing/selftests/mm/uffd-unit-tests.c
+++ b/tools/testing/selftests/mm/uffd-unit-tests.c
@@ -2049,28 +2049,29 @@ static void uffd_move_swap_test_common(uffd_global_test_opts_t *gopts,
bool rwp)
{
unsigned long page_size = gopts->page_size;
- struct uffdio_move move = { };
- int pagemap_fd;
+ struct uffdio_move move = {
+ .dst = (unsigned long)gopts->area_dst,
+ .src = (unsigned long)gopts->area_src,
+ .len = page_size,
+ };
+ int pagemap_fd = pagemap_open();
if (rwp) {
if (uffd_register_rwp(gopts->uffd, gopts->area_src, page_size))
err("register src failure");
- } else if (uffd_register(gopts->uffd, gopts->area_src, page_size,
- false, true, false)) {
- err("register src failure");
- }
- if (uffd_register(gopts->uffd, gopts->area_dst, page_size,
- true, false, false))
- err("register dst failure");
-
- if (rwp)
rwprotect_range(gopts->uffd, (unsigned long)gopts->area_src,
page_size, true);
- else
+ } else {
+ if (uffd_register(gopts->uffd, gopts->area_src, page_size,
+ false, true, false))
+ err("register src failure");
wp_range(gopts->uffd, (unsigned long)gopts->area_src,
page_size, true);
+ }
+ if (uffd_register(gopts->uffd, gopts->area_dst, page_size,
+ true, false, false))
+ err("register dst failure");
- pagemap_fd = pagemap_open();
if (madvise(gopts->area_src, page_size, MADV_PAGEOUT))
err("MADV_PAGEOUT");
if (!pagemap_is_swapped(pagemap_fd, gopts->area_src)) {
@@ -2078,11 +2079,10 @@ static void uffd_move_swap_test_common(uffd_global_test_opts_t *gopts,
goto out;
}
- move.dst = (unsigned long)gopts->area_dst;
- move.src = (unsigned long)gopts->area_src;
- move.len = page_size;
- if (ioctl(gopts->uffd, UFFDIO_MOVE, &move))
- err("UFFDIO_MOVE");
+ if (ioctl(gopts->uffd, UFFDIO_MOVE, &move)) {
+ uffd_test_fail("UFFDIO_MOVE failed: %s", strerror(errno));
+ goto out;
+ }
if (pagemap_get_entry(pagemap_fd, gopts->area_dst) & PM_UFFD_WP)
uffd_test_fail("uffd bit moved into an area registered for missing faults only");
diff --git a/tools/testing/selftests/net/.gitignore b/tools/testing/selftests/net/.gitignore
index dacd36ed8455..21412510ac26 100644
--- a/tools/testing/selftests/net/.gitignore
+++ b/tools/testing/selftests/net/.gitignore
@@ -41,6 +41,7 @@ skf_net_off
socket
so_incoming_cpu
so_netns_cookie
+so_reserve_mem
so_rcv_listener
stress_reuseport_listen
tap
diff --git a/tools/testing/selftests/net/Makefile b/tools/testing/selftests/net/Makefile
index cab3f2c90039..54beea2e348c 100644
--- a/tools/testing/selftests/net/Makefile
+++ b/tools/testing/selftests/net/Makefile
@@ -197,6 +197,7 @@ TEST_GEN_PROGS := \
sk_connect_zero_addr \
sk_so_peek_off \
so_incoming_cpu \
+ so_reserve_mem \
tap \
tcp_port_share \
tls \
diff --git a/tools/testing/selftests/net/config b/tools/testing/selftests/net/config
index 737e7e6327b3..d355cf980597 100644
--- a/tools/testing/selftests/net/config
+++ b/tools/testing/selftests/net/config
@@ -7,6 +7,7 @@ CONFIG_BRIDGE_VLAN_FILTERING=y
CONFIG_CAN=m
CONFIG_CAN_DEV=m
CONFIG_CAN_VXCAN=m
+CONFIG_CGROUPS=y
CONFIG_CRYPTO_ARIA=y
CONFIG_CRYPTO_CHACHA20POLY1305=m
CONFIG_CRYPTO_SHA1=y
@@ -61,6 +62,7 @@ CONFIG_L2TP_V3=y
CONFIG_MACSEC=m
CONFIG_MACVLAN=y
CONFIG_MACVTAP=y
+CONFIG_MEMCG=y
CONFIG_MPLS=y
CONFIG_MPLS_IPTUNNEL=m
CONFIG_MPLS_ROUTING=m
diff --git a/tools/testing/selftests/net/cork_fragsize.py b/tools/testing/selftests/net/cork_fragsize.py
index 7afd643d07ec..ae6be804b7a2 100755
--- a/tools/testing/selftests/net/cork_fragsize.py
+++ b/tools/testing/selftests/net/cork_fragsize.py
@@ -54,7 +54,7 @@ def check_kernel_config(option: str) -> bool | None:
return False
except OSError:
continue
- return None
+ return None
def assert_debug_kernel() -> None:
@@ -71,7 +71,8 @@ def assert_debug_kernel() -> None:
def check_dmesg_clean(func: str) -> bool:
'''
- Check if the given function produced a WARN in dmesg.
+ Check if the given function produced a WARN in dmesg. Returns True if dmesg
+ is clean, i.e. doesn't contain traces of a WARNING in the given function.
'''
with subprocess.Popen(['dmesg'], stdout=subprocess.PIPE) as dmesg:
diff --git a/tools/testing/selftests/net/so_reserve_mem.c b/tools/testing/selftests/net/so_reserve_mem.c
new file mode 100644
index 000000000000..aca5960ddff2
--- /dev/null
+++ b/tools/testing/selftests/net/so_reserve_mem.c
@@ -0,0 +1,448 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <limits.h>
+#include <linux/sock_diag.h>
+#include <net/if.h>
+#include <netinet/in.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/ioctl.h>
+#include <sys/mount.h>
+#include <sys/socket.h>
+#include <sys/stat.h>
+#include <unistd.h>
+
+#include "kselftest_harness.h"
+
+#ifndef IPPROTO_MPTCP
+#define IPPROTO_MPTCP 262
+#endif
+
+#define SO_RESERVE_MEM_MAX (1 << 30)
+
+/* cgroup2 is mounted here, on a tmpfs, in a private mount namespace. */
+#define CG_TMP "/tmp"
+#define CG_MNT CG_TMP "/cgroup2"
+
+static int cg_write(const char *dir, const char *file, const char *buf,
+ int flags)
+{
+ ssize_t len = strlen(buf);
+ char path[PATH_MAX];
+ int fd, ret;
+
+ snprintf(path, sizeof(path), "%s/%s", dir, file);
+ fd = open(path, O_WRONLY | flags);
+ if (fd < 0)
+ return -1;
+ ret = write(fd, buf, len) == len ? 0 : -1;
+ close(fd);
+ return ret;
+}
+
+static int cg_read(const char *dir, const char *file, char *buf, size_t size)
+{
+ char path[PATH_MAX];
+ ssize_t n;
+ int fd;
+
+ snprintf(path, sizeof(path), "%s/%s", dir, file);
+ fd = open(path, O_RDONLY);
+ if (fd < 0)
+ return -1;
+ n = read(fd, buf, size - 1);
+ close(fd);
+ if (n < 0)
+ return -1;
+ buf[n] = '\0';
+ return 0;
+}
+
+static int cg_enter(const char *dir)
+{
+ char buf[32];
+
+ snprintf(buf, sizeof(buf), "%d\n", getpid());
+ return cg_write(dir, "cgroup.procs", buf, 0);
+}
+
+/* Path of the current cgroup, below CG_MNT. */
+static int cg_get_current(char *buf, size_t size)
+{
+ char line[PATH_MAX];
+ int ret = -1;
+ FILE *f;
+
+ f = fopen("/proc/self/cgroup", "r");
+ if (!f)
+ return -1;
+ while (fgets(line, sizeof(line), f)) {
+ if (strncmp(line, "0::", 3))
+ continue;
+ line[strcspn(line, "\n")] = '\0';
+ if (snprintf(buf, size, "%s%s", CG_MNT, line + 3) < (int)size)
+ ret = 0;
+ break;
+ }
+ fclose(f);
+ return ret;
+}
+
+static int get_reserve_mem(struct __test_metadata *_metadata, int fd)
+{
+ int val = -1;
+ socklen_t len = sizeof(val);
+
+ EXPECT_EQ(getsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, &len), 0);
+ return val;
+}
+
+static __u32 get_fwd_alloc(struct __test_metadata *_metadata, int fd)
+{
+ __u32 meminfo[SK_MEMINFO_VARS] = {};
+ socklen_t len = sizeof(meminfo);
+
+ EXPECT_EQ(getsockopt(fd, SOL_SOCKET, SO_MEMINFO, meminfo, &len), 0);
+ return meminfo[SK_MEMINFO_FWD_ALLOC];
+}
+
+/* Wait until a single SO_MEMINFO snapshot shows an empty write queue
+ * and at least @min_fwd_alloc bytes of forward alloc: ACK processing
+ * (possibly running on another CPU) first decrements sk_wmem_queued,
+ * then uncharges sk_forward_alloc.
+ */
+static void wait_wmem_drained(struct __test_metadata *_metadata, int fd,
+ __u32 min_fwd_alloc)
+{
+ __u32 meminfo[SK_MEMINFO_VARS] = {};
+ socklen_t len;
+ int i;
+
+ for (i = 0; i < 5000; i++) {
+ len = sizeof(meminfo);
+ ASSERT_EQ(getsockopt(fd, SOL_SOCKET, SO_MEMINFO, meminfo, &len), 0);
+ if (meminfo[SK_MEMINFO_WMEM_QUEUED] == 0 &&
+ meminfo[SK_MEMINFO_FWD_ALLOC] >= min_fwd_alloc)
+ return;
+ usleep(1000);
+ }
+ EXPECT_EQ(meminfo[SK_MEMINFO_WMEM_QUEUED], 0U);
+ EXPECT_GE(meminfo[SK_MEMINFO_FWD_ALLOC], min_fwd_alloc);
+}
+
+FIXTURE(so_reserve_mem)
+{
+ char cg_orig[PATH_MAX]; /* cgroup the test started in */
+ char cg_test[PATH_MAX]; /* cgroup the test sockets are charged to */
+ long page_size;
+ bool cg_created;
+};
+
+/* The cgroup2 mount lives in a private mount namespace which goes away
+ * with the test process, no need to unmount it.
+ */
+FIXTURE_TEARDOWN(so_reserve_mem)
+{
+ if (!self->cg_created)
+ return;
+
+ /* Leave cg_test so that it can be removed. */
+ EXPECT_EQ(cg_enter(self->cg_orig), 0);
+ EXPECT_EQ(rmdir(self->cg_test), 0);
+ self->cg_created = false;
+}
+
+FIXTURE_SETUP(so_reserve_mem)
+{
+ struct ifreq ifr = {
+ .ifr_name = "lo",
+ .ifr_flags = IFF_UP,
+ };
+ int fd, err, ret, val = 0;
+ char buf[256];
+
+ self->page_size = sysconf(_SC_PAGESIZE);
+ ASSERT_GT(self->page_size, 0);
+
+ if (unshare(CLONE_NEWNS | CLONE_NEWNET))
+ SKIP(return, "Failed to unshare namespaces (need root)");
+
+ /* Make sure the mounts below do not propagate to the host. */
+ ASSERT_EQ(mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+
+ if (mount("none", CG_TMP, "tmpfs", 0, NULL) ||
+ mkdir(CG_MNT, 0755) ||
+ mount("none", CG_MNT, "cgroup2", 0, NULL))
+ SKIP(return, "Failed to mount cgroup2");
+
+ ASSERT_EQ(cg_get_current(self->cg_orig, sizeof(self->cg_orig)), 0);
+
+ /* cg_test needs the memory controller to be enabled in the root of
+ * the hierarchy (which is not necessarily the global root cgroup
+ * when running in a cgroup namespace). Like selftests/cgroup, leave
+ * it enabled on exit: disabling it could break other users of the
+ * hierarchy.
+ */
+ ASSERT_EQ(cg_read(CG_MNT, "cgroup.subtree_control", buf, sizeof(buf)), 0);
+ if (!strstr(buf, "memory") &&
+ cg_write(CG_MNT, "cgroup.subtree_control", "+memory", 0))
+ SKIP(return, "cgroup2 memory controller not available");
+
+ /* cg_test is a leaf cgroup, it can always host the test process. */
+ snprintf(self->cg_test, sizeof(self->cg_test),
+ "%s/ksft_so_reserve_mem_%d", CG_MNT, getpid());
+ if (mkdir(self->cg_test, 0755))
+ SKIP(return, "Failed to create test cgroup");
+ self->cg_created = true;
+
+ ret = cg_enter(self->cg_test);
+ if (ret)
+ so_reserve_mem_teardown(_metadata, self, variant);
+ ASSERT_EQ(ret, 0);
+
+ /* Bring up loopback and verify memcg socket accounting is enabled */
+ fd = socket(AF_INET, SOCK_STREAM, 0);
+ ret = fd < 0 ? -1 : ioctl(fd, SIOCSIFFLAGS, &ifr);
+ if (ret) {
+ if (fd >= 0)
+ close(fd);
+ so_reserve_mem_teardown(_metadata, self, variant);
+ }
+ ASSERT_EQ(ret, 0);
+
+ ret = setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val));
+ err = errno;
+ close(fd);
+ if (ret) {
+ so_reserve_mem_teardown(_metadata, self, variant);
+ if (err == EOPNOTSUPP)
+ SKIP(return, "memcg socket accounting not enabled");
+ }
+ ASSERT_EQ(ret, 0);
+}
+
+static void check_non_tcp_rejected(struct __test_metadata *_metadata,
+ int domain, int type, int protocol,
+ int val)
+{
+ int fd = socket(domain, type, protocol);
+ int zero = 0;
+
+ if (fd < 0) {
+ /* Protocol not available, or no CAP_NET_RAW */
+ EXPECT_TRUE(errno == EAFNOSUPPORT ||
+ errno == EPROTONOSUPPORT ||
+ errno == ENOPROTOOPT ||
+ errno == EPERM ||
+ errno == EACCES);
+ TH_LOG("socket(%d, %d, %d): %s, skipped",
+ domain, type, protocol, strerror(errno));
+ return;
+ }
+ EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &zero, sizeof(zero)), -1);
+ EXPECT_EQ(errno, EOPNOTSUPP);
+ EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1);
+ EXPECT_EQ(errno, EOPNOTSUPP);
+ close(fd);
+}
+
+TEST_F(so_reserve_mem, non_tcp_rejected)
+{
+ int val = self->page_size * 4;
+
+ check_non_tcp_rejected(_metadata, AF_INET, SOCK_DGRAM, 0, val);
+ check_non_tcp_rejected(_metadata, AF_UNIX, SOCK_STREAM, 0, val);
+ check_non_tcp_rejected(_metadata, AF_INET, SOCK_RAW, IPPROTO_ICMP, val);
+ check_non_tcp_rejected(_metadata, AF_INET, SOCK_STREAM, IPPROTO_MPTCP, val);
+}
+
+TEST_F(so_reserve_mem, grow_shrink_and_rounding)
+{
+ int ps = self->page_size;
+ int fd, val;
+
+ fd = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(fd, 0);
+
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 0);
+ EXPECT_EQ(get_fwd_alloc(_metadata, fd), 0U);
+
+ /* Negative or > SO_RESERVE_MEM_MAX value -> EINVAL */
+ val = -1;
+ EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1);
+ EXPECT_EQ(errno, EINVAL);
+
+ val = SO_RESERVE_MEM_MAX + 1;
+ EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1);
+ EXPECT_EQ(errno, EINVAL);
+
+ val = INT_MAX;
+ EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1);
+ EXPECT_EQ(errno, EINVAL);
+
+ /* 1 byte rounds up to 1 page */
+ val = 1;
+ ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)ps);
+
+ /* Grow to 16 pages */
+ val = 16 * ps;
+ ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 16 * ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)(16 * ps));
+
+ /* Shrink by 1 byte (rounds delta down to 0 -> stays 16 pages) */
+ val = 16 * ps - 1;
+ ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 16 * ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)(16 * ps));
+
+ /* Shrink to 4 pages */
+ val = 4 * ps;
+ ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 4 * ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)(4 * ps));
+
+ /* Release all */
+ val = 0;
+ ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 0);
+ EXPECT_EQ(get_fwd_alloc(_metadata, fd), 0U);
+
+ close(fd);
+}
+
+TEST_F(so_reserve_mem, cgroup_memory_max)
+{
+ int ps = self->page_size;
+ char buf[32];
+ int fd, val;
+
+ /* The socket is charged to cg_test */
+ fd = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(fd, 0);
+
+ /* Move the test process back to its original cgroup before lowering
+ * cg_test's memory.max, so that the limit only governs the socket's
+ * memcg charges and cannot trigger OOM on the test process itself.
+ */
+ ASSERT_EQ(cg_enter(self->cg_orig), 0);
+
+ /* Limit cg_test memory to 8 pages and try to reserve 128 pages.
+ * O_NONBLOCK: do not try to reclaim cg_test usage above the new limit.
+ */
+ snprintf(buf, sizeof(buf), "%d\n", 8 * ps);
+ ASSERT_EQ(cg_write(self->cg_test, "memory.max", buf, O_NONBLOCK), 0);
+
+ val = 128 * ps;
+ EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1);
+ EXPECT_EQ(errno, ENOMEM);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 0);
+
+ /* Restore unlimited memory.max */
+ ASSERT_EQ(cg_write(self->cg_test, "memory.max", "max\n", 0), 0);
+
+ val = 4 * ps;
+ ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, fd), 4 * ps);
+
+ close(fd);
+}
+
+TEST_F(so_reserve_mem, accept_child_zero_reserve)
+{
+ struct sockaddr_in addr = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ };
+ socklen_t alen = sizeof(addr);
+ int ps = self->page_size;
+ int lfd, cfd, sfd, val;
+
+ lfd = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(lfd, 0);
+
+ /* Set SO_RESERVE_MEM on listener before listen() */
+ val = 4 * ps;
+ ASSERT_EQ(setsockopt(lfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ ASSERT_EQ(bind(lfd, (struct sockaddr *)&addr, sizeof(addr)), 0);
+ ASSERT_EQ(listen(lfd, 2), 0);
+ ASSERT_EQ(getsockname(lfd, (struct sockaddr *)&addr, &alen), 0);
+
+ /* Grow SO_RESERVE_MEM on listener after listen() */
+ val = 8 * ps;
+ ASSERT_EQ(setsockopt(lfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, lfd), 8 * ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, lfd), (__u32)(8 * ps));
+
+ cfd = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(cfd, 0);
+ ASSERT_EQ(connect(cfd, (struct sockaddr *)&addr, sizeof(addr)), 0);
+
+ sfd = accept(lfd, NULL, NULL);
+ ASSERT_GE(sfd, 0);
+
+ /* Child after accept() must have 0 reserve while listener keeps 8 pages */
+ EXPECT_EQ(get_reserve_mem(_metadata, sfd), 0);
+ EXPECT_EQ(get_fwd_alloc(_metadata, sfd), 0U);
+ EXPECT_EQ(get_reserve_mem(_metadata, lfd), 8 * ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, lfd), (__u32)(8 * ps));
+
+ /* Child can still independently set its own SO_RESERVE_MEM */
+ val = 6 * ps;
+ ASSERT_EQ(setsockopt(sfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ EXPECT_EQ(get_reserve_mem(_metadata, sfd), 6 * ps);
+ EXPECT_EQ(get_fwd_alloc(_metadata, sfd), (__u32)(6 * ps));
+
+ close(sfd);
+ close(cfd);
+ close(lfd);
+}
+
+TEST_F(so_reserve_mem, preserved_after_traffic)
+{
+ struct sockaddr_in addr = {
+ .sin_family = AF_INET,
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK),
+ };
+ socklen_t alen = sizeof(addr);
+ int ps = self->page_size;
+ int lfd, cfd, sfd, val;
+ char buf[8192] = {};
+
+ lfd = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(lfd, 0);
+ ASSERT_EQ(bind(lfd, (struct sockaddr *)&addr, sizeof(addr)), 0);
+ ASSERT_EQ(listen(lfd, 1), 0);
+ ASSERT_EQ(getsockname(lfd, (struct sockaddr *)&addr, &alen), 0);
+
+ cfd = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(cfd, 0);
+ val = 16 * ps;
+ ASSERT_EQ(setsockopt(cfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0);
+ ASSERT_EQ(connect(cfd, (struct sockaddr *)&addr, sizeof(addr)), 0);
+
+ sfd = accept(lfd, NULL, NULL);
+ ASSERT_GE(sfd, 0);
+
+ /* Send & drain traffic; cfd must retain its 16-page forward alloc,
+ * while sfd (0 reserve) reclaims its forward alloc back to 0.
+ */
+ ASSERT_EQ(send(cfd, buf, sizeof(buf), 0), (ssize_t)sizeof(buf));
+ ASSERT_EQ(recv(sfd, buf, sizeof(buf), MSG_WAITALL), (ssize_t)sizeof(buf));
+ wait_wmem_drained(_metadata, cfd, 16 * ps);
+
+ EXPECT_EQ(get_fwd_alloc(_metadata, sfd), 0U);
+
+ close(sfd);
+ close(cfd);
+ close(lfd);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/nfsd/.gitignore b/tools/testing/selftests/nfsd/.gitignore
new file mode 100644
index 000000000000..19e6dec04d8e
--- /dev/null
+++ b/tools/testing/selftests/nfsd/.gitignore
@@ -0,0 +1 @@
+nfsd_netlink_listener
diff --git a/tools/testing/selftests/nfsd/Makefile b/tools/testing/selftests/nfsd/Makefile
new file mode 100644
index 000000000000..15ac65549d25
--- /dev/null
+++ b/tools/testing/selftests/nfsd/Makefile
@@ -0,0 +1,6 @@
+# SPDX-License-Identifier: GPL-2.0
+CFLAGS += $(KHDR_INCLUDES) -Wall
+
+TEST_GEN_PROGS := nfsd_netlink_listener
+
+include ../lib.mk
diff --git a/tools/testing/selftests/nfsd/config b/tools/testing/selftests/nfsd/config
new file mode 100644
index 000000000000..0eef03af3503
--- /dev/null
+++ b/tools/testing/selftests/nfsd/config
@@ -0,0 +1,14 @@
+CONFIG_NAMESPACES=y
+CONFIG_NET_NS=y
+CONFIG_SHMEM=y
+CONFIG_TMPFS=y
+CONFIG_UNIX=y
+CONFIG_INET=y
+CONFIG_IPV6=y
+CONFIG_MULTIUSER=y
+CONFIG_PROC_FS=y
+CONFIG_FILE_LOCKING=y
+CONFIG_INOTIFY_USER=y
+CONFIG_SUNRPC=y
+CONFIG_NFSD=y
+CONFIG_NFSD_V4=y
diff --git a/tools/testing/selftests/nfsd/nfsd_netlink_listener.c b/tools/testing/selftests/nfsd/nfsd_netlink_listener.c
new file mode 100644
index 000000000000..106360f87b99
--- /dev/null
+++ b/tools/testing/selftests/nfsd/nfsd_netlink_listener.c
@@ -0,0 +1,1323 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Regression tests for the NFSD generic-netlink listener interface
+ * (NFSD_CMD_LISTENER_SET / NFSD_CMD_LISTENER_GET).
+ *
+ * Three groups:
+ * validation - malformed/abusive LISTENER_SET requests are rejected by
+ * nfsd_nl_validate_listeners(), before nfsd_mutex is taken.
+ * functional - create/add/remove listeners and verify LISTENER_GET
+ * reflects the set (round-trip of transport + addr:port).
+ * semantics - once threads are running (THREADS_SET) a listener change
+ * is refused with -EBUSY.
+ *
+ * Each test runs in its own private net + mount namespace (unshare in
+ * FIXTURE_SETUP). /run is masked there: a pathname AF_LOCAL connect is not
+ * scoped by the network namespace, since unix_find_bsd() resolves by inode
+ * and takes no struct net, so the kernel's rpcbind client would otherwise be
+ * able to reach the rpcbind running on the host. Anything that creates a
+ * serv is served by the per-netns rpcbind stub below instead.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <poll.h>
+#include <sched.h>
+#include <signal.h>
+#include <stddef.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mman.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/socket.h>
+#include <sys/ioctl.h>
+#include <sys/stat.h>
+#include <sys/time.h>
+#include <sys/un.h>
+#include <sys/wait.h>
+#include <net/if.h>
+#include <netinet/in.h>
+#include <linux/netlink.h>
+#include <linux/genetlink.h>
+#include <linux/nfsd_netlink.h>
+
+#include "../kselftest_harness.h"
+
+#define NLA_ALIGN4(len) (((len) + 3) & ~3)
+#define TEST_PORT 20049
+#define MAX_LISTENERS 8
+#define RECV_TIMEO_SEC 30
+
+static int nfsd_family = -1; /* set per-test in FIXTURE_SETUP */
+
+/* Extack message from the last genl_request(); empty if there was none. */
+static char last_extack[128];
+
+static void die(const char *msg)
+{
+ perror(msg);
+ exit(1);
+}
+
+/* ------------------- minimal generic-netlink plumbing ------------------- */
+
+static int genl_open(void)
+{
+ struct sockaddr_nl sa = { .nl_family = AF_NETLINK };
+ struct timeval tv = { .tv_sec = RECV_TIMEO_SEC };
+ int fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_GENERIC);
+ int on = 1;
+
+ if (fd < 0)
+ die("socket(NETLINK_GENERIC)");
+ if (bind(fd, (void *)&sa, sizeof(sa)) < 0)
+ die("bind(netlink)");
+ setsockopt(fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv));
+ /*
+ * Ask for extack, and cap the ack so the request is not echoed back:
+ * the TLVs then always follow the fixed part of the error message.
+ */
+ setsockopt(fd, SOL_NETLINK, NETLINK_EXT_ACK, &on, sizeof(on));
+ setsockopt(fd, SOL_NETLINK, NETLINK_CAP_ACK, &on, sizeof(on));
+ return fd;
+}
+
+/* Stash the extack message of an ack, if it carries one. */
+static void parse_extack(const char *rbuf)
+{
+ const struct nlmsghdr *nlh = (const void *)rbuf;
+ const struct nlattr *na;
+ int off, left;
+
+ last_extack[0] = '\0';
+ if (nlh->nlmsg_type != NLMSG_ERROR ||
+ !(nlh->nlmsg_flags & NLM_F_ACK_TLVS))
+ return;
+
+ off = NLMSG_HDRLEN + NLMSG_ALIGN(sizeof(struct nlmsgerr));
+ left = nlh->nlmsg_len - off;
+ na = (const void *)(rbuf + off);
+
+ while (left >= (int)NLA_HDRLEN) {
+ if ((na->nla_type & NLA_TYPE_MASK) == NLMSGERR_ATTR_MSG) {
+ strncpy(last_extack, (const char *)na + NLA_HDRLEN,
+ sizeof(last_extack) - 1);
+ last_extack[sizeof(last_extack) - 1] = '\0';
+ return;
+ }
+ left -= NLA_ALIGN4(na->nla_len);
+ na = (const void *)((const char *)na + NLA_ALIGN4(na->nla_len));
+ }
+}
+
+/* Append an attribute at @off; return the new (aligned) offset. */
+static int put_attr(char *buf, int off, uint16_t type,
+ const void *data, int len)
+{
+ struct nlattr *na = (void *)(buf + off);
+
+ na->nla_type = type;
+ na->nla_len = NLA_HDRLEN + len;
+ if (len)
+ memcpy(buf + off + NLA_HDRLEN, data, len);
+ return off + NLA_ALIGN4(NLA_HDRLEN + len);
+}
+
+/* Build a genl message header into @buf; return the offset past it. */
+static int genl_hdr(char *buf, uint16_t type, uint16_t flags, uint8_t cmd)
+{
+ struct nlmsghdr *nlh = (void *)buf;
+ struct genlmsghdr *gnl = (void *)(buf + NLMSG_HDRLEN);
+
+ memset(buf, 0, NLMSG_HDRLEN + GENL_HDRLEN);
+ nlh->nlmsg_type = type;
+ nlh->nlmsg_flags = flags;
+ nlh->nlmsg_seq = 1;
+ gnl->cmd = cmd;
+ gnl->version = 1;
+ return NLMSG_HDRLEN + GENL_HDRLEN;
+}
+
+/* Send an nfsd command with an ACK; return the ACK errno (<= 0). */
+static int genl_request(uint8_t cmd, const char *attrs, int attrs_len)
+{
+ char buf[1 << 20], rbuf[4096];
+ struct nlmsghdr *nlh = (void *)buf;
+ int fd = genl_open();
+ int off, n, ret;
+
+ off = genl_hdr(buf, nfsd_family, NLM_F_REQUEST | NLM_F_ACK, cmd);
+ if (attrs_len) {
+ memcpy(buf + off, attrs, attrs_len);
+ off += attrs_len;
+ }
+ nlh->nlmsg_len = off;
+
+ if (send(fd, buf, off, 0) < 0)
+ die("send(genl)");
+
+ last_extack[0] = '\0';
+ n = recv(fd, rbuf, sizeof(rbuf), 0);
+ if (n < 0) {
+ ret = (errno == EAGAIN || errno == EWOULDBLOCK) ? -ETIMEDOUT : -errno;
+ } else if (((struct nlmsghdr *)rbuf)->nlmsg_type == NLMSG_ERROR) {
+ ret = ((struct nlmsgerr *)NLMSG_DATA(rbuf))->error;
+ parse_extack(rbuf);
+ } else {
+ ret = 0;
+ }
+ close(fd);
+ return ret;
+}
+
+/* Send a command and return the full reply message; -errno on failure. */
+static int genl_request_reply(uint8_t cmd, char *rbuf, size_t rlen)
+{
+ char buf[256];
+ struct nlmsghdr *nlh = (void *)buf;
+ int fd = genl_open();
+ int off, n, ret;
+
+ off = genl_hdr(buf, nfsd_family, NLM_F_REQUEST, cmd);
+ nlh->nlmsg_len = off;
+
+ if (send(fd, buf, off, 0) < 0)
+ die("send(genl reply)");
+
+ n = recv(fd, rbuf, rlen, 0);
+ if (n < 0)
+ ret = (errno == EAGAIN || errno == EWOULDBLOCK) ? -ETIMEDOUT : -errno;
+ else if (((struct nlmsghdr *)rbuf)->nlmsg_type == NLMSG_ERROR)
+ ret = ((struct nlmsgerr *)NLMSG_DATA(rbuf))->error;
+ else
+ ret = n;
+ close(fd);
+ return ret;
+}
+
+/* Resolve the "nfsd" genl family id; -1 if not registered. */
+static int genl_resolve_nfsd(void)
+{
+ char buf[1024], rbuf[4096];
+ struct nlmsghdr *nlh = (void *)buf;
+ struct nlmsghdr *rh = (void *)rbuf;
+ struct nlattr *na;
+ int fd, off, left, id = -1;
+
+ fd = genl_open();
+ off = genl_hdr(buf, GENL_ID_CTRL, NLM_F_REQUEST, CTRL_CMD_GETFAMILY);
+ off = put_attr(buf, off, CTRL_ATTR_FAMILY_NAME,
+ NFSD_FAMILY_NAME, sizeof(NFSD_FAMILY_NAME));
+ nlh->nlmsg_len = off;
+
+ if (send(fd, buf, off, 0) < 0)
+ die("send(GETFAMILY)");
+ if (recv(fd, rbuf, sizeof(rbuf), 0) < 0)
+ die("recv(GETFAMILY)");
+ close(fd);
+
+ if (rh->nlmsg_type == NLMSG_ERROR)
+ return -1;
+
+ na = (void *)((char *)NLMSG_DATA(rh) + GENL_HDRLEN);
+ left = rh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN;
+ while (left >= (int)NLA_HDRLEN) {
+ if (na->nla_type == CTRL_ATTR_FAMILY_ID) {
+ id = *(uint16_t *)((char *)na + NLA_HDRLEN);
+ break;
+ }
+ left -= NLA_ALIGN4(na->nla_len);
+ na = (void *)((char *)na + NLA_ALIGN4(na->nla_len));
+ }
+ return id;
+}
+
+/* ------------------- listener request builders ------------------- */
+
+/* Fine-grained control for negative tests: any field can be omitted/malformed. */
+struct raw_listener {
+ const char *xprt; /* NULL -> omit NFSD_A_SOCK_TRANSPORT_NAME */
+ int emit_addr; /* 0 -> omit NFSD_A_SOCK_ADDR */
+ const void *addr;
+ int addr_len; /* bytes to emit for NFSD_A_SOCK_ADDR */
+};
+
+static int put_raw_listener(char *buf, int off, const struct raw_listener *r)
+{
+ struct nlattr *nest = (void *)(buf + off);
+ int inner = off + NLA_HDRLEN;
+
+ if (r->emit_addr)
+ inner = put_attr(buf, inner, NFSD_A_SOCK_ADDR, r->addr, r->addr_len);
+ if (r->xprt)
+ inner = put_attr(buf, inner, NFSD_A_SOCK_TRANSPORT_NAME,
+ r->xprt, strlen(r->xprt) + 1);
+ nest->nla_type = NFSD_A_SERVER_SOCK_ADDR | NLA_F_NESTED;
+ nest->nla_len = inner - off;
+ return off + NLA_ALIGN4(nest->nla_len);
+}
+
+/* Well-formed loopback listener for @family (AF_INET or AF_INET6). */
+static int put_listener_af(char *buf, int off, const char *xprt, int family,
+ uint16_t port)
+{
+ struct sockaddr_storage ss = {0};
+ struct raw_listener r = { .xprt = xprt, .emit_addr = 1, .addr = &ss };
+
+ if (family == AF_INET6) {
+ struct sockaddr_in6 *s6 = (void *)&ss;
+
+ s6->sin6_family = AF_INET6;
+ s6->sin6_port = htons(port);
+ s6->sin6_addr = in6addr_loopback;
+ r.addr_len = sizeof(*s6);
+ } else {
+ struct sockaddr_in *s4 = (void *)&ss;
+
+ s4->sin_family = AF_INET;
+ s4->sin_port = htons(port);
+ s4->sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ r.addr_len = sizeof(*s4);
+ }
+ return put_raw_listener(buf, off, &r);
+}
+
+static int put_listener(char *buf, int off, const char *xprt, uint16_t port)
+{
+ return put_listener_af(buf, off, xprt, AF_INET, port);
+}
+
+/* ------------------- LISTENER_GET parsing ------------------- */
+
+struct listener_ent {
+ char xprt[16];
+ int family;
+ uint16_t port;
+ struct in_addr a4;
+ struct in6_addr a6;
+};
+
+static int parse_listener_get(const char *rbuf, int len,
+ struct listener_ent *out, int max)
+{
+ const struct nlmsghdr *nlh = (const void *)rbuf;
+ const struct nlattr *na;
+ int left, count = 0;
+
+ (void)len;
+ na = (const void *)(rbuf + NLMSG_HDRLEN + GENL_HDRLEN);
+ left = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN;
+
+ while (left >= (int)NLA_HDRLEN) {
+ int alen = na->nla_len;
+
+ if ((na->nla_type & NLA_TYPE_MASK) == NFSD_A_SERVER_SOCK_ADDR &&
+ count < max) {
+ const struct nlattr *in = (const void *)((char *)na + NLA_HDRLEN);
+ int ileft = alen - NLA_HDRLEN;
+ struct listener_ent *e = &out[count];
+
+ memset(e, 0, sizeof(*e));
+ while (ileft >= (int)NLA_HDRLEN) {
+ const void *d = (const char *)in + NLA_HDRLEN;
+ int t = in->nla_type & NLA_TYPE_MASK;
+
+ if (t == NFSD_A_SOCK_TRANSPORT_NAME) {
+ strncpy(e->xprt, d, sizeof(e->xprt) - 1);
+ } else if (t == NFSD_A_SOCK_ADDR) {
+ const struct sockaddr_storage *ss = d;
+
+ e->family = ss->ss_family;
+ if (ss->ss_family == AF_INET) {
+ const struct sockaddr_in *s = d;
+
+ e->a4 = s->sin_addr;
+ e->port = ntohs(s->sin_port);
+ } else if (ss->ss_family == AF_INET6) {
+ const struct sockaddr_in6 *s = d;
+
+ e->a6 = s->sin6_addr;
+ e->port = ntohs(s->sin6_port);
+ }
+ }
+ ileft -= NLA_ALIGN4(in->nla_len);
+ in = (const void *)((char *)in + NLA_ALIGN4(in->nla_len));
+ }
+ count++;
+ }
+ left -= NLA_ALIGN4(alen);
+ na = (const void *)((char *)na + NLA_ALIGN4(alen));
+ }
+ return count;
+}
+
+/* ------------------- convenience wrappers ------------------- */
+
+static int listener_set(const char *attrs, int len)
+{
+ return genl_request(NFSD_CMD_LISTENER_SET, attrs, len);
+}
+
+/*
+ * Enable exactly one NFS version in this netns. NFSD_CMD_VERSION_SET clears
+ * every version first, so one nest is enough to leave the server v4-only.
+ * It refuses once a serv exists, so call it before any listener.
+ */
+static int version_set_only(uint32_t major, uint32_t minor)
+{
+ char attrs[64];
+ struct nlattr *nest = (void *)attrs;
+ int inner = NLA_HDRLEN;
+
+ inner = put_attr(attrs, inner, NFSD_A_VERSION_MAJOR,
+ &major, sizeof(major));
+ inner = put_attr(attrs, inner, NFSD_A_VERSION_MINOR,
+ &minor, sizeof(minor));
+ inner = put_attr(attrs, inner, NFSD_A_VERSION_ENABLED, NULL, 0);
+ nest->nla_type = NFSD_A_SERVER_PROTO_VERSION | NLA_F_NESTED;
+ nest->nla_len = inner;
+
+ return genl_request(NFSD_CMD_VERSION_SET, attrs, NLA_ALIGN4(inner));
+}
+
+/* Fetch the current listeners; returns count (>=0) or -errno. */
+static int listener_get(struct listener_ent *out, int max)
+{
+ char rbuf[8192];
+ int n = genl_request_reply(NFSD_CMD_LISTENER_GET, rbuf, sizeof(rbuf));
+
+ if (n < 0)
+ return n;
+ return parse_listener_get(rbuf, n, out, max);
+}
+
+/*
+ * Every listener these tests create comes from put_listener_af(), so the
+ * address is always loopback. Match on it too: without that, a reply that
+ * gave the right transport and port on the wrong address (0.0.0.0, say)
+ * would pass.
+ */
+static struct listener_ent *find_listener(struct listener_ent *e, int n,
+ const char *xprt, int family,
+ uint16_t port)
+{
+ int i;
+
+ for (i = 0; i < n; i++) {
+ if (e[i].family != family || e[i].port != port ||
+ strcmp(e[i].xprt, xprt))
+ continue;
+ if (family == AF_INET6) {
+ if (memcmp(&e[i].a6, &in6addr_loopback, sizeof(e[i].a6)))
+ continue;
+ } else if (e[i].a4.s_addr != htonl(INADDR_LOOPBACK)) {
+ continue;
+ }
+ return &e[i];
+ }
+ return NULL;
+}
+
+/* Start (@n > 0) or stop (@n == 0) nfsd threads in this netns. */
+static int threads_set(int n)
+{
+ char attrs[64];
+ uint32_t v = n;
+ int off = put_attr(attrs, 0, NFSD_A_SERVER_THREADS, &v, sizeof(v));
+
+ return genl_request(NFSD_CMD_THREADS_SET, attrs, off);
+}
+
+/* ------------------- per-netns local rpcbind stub ------------------- */
+
+/*
+ * Creating a listener registers with rpcbind: nfsd_nl_listener_set_doit()
+ * passes no SVC_SOCK_ANONYMOUS for the first entry of a request, so
+ * pmap_register is true in svc_setup_socket(). The fixture's server has v3
+ * enabled, and nfsd_version3 does not set vs_rpcb_optnl, so a failure there
+ * comes back out of svc_register() and takes the listener down with it.
+ * With nothing listening, every attempt first waits out the local rpcbind
+ * timeout. The abstract AF_LOCAL name the kernel tries first is per-netns
+ * (unix_find_abstract() takes a struct net), so answer it here and stay out
+ * of the host's rpcbind.
+ *
+ * Arguments are never decoded. The NULL procedure gets an empty success and
+ * SET/UNSET get TRUE, for both RPCBVERS_2 and RPCBVERS_4. v4 has to be
+ * answered because __svc_rpcb_register6() turns a v4 refusal into
+ * -EAFNOSUPPORT, which would leave every IPv6 listener unregistered.
+ *
+ * In RPCB_STUB_REFUSE mode SET is answered FALSE instead, which
+ * rpcb_register_call() reports as -EACCES. UNSET is left alone: only
+ * svc_unregister() issues it, and it discards the result.
+ *
+ * In RPCB_STUB_SILENT mode a SET or an UNSET is read and nothing is written
+ * back, so the kernel waits out its own timeout. That is the only mode that
+ * makes rpcb_register_call() report a call that got no answer, which is what
+ * the per-net failure count records. The NULL procedure is still answered:
+ * rpcb_create_af_local() builds its client without RPC_CLNT_CREATE_NOPING, so
+ * rpc_create() pings, and a ping that goes unanswered drops the kernel onto
+ * the loopback rpcb_create_local_net() client, which never reaches this stub.
+ *
+ * The stub also keeps counters and the mode in a page shared with the test, so
+ * a test can assert that the kernel never talked to rpcbind at all, or that it
+ * dropped the local rpcbind client and had to reconnect.
+ *
+ * The mode lives there rather than in the child so that a test can change it
+ * with a serv already up. Killing and restarting the stub would close the
+ * connection the kernel holds, and rpcb_register_call() issues UNSET over
+ * AF_LOCAL with RPC_TASK_NOCONNECT, so the next call would fail at once with
+ * -ENOTCONN instead of waiting out a timeout.
+ */
+#define RPCB_PROGRAM 100000
+#define RPCB_PROC_NULL 0
+#define RPCB_PROC_SET 1
+#define RPCB_PROC_UNSET 2
+#define RPCB_ABSTRACT_NAME "/run/rpcbind.sock"
+#define RPCB_STUB_MAXCONN 4
+
+enum { RPCB_STUB_ACCEPT, RPCB_STUB_REFUSE, RPCB_STUB_SILENT };
+
+struct rpcb_stub_stats {
+ unsigned int conns; /* connections accepted */
+ unsigned int calls; /* calls received */
+ unsigned int mode; /* RPCB_STUB_*, read on every call */
+};
+
+static volatile struct rpcb_stub_stats *rpcb_stats; /* MAP_SHARED */
+
+static int rpcb_stats_alloc(void)
+{
+ void *p = mmap(NULL, sizeof(*rpcb_stats), PROT_READ | PROT_WRITE,
+ MAP_SHARED | MAP_ANONYMOUS, -1, 0);
+
+ if (p == MAP_FAILED)
+ return -1;
+ rpcb_stats = p;
+ return 0;
+}
+
+/*
+ * The stub bumps these before it replies and the kernel waits for that reply,
+ * so whatever a netlink request provoked is visible once it returns.
+ */
+static int rpcb_calls(void)
+{
+ return rpcb_stats ? (int)rpcb_stats->calls : 0;
+}
+
+static int rpcb_conns(void)
+{
+ return rpcb_stats ? (int)rpcb_stats->conns : 0;
+}
+
+/* Takes effect on the stub's next call; the caller has not sent one yet. */
+static void rpcb_stub_set_mode(int mode)
+{
+ rpcb_stats->mode = mode;
+}
+
+static int rpcb_stub_listen(void)
+{
+ struct sockaddr_un sun = { .sun_family = AF_UNIX };
+ size_t nlen = strlen(RPCB_ABSTRACT_NAME);
+ socklen_t alen;
+ int fd;
+
+ /* Abstract names are length-delimited, so the length must match. */
+ memcpy(sun.sun_path + 1, RPCB_ABSTRACT_NAME, nlen);
+ alen = offsetof(struct sockaddr_un, sun_path) + 1 + nlen;
+
+ fd = socket(AF_UNIX, SOCK_STREAM, 0);
+ if (fd < 0)
+ return -1;
+ if (bind(fd, (struct sockaddr *)&sun, alen) < 0 ||
+ listen(fd, RPCB_STUB_MAXCONN) < 0) {
+ close(fd);
+ return -1;
+ }
+ return fd;
+}
+
+static int rpcb_stub_read(int fd, void *buf, size_t len)
+{
+ size_t done = 0;
+
+ while (done < len) {
+ ssize_t n = read(fd, (char *)buf + done, len - done);
+
+ if (n <= 0)
+ return -1;
+ done += n;
+ }
+ return 0;
+}
+
+/* Handle one record-marked RPC call. Returns -1 when the peer is done. */
+static int rpcb_stub_call(int fd)
+{
+ unsigned int len, nrep = 6, mode = rpcb_stats->mode;
+ uint32_t mark, call[6], rep[7];
+ size_t replen;
+
+ if (rpcb_stub_read(fd, &mark, sizeof(mark)))
+ return -1;
+ len = ntohl(mark) & 0x7fffffff;
+ if (len < sizeof(call) || len > 4096)
+ return -1;
+ if (rpcb_stub_read(fd, call, sizeof(call)))
+ return -1;
+
+ /* xid, msg_type, rpcvers, prog, vers, proc; the rest is discarded */
+ for (len -= sizeof(call); len; ) {
+ char sink[256];
+ unsigned int n = len > sizeof(sink) ? sizeof(sink) : len;
+
+ if (rpcb_stub_read(fd, sink, n))
+ return -1;
+ len -= n;
+ }
+
+ if (rpcb_stats)
+ rpcb_stats->calls++;
+
+ rep[0] = call[0]; /* xid */
+ rep[1] = htonl(1); /* REPLY */
+ rep[2] = htonl(0); /* MSG_ACCEPTED */
+ rep[3] = htonl(0); /* verifier flavor AUTH_NULL */
+ rep[4] = htonl(0); /* verifier length */
+ rep[5] = htonl(0); /* SUCCESS */
+
+ if (ntohl(call[3]) != RPCB_PROGRAM) {
+ rep[5] = htonl(1); /* PROG_UNAVAIL */
+ } else {
+ unsigned int proc = ntohl(call[5]);
+
+ switch (proc) {
+ case RPCB_PROC_NULL:
+ break;
+ case RPCB_PROC_SET:
+ rep[6] = htonl(mode == RPCB_STUB_REFUSE ? 0 : 1);
+ nrep = 7;
+ break;
+ case RPCB_PROC_UNSET:
+ rep[6] = htonl(1); /* TRUE */
+ nrep = 7;
+ break;
+ default:
+ rep[5] = htonl(3); /* PROC_UNAVAIL */
+ }
+
+ /*
+ * Answer nothing, so the caller waits out its timeout. The
+ * NULL procedure is answered even here: the kernel pings at
+ * client creation, and a ping with no answer takes it off
+ * this socket entirely.
+ */
+ if (mode == RPCB_STUB_SILENT && proc != RPCB_PROC_NULL)
+ return 0;
+ }
+
+ replen = nrep * sizeof(rep[0]);
+ mark = htonl(0x80000000 | replen);
+ if (write(fd, &mark, sizeof(mark)) != (ssize_t)sizeof(mark) ||
+ write(fd, rep, replen) != (ssize_t)replen)
+ return -1;
+ return 0;
+}
+
+static void rpcb_stub_serve(int lfd)
+{
+ struct pollfd pfd[1 + RPCB_STUB_MAXCONN];
+ nfds_t n = 1, i;
+
+ pfd[0].fd = lfd;
+
+ for (;;) {
+ /* stop polling the listener when full, or poll() spins */
+ pfd[0].events = n < 1 + RPCB_STUB_MAXCONN ? POLLIN : 0;
+
+ if (poll(pfd, n, -1) < 0)
+ return;
+
+ if (pfd[0].revents & POLLIN) {
+ int c = accept(lfd, NULL, NULL);
+
+ if (c >= 0) {
+ pfd[n].fd = c;
+ pfd[n].events = POLLIN;
+ /*
+ * poll() ran with the old n, so it did not
+ * write this revents. The loop below reads it.
+ */
+ pfd[n].revents = 0;
+ n++;
+ if (rpcb_stats)
+ rpcb_stats->conns++;
+ }
+ }
+
+ for (i = 1; i < n; i++) {
+ if (!(pfd[i].revents & (POLLIN | POLLHUP | POLLERR)))
+ continue;
+ if (rpcb_stub_call(pfd[i].fd)) {
+ close(pfd[i].fd);
+ pfd[i] = pfd[--n];
+ }
+ }
+ }
+}
+
+/* Returns the stub's pid, or -1. The socket is listening before we fork. */
+static pid_t rpcb_stub_start(int mode)
+{
+ int lfd = rpcb_stub_listen();
+ pid_t pid;
+
+ if (lfd < 0)
+ return -1;
+
+ rpcb_stats->mode = mode;
+
+ pid = fork();
+ if (pid < 0) {
+ close(lfd);
+ return -1;
+ }
+ if (pid == 0) {
+ signal(SIGPIPE, SIG_IGN);
+ prctl(PR_SET_PDEATHSIG, SIGKILL);
+ if (getppid() == 1) /* raced with parent exit */
+ _exit(0);
+ rpcb_stub_serve(lfd);
+ _exit(0);
+ }
+
+ close(lfd);
+ return pid;
+}
+
+/* --------------------------- fixture --------------------------- */
+
+FIXTURE(nfsd_listener) {
+ pid_t rpcbd;
+};
+
+FIXTURE_SETUP(nfsd_listener)
+{
+ struct ifreq ifr = {0};
+ struct stat st;
+ int s;
+
+ if (geteuid() != 0)
+ SKIP(return, "must be run as root");
+ if (unshare(CLONE_NEWNET | CLONE_NEWNS) < 0)
+ SKIP(return, "unshare(NEWNET|NEWNS): %s", strerror(errno));
+ if (mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL) < 0)
+ SKIP(return, "mount(/ private): %s", strerror(errno));
+
+ /*
+ * Keep the kernel's rpcbind client inside this namespace. The
+ * abstract socket it tries first is per-netns, but the
+ * "/var/run/rpcbind.sock" fallback is not, so hide the path.
+ */
+ if (mount("tmpfs", "/run", "tmpfs", 0, NULL) < 0)
+ SKIP(return, "mount(tmpfs on /run): %s", strerror(errno));
+ if (lstat("/var/run", &st) == 0 && S_ISDIR(st.st_mode) &&
+ mount("tmpfs", "/var/run", "tmpfs", 0, NULL) < 0)
+ SKIP(return, "mount(tmpfs on /var/run): %s", strerror(errno));
+
+ /*
+ * Bring loopback up so listener binds (127.0.0.1 / ::1) work. Root
+ * without CAP_NET_ADMIN in this netns gets -EPERM here, so skip.
+ */
+ s = socket(AF_INET, SOCK_DGRAM, 0);
+ ASSERT_GE(s, 0);
+ strcpy(ifr.ifr_name, "lo");
+ if (ioctl(s, SIOCGIFFLAGS, &ifr) < 0) {
+ close(s);
+ SKIP(return, "SIOCGIFFLAGS(lo): %s", strerror(errno));
+ }
+ ifr.ifr_flags |= IFF_UP | IFF_RUNNING;
+ if (ioctl(s, SIOCSIFFLAGS, &ifr) < 0) {
+ close(s);
+ SKIP(return, "SIOCSIFFLAGS(lo): %s", strerror(errno));
+ }
+ close(s);
+
+ nfsd_family = genl_resolve_nfsd();
+ if (nfsd_family < 0)
+ SKIP(return, "nfsd genl family not found (modprobe nfsd?)");
+
+ if (rpcb_stats_alloc() < 0)
+ SKIP(return, "mmap(rpcbind stub counters): %s", strerror(errno));
+
+ self->rpcbd = rpcb_stub_start(RPCB_STUB_ACCEPT);
+ if (self->rpcbd < 0)
+ SKIP(return, "cannot start the rpcbind stub: %s",
+ strerror(errno));
+}
+
+FIXTURE_TEARDOWN(nfsd_listener)
+{
+ /*
+ * A listener holds a reference to this netns, which outlives the test
+ * process, so anything still up leaks it. Threads pin the listeners in
+ * turn; dropping them destroys the serv and everything under it.
+ */
+ if (nfsd_family >= 0 && listener_set(NULL, 0) == -EBUSY)
+ threads_set(0);
+
+ if (self->rpcbd > 0) {
+ kill(self->rpcbd, SIGKILL);
+ waitpid(self->rpcbd, NULL, 0);
+ }
+ if (rpcb_stats) {
+ munmap((void *)rpcb_stats, sizeof(*rpcb_stats));
+ rpcb_stats = NULL;
+ }
+}
+
+/* ===================== validation / negative ===================== */
+
+TEST_F(nfsd_listener, val_empty_list_ok)
+{
+ EXPECT_EQ(0, listener_set(NULL, 0));
+}
+
+TEST_F(nfsd_listener, val_too_many)
+{
+ static char attrs[1 << 20];
+ int i, off = 0;
+
+ for (i = 0; i < 1025; i++) /* > NFSD_NL_LISTENER_MAX (1024) */
+ off = put_listener(attrs, off, "udp", TEST_PORT);
+ EXPECT_EQ(-E2BIG, listener_set(attrs, off));
+}
+
+TEST_F(nfsd_listener, val_missing_addr)
+{
+ char attrs[64];
+ struct raw_listener r = { .xprt = "tcp", .emit_addr = 0 };
+ int off = put_raw_listener(attrs, 0, &r);
+
+ EXPECT_EQ(-EINVAL, listener_set(attrs, off));
+}
+
+TEST_F(nfsd_listener, val_missing_transport)
+{
+ struct sockaddr_in s4 = { .sin_family = AF_INET, .sin_port = htons(TEST_PORT) };
+ struct raw_listener r = { .xprt = NULL, .emit_addr = 1,
+ .addr = &s4, .addr_len = sizeof(s4) };
+ char attrs[64];
+ int off = put_raw_listener(attrs, 0, &r);
+
+ EXPECT_EQ(-EINVAL, listener_set(attrs, off));
+}
+
+/*
+ * A name matching no transport class must be refused before nfsd_mutex is
+ * taken, so it never reaches svc_xprt_create_from_sa() and its
+ * request_module("svc%s", name) upcall.
+ *
+ * The errno cannot show that -- svc_xprt_create_from_sa() returns
+ * -EPROTONOSUPPORT for an unknown name too. The rpcbind traffic can:
+ * getting that far means nfsd_create_serv() ran, and svc_bind() pings
+ * rpcbind at client creation and then sweeps stale entries with
+ * svc_unregister(). A silent stub is the proof nothing was created.
+ */
+TEST_F(nfsd_listener, val_bad_transport)
+{
+ char attrs[64];
+ int off = put_listener(attrs, 0, "bogus_xprt", TEST_PORT);
+
+ ASSERT_EQ(0, rpcb_calls());
+ EXPECT_EQ(-EPROTONOSUPPORT, listener_set(attrs, off));
+ EXPECT_EQ(0, rpcb_calls());
+}
+
+TEST_F(nfsd_listener, val_addr_too_short)
+{
+ unsigned char tiny = 0;
+ struct raw_listener r = { .xprt = "tcp", .emit_addr = 1,
+ .addr = &tiny, .addr_len = 1 };
+ char attrs[64];
+ int off = put_raw_listener(attrs, 0, &r);
+
+ EXPECT_EQ(-EINVAL, listener_set(attrs, off));
+}
+
+TEST_F(nfsd_listener, val_inet_short)
+{
+ struct sockaddr_in s4 = { .sin_family = AF_INET, .sin_port = htons(TEST_PORT) };
+ struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, .addr = &s4,
+ .addr_len = sizeof(sa_family_t) + 2 };
+ char attrs[64];
+ int off = put_raw_listener(attrs, 0, &r);
+
+ EXPECT_EQ(-EINVAL, listener_set(attrs, off));
+}
+
+TEST_F(nfsd_listener, val_inet6_short)
+{
+ struct sockaddr_in6 s6 = { .sin6_family = AF_INET6, .sin6_port = htons(TEST_PORT) };
+ struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, .addr = &s6,
+ .addr_len = sizeof(struct sockaddr_in) };
+ char attrs[64];
+ int off = put_raw_listener(attrs, 0, &r);
+
+ EXPECT_EQ(-EINVAL, listener_set(attrs, off));
+}
+
+TEST_F(nfsd_listener, val_bad_family)
+{
+ struct sockaddr_storage ss = { .ss_family = AF_UNIX };
+ struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, .addr = &ss,
+ .addr_len = sizeof(struct sockaddr_in) };
+ char attrs[64];
+ int off = put_raw_listener(attrs, 0, &r);
+
+ EXPECT_EQ(-EAFNOSUPPORT, listener_set(attrs, off));
+}
+
+TEST_F(nfsd_listener, val_second_entry_bad)
+{
+ struct sockaddr_storage ss = { .ss_family = AF_UNIX };
+ struct raw_listener bad = { .xprt = "tcp", .emit_addr = 1, .addr = &ss,
+ .addr_len = sizeof(struct sockaddr_in) };
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[128];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+
+ off = put_raw_listener(attrs, off, &bad);
+ /* The whole request is rejected during validation; nothing applied. */
+ EXPECT_EQ(-EAFNOSUPPORT, listener_set(attrs, off));
+ /*
+ * Again the errno alone does not say so: svc_xprt_create_from_sa()
+ * also returns -EAFNOSUPPORT, and the doit keeps the listeners it did
+ * manage to create, so the well-formed tcp entry ahead of the bad one
+ * would still be up.
+ */
+ EXPECT_EQ(0, listener_get(got, MAX_LISTENERS));
+}
+
+/*
+ * A rejected request must leave the listeners that are already up alone.
+ * The errno alone does not show that: svc_xprt_create_from_sa() returns
+ * -EPROTONOSUPPORT for an unknown name too. What differs is how far the
+ * request gets -- without the check in nfsd_nl_validate_listeners(),
+ * nfsd_nl_listener_set_doit() has already moved the unmatched tcp listener
+ * off sv_permsocks and run svc_xprt_destroy_all() on it by the time the
+ * name fails.
+ */
+TEST_F(nfsd_listener, val_reject_keeps_listeners)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char good[64], bad[64];
+ int og = put_listener(good, 0, "tcp", TEST_PORT);
+ int ob = put_listener(bad, 0, "bogus_xprt", TEST_PORT);
+
+ ASSERT_EQ(0, listener_set(good, og));
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+
+ EXPECT_EQ(-EPROTONOSUPPORT, listener_set(bad, ob));
+
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT));
+}
+
+/* ===================== functional / round-trip ===================== */
+
+/* LISTENER_GET with no serv in this netns returns an empty list. */
+TEST_F(nfsd_listener, func_get_empty)
+{
+ struct listener_ent got[MAX_LISTENERS];
+
+ EXPECT_EQ(0, listener_get(got, MAX_LISTENERS));
+}
+
+TEST_F(nfsd_listener, func_create_tcp)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+
+ ASSERT_EQ(0, listener_set(attrs, off));
+ EXPECT_STREQ("", last_extack); /* nothing to warn about */
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT));
+}
+
+TEST_F(nfsd_listener, func_create_udp)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off = put_listener(attrs, 0, "udp", TEST_PORT);
+
+ ASSERT_EQ(0, listener_set(attrs, off));
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "udp", AF_INET, TEST_PORT));
+}
+
+TEST_F(nfsd_listener, func_create_multi)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[128];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+
+ off = put_listener(attrs, off, "udp", TEST_PORT);
+ ASSERT_EQ(0, listener_set(attrs, off));
+ ASSERT_EQ(2, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 2, "tcp", AF_INET, TEST_PORT));
+ EXPECT_NE(NULL, find_listener(got, 2, "udp", AF_INET, TEST_PORT));
+}
+
+TEST_F(nfsd_listener, func_idempotent)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+
+ ASSERT_EQ(0, listener_set(attrs, off));
+ EXPECT_EQ(0, listener_set(attrs, off)); /* re-set same list */
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT));
+}
+
+TEST_F(nfsd_listener, func_add)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char one[64], two[128];
+ int o1 = put_listener(one, 0, "tcp", TEST_PORT);
+ int o2 = put_listener(two, 0, "tcp", TEST_PORT);
+
+ o2 = put_listener(two, o2, "udp", TEST_PORT);
+ ASSERT_EQ(0, listener_set(one, o1));
+ ASSERT_EQ(0, listener_set(two, o2)); /* add udp, keep tcp */
+ ASSERT_EQ(2, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 2, "tcp", AF_INET, TEST_PORT));
+ EXPECT_NE(NULL, find_listener(got, 2, "udp", AF_INET, TEST_PORT));
+}
+
+TEST_F(nfsd_listener, func_remove_subset)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char both[128], one[64];
+ int ob = put_listener(both, 0, "tcp", TEST_PORT);
+ int oo = put_listener(one, 0, "tcp", TEST_PORT);
+
+ ob = put_listener(both, ob, "udp", TEST_PORT);
+ ASSERT_EQ(0, listener_set(both, ob));
+ ASSERT_EQ(0, listener_set(one, oo)); /* drop udp */
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT));
+}
+
+/*
+ * LISTENER_GET cannot tell a destroyed serv from a live one with no
+ * permsocks: nfsd_nl_listener_get_doit() replies empty either way. The
+ * rpcbind client can. nfsd_destroy_serv() is the only path that reaches
+ * svc_xprt_destroy_all(..., unregister=true) -> svc_rpcb_cleanup() ->
+ * rpcb_put_local(), which drops the last user and shuts the local client
+ * down; the next serv then has to connect again. Leaving the serv in place
+ * would keep the first connection and the stub would see just the one.
+ */
+TEST_F(nfsd_listener, func_empty_destroys)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ int conns;
+
+ ASSERT_EQ(0, listener_set(attrs, off));
+ conns = rpcb_conns();
+ ASSERT_GT(conns, 0);
+
+ EXPECT_EQ(0, listener_set(NULL, 0)); /* empty -> destroy serv */
+ EXPECT_EQ(0, listener_get(got, MAX_LISTENERS));
+
+ ASSERT_EQ(0, listener_set(attrs, off));
+ EXPECT_GT(rpcb_conns(), conns);
+}
+
+TEST_F(nfsd_listener, func_ipv6)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off, s;
+
+ s = socket(AF_INET6, SOCK_STREAM, 0);
+ if (s < 0)
+ SKIP(return, "IPv6 unavailable: %s", strerror(errno));
+ close(s);
+
+ off = put_listener_af(attrs, 0, "tcp", AF_INET6, TEST_PORT);
+ ASSERT_EQ(0, listener_set(attrs, off));
+ ASSERT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET6, TEST_PORT));
+}
+
+/* ===================== rpcbind registration ===================== */
+
+/*
+ * A rpcbind that refuses the registration takes the listener down with it.
+ * svc_register() fails, so svc_setup_socket() fails, so no listener is
+ * created. -EACCES alone does not show that, since a bind can return it
+ * too, so read the listener set back as well.
+ */
+TEST_F(nfsd_listener, sem_register_refused)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+
+ rpcb_stub_set_mode(RPCB_STUB_REFUSE);
+
+ EXPECT_EQ(-EACCES, listener_set(attrs, off));
+ EXPECT_STRNE("", last_extack);
+ EXPECT_EQ(0, listener_get(got, MAX_LISTENERS));
+}
+
+/*
+ * A listener that cannot be created reports which one it was: the errno
+ * alone does not name the entry in a multi-listener request.
+ */
+TEST_F(nfsd_listener, sem_create_failure_extack)
+{
+ struct sockaddr_in s4 = { .sin_family = AF_INET,
+ .sin_port = htons(TEST_PORT),
+ .sin_addr.s_addr = htonl(INADDR_LOOPBACK) };
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[64];
+ int off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ int s;
+
+ /* squat on the port so the listener cannot bind */
+ s = socket(AF_INET, SOCK_STREAM, 0);
+ ASSERT_GE(s, 0);
+ ASSERT_EQ(0, bind(s, (struct sockaddr *)&s4, sizeof(s4)));
+
+ EXPECT_EQ(-EADDRINUSE, listener_set(attrs, off));
+ EXPECT_STRNE("", last_extack);
+ EXPECT_EQ(0, listener_get(got, MAX_LISTENERS));
+ close(s);
+}
+
+/* ============ one rpcbind attempt for each request ============ */
+
+/*
+ * Every listener used to register on its own, so a rpcbind that never
+ * answers cost one timeout for each entry. Ask for one listener, then for
+ * three, and compare what the stub saw. Three entries must not cost three
+ * times as much.
+ *
+ * The stub has to stay silent rather than refuse. A refusal is an answer,
+ * and rpcbind refuses one entry at a time, so the count ignores it.
+ */
+TEST_F(nfsd_listener, rpcb_stop_after_failure)
+{
+ int before, one, three, off;
+ char attrs[192];
+
+ rpcb_stub_set_mode(RPCB_STUB_SILENT);
+
+ before = rpcb_calls();
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ listener_set(attrs, off);
+ one = rpcb_calls() - before;
+ ASSERT_GT(one, 0);
+
+ ASSERT_EQ(0, listener_set(attrs, 0));
+
+ before = rpcb_calls();
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 1);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 2);
+ listener_set(attrs, off);
+ three = rpcb_calls() - before;
+
+ /* the second and third entries must not reach rpcbind at all */
+ EXPECT_LE(three, one);
+}
+
+/*
+ * The entry that finds rpcbind silent is the one that pays for the
+ * discovery, and v3 has no vs_rpcb_optnl to discard the error, so it is the
+ * only entry whose listener would be lost. Nothing distinguishes it from the
+ * rest of the request, and a retry of the same request would fail the same
+ * entry again, so the set would stay short for as long as rpcbind was quiet.
+ *
+ * Ask for three listeners against a silent stub and require the whole set,
+ * a success, and a warning that says why.
+ */
+TEST_F(nfsd_listener, rpcb_silent_set_complete)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char attrs[192];
+ int off;
+
+ rpcb_stub_set_mode(RPCB_STUB_SILENT);
+
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 1);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 2);
+ EXPECT_EQ(0, listener_set(attrs, off));
+
+ /* the first entry is not the odd one out */
+ EXPECT_EQ(3, listener_get(got, MAX_LISTENERS));
+ /* no errno reports this, so the ack has to */
+ EXPECT_STRNE("", last_extack);
+}
+
+/*
+ * The case that needs the count rather than a failed listener. NFSv4 sets
+ * vs_rpcb_optnl, so svc_generic_rpcbind_set() discards the error, every
+ * listener comes up, and nothing reports a failure. Without the fix each
+ * entry still waits for rpcbind on its own.
+ *
+ * Make the server v4-only, answer no SET, and require three things: the
+ * listeners come up, the ack warns that they are not registered, and the
+ * stub does not see one round trip for each entry.
+ */
+TEST_F(nfsd_listener, rpcb_v4_only_bounded)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ int before, one, three, off;
+ char attrs[192];
+
+ /* refuses once a serv exists, so this has to come first */
+ ASSERT_EQ(0, version_set_only(4, 1));
+ rpcb_stub_set_mode(RPCB_STUB_SILENT);
+
+ before = rpcb_calls();
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ ASSERT_EQ(0, listener_set(attrs, off));
+ one = rpcb_calls() - before;
+ ASSERT_GT(one, 0);
+
+ /* start over, so the second measurement also builds a serv */
+ ASSERT_EQ(0, listener_set(attrs, 0));
+
+ before = rpcb_calls();
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 1);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 2);
+ ASSERT_EQ(0, listener_set(attrs, off));
+ three = rpcb_calls() - before;
+
+ /* the listeners are up even though rpcbind never answered */
+ EXPECT_EQ(3, listener_get(got, MAX_LISTENERS));
+ /* and the ack says they are unregistered, since no errno can */
+ EXPECT_STRNE("", last_extack);
+ EXPECT_LE(three, one);
+}
+
+/*
+ * The stop applies to one request only. After rpcbind starts answering,
+ * the next request must register without any other step.
+ */
+TEST_F(nfsd_listener, rpcb_retry_next_request)
+{
+ int before, after, off;
+ char attrs[192];
+
+ rpcb_stub_set_mode(RPCB_STUB_SILENT);
+
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 1);
+ listener_set(attrs, off);
+ ASSERT_EQ(0, listener_set(attrs, 0));
+
+ /* rpcbind recovers */
+ rpcb_stub_set_mode(RPCB_STUB_ACCEPT);
+
+ before = rpcb_calls();
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ EXPECT_EQ(0, listener_set(attrs, off));
+ after = rpcb_calls();
+
+ /* a fresh request starts from a fresh reading and tries again */
+ EXPECT_GT(after, before);
+ EXPECT_STREQ("", last_extack);
+}
+
+/*
+ * The same rule on the way out. Removing a listener unregisters it, so a
+ * rpcbind that stops answering used to cost one timeout for each listener
+ * removed. Register one listener while the stub answers, silence the stub,
+ * remove it and count; then do the same with three.
+ *
+ * Both measurements also pay the svc_unregister() sweep that
+ * nfsd_destroy_serv() runs once the last listener is gone, so that cancels
+ * out of the comparison.
+ */
+TEST_F(nfsd_listener, rpcb_unreg_stop_after_failure)
+{
+ int before, one, three, off;
+ char attrs[192];
+
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ ASSERT_EQ(0, listener_set(attrs, off));
+
+ rpcb_stub_set_mode(RPCB_STUB_SILENT);
+ before = rpcb_calls();
+ ASSERT_EQ(0, listener_set(NULL, 0));
+ one = rpcb_calls() - before;
+ ASSERT_GT(one, 0);
+
+ rpcb_stub_set_mode(RPCB_STUB_ACCEPT);
+ off = put_listener(attrs, 0, "tcp", TEST_PORT);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 1);
+ off = put_listener(attrs, off, "tcp", TEST_PORT + 2);
+ ASSERT_EQ(0, listener_set(attrs, off));
+
+ rpcb_stub_set_mode(RPCB_STUB_SILENT);
+ before = rpcb_calls();
+ ASSERT_EQ(0, listener_set(NULL, 0));
+ three = rpcb_calls() - before;
+
+ /* the second and third removals must not reach rpcbind at all */
+ EXPECT_LE(three, one);
+}
+
+/* ===================== threads / -EBUSY semantics ===================== */
+
+TEST_F(nfsd_listener, sem_busy_on_change)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char one[64], two[128];
+ int o1 = put_listener(one, 0, "tcp", TEST_PORT);
+ int o2 = put_listener(two, 0, "tcp", TEST_PORT);
+
+ o2 = put_listener(two, o2, "udp", TEST_PORT);
+ ASSERT_EQ(0, listener_set(one, o1));
+ ASSERT_EQ(0, threads_set(1)); /* threads now running */
+ EXPECT_EQ(-EBUSY, listener_set(two, o2)); /* add refused */
+
+ /* refused means refused: the udp listener must not have been added */
+ EXPECT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT));
+
+ threads_set(0); /* stop before netns exit */
+}
+
+TEST_F(nfsd_listener, sem_busy_on_remove)
+{
+ struct listener_ent got[MAX_LISTENERS];
+ char one[64];
+ int o1 = put_listener(one, 0, "tcp", TEST_PORT);
+
+ ASSERT_EQ(0, listener_set(one, o1));
+ ASSERT_EQ(0, threads_set(1));
+ EXPECT_EQ(-EBUSY, listener_set(NULL, 0)); /* remove refused */
+
+ /* the doit moves the permsocks to a temp list before it can fail */
+ EXPECT_EQ(1, listener_get(got, MAX_LISTENERS));
+ EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT));
+
+ threads_set(0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/nfsd/settings b/tools/testing/selftests/nfsd/settings
new file mode 100644
index 000000000000..6091b45d226b
--- /dev/null
+++ b/tools/testing/selftests/nfsd/settings
@@ -0,0 +1 @@
+timeout=120
diff --git a/tools/testing/selftests/nommu/Makefile b/tools/testing/selftests/nommu/Makefile
new file mode 100644
index 000000000000..8e7cd7315c53
--- /dev/null
+++ b/tools/testing/selftests/nommu/Makefile
@@ -0,0 +1,8 @@
+# SPDX-License-Identifier: GPL-2.0
+# Makefile for nommu selftests
+
+TEST_GEN_PROGS += nommu_mmap_test
+TEST_GEN_PROGS += nommu_mremap_test
+
+include ../lib.mk
+include local.mk
diff --git a/tools/testing/selftests/nommu/local.mk b/tools/testing/selftests/nommu/local.mk
new file mode 100644
index 000000000000..0bd1300f00f4
--- /dev/null
+++ b/tools/testing/selftests/nommu/local.mk
@@ -0,0 +1,7 @@
+# detect if users request NOMMU build or not
+# User can set NOMMU to 1 to build/test for NOMMU platforms
+NOMMU ?= 0
+ifeq ($(NOMMU),1)
+CFLAGS += -DNOMMU
+export NOMMU
+endif
diff --git a/tools/testing/selftests/nommu/nommu_mmap_test.c b/tools/testing/selftests/nommu/nommu_mmap_test.c
new file mode 100644
index 000000000000..a1f5fdda554a
--- /dev/null
+++ b/tools/testing/selftests/nommu/nommu_mmap_test.c
@@ -0,0 +1,261 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <sys/mman.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <errno.h>
+#include <string.h>
+#include <limits.h>
+#include "kselftest.h"
+
+#include <sys/vfs.h>
+#ifndef RAMFS_MAGIC
+#define RAMFS_MAGIC 0x858458f6
+#endif
+
+static size_t ps;
+
+struct test_case_t {
+ const char *name;
+ const char *pathname;
+ int open_flags;
+ int mmap_prot;
+ int mmap_flags;
+ int exp_err;
+ int (*resolve_exp_err)(const char *path);
+};
+
+static int get_shm_expected_error(const char *path)
+{
+ struct statfs fs;
+
+ if (statfs(path, &fs) == 0) {
+ if (fs.f_type == RAMFS_MAGIC)
+ return 0; /* ramfs succeed with contiguous memory */
+ }
+ /* hostfs, etc returns ENODEV due to lack of contiguous allocation */
+ return ENODEV;
+}
+
+static struct test_case_t test_cases[] = {
+ {
+ .name = "anonymous private allocation",
+ .pathname = NULL,
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ | PROT_WRITE,
+ .mmap_flags = MAP_ANONYMOUS | MAP_PRIVATE,
+ .exp_err = 0,
+ .resolve_exp_err = NULL,
+ },
+ {
+ .name = "non-anonymous private file mapping (rw-)",
+ .pathname = "/tmp/ksft.nommu-reg-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ | PROT_WRITE,
+ .mmap_flags = MAP_PRIVATE,
+ .exp_err = 0,
+ .resolve_exp_err = NULL,
+ },
+ {
+ .name = "non-anonymous private file mapping (r--)",
+ .pathname = "/tmp/ksft.nommu-reg-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ,
+ .mmap_flags = MAP_PRIVATE,
+ .exp_err = 0,
+ .resolve_exp_err = NULL,
+ },
+ {
+ .name = "non-anonymous shared file mapping (rw-)",
+ .pathname = "/tmp/ksft.nommu-shm-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ | PROT_WRITE,
+ .mmap_flags = MAP_SHARED,
+ .exp_err = 0,
+#ifdef NOMMU
+ .resolve_exp_err = get_shm_expected_error,
+#else
+ .resolve_exp_err = NULL,
+#endif
+ },
+ {
+ .name = "non-anonymous shared file mapping (r--)",
+ .pathname = "/tmp/ksft.nommu-shm-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ,
+ .mmap_flags = MAP_SHARED,
+ .exp_err = 0,
+#ifdef NOMMU
+ .resolve_exp_err = get_shm_expected_error,
+#else
+ .resolve_exp_err = 0,
+#endif
+ },
+};
+
+static int run_mapping_matrix_test(struct test_case_t *tcase)
+{
+ int fd;
+ void *ptr;
+ char path_buf[PATH_MAX];
+ const char *path = tcase->pathname;
+ int rc = KSFT_PASS;
+ int expected_error;
+
+ ksft_print_msg("[RUN] %s\n", tcase->name);
+
+ if (tcase->pathname == NULL) {
+ fd = -1;
+ } else if (strstr(tcase->pathname, "XXXXXX")) {
+ strncpy(path_buf, tcase->pathname, sizeof(path_buf) - 1);
+ path_buf[sizeof(path_buf) - 1] = '\0';
+ fd = mkstemp(path_buf);
+ if (fd < 0) {
+ ksft_print_msg("Failed to setup temp node: %s\n",
+ tcase->pathname);
+ ksft_test_result_skip("%s\n", tcase->name);
+ return KSFT_SKIP;
+ }
+ if (ftruncate(fd, ps) != 0) {
+ ksft_print_msg("ftruncate failed for: %s\n", tcase->pathname);
+ ksft_test_result_fail("%s\n", tcase->name);
+ close(fd);
+ unlink(path_buf);
+ return KSFT_FAIL;
+ }
+ path = path_buf;
+ } else {
+ fd = open(tcase->pathname, tcase->open_flags, 0600);
+ if (fd < 0) {
+ ksft_print_msg("Device node not accessible: %s\n",
+ tcase->pathname);
+ ksft_test_result_skip("%s\n", tcase->name);
+ return KSFT_SKIP;
+ }
+ }
+
+ expected_error = tcase->exp_err;
+ if (tcase->resolve_exp_err && fd >= 0)
+ expected_error = tcase->resolve_exp_err(path);
+
+ ptr = mmap(NULL, ps, tcase->mmap_prot, tcase->mmap_flags, fd, 0);
+
+ if (expected_error != 0) {
+ if (ptr != MAP_FAILED) {
+ ksft_print_msg("mmap unexpectedly succeeded (exp error %d)\n",
+ expected_error);
+ ksft_test_result_fail("%s\n", tcase->name);
+ munmap(ptr, ps);
+ rc = KSFT_FAIL;
+ goto cleanup;
+ }
+ if (errno != expected_error) {
+ ksft_print_msg("mmap failed with %d (%s), but expected %d\n",
+ errno, strerror(errno), expected_error);
+ ksft_test_result_fail("%s\n", tcase->name);
+ rc = KSFT_FAIL;
+ goto cleanup;
+ }
+ ksft_print_msg("Correctly rejected with expected error %s(%d)\n",
+ strerror(expected_error), expected_error);
+ ksft_test_result_pass("%s\n", tcase->name);
+ rc = KSFT_PASS;
+ goto cleanup;
+ }
+
+ if (ptr == MAP_FAILED) {
+ ksft_print_msg("mmap failed unexpectedly: %s\n", strerror(errno));
+ ksft_test_result_fail("%s\n", tcase->name);
+ rc = KSFT_FAIL;
+ goto cleanup;
+ }
+
+ ksft_test_result_pass("%s\n", tcase->name);
+ munmap(ptr, ps);
+
+cleanup:
+ if (fd >= 0) {
+ close(fd);
+ if (tcase->pathname && strstr(tcase->pathname, "XXXXXX"))
+ unlink(path_buf);
+ }
+ return rc;
+}
+
+static int test_map_fixed(void)
+{
+ void *fixed_addr;
+ void *ptr;
+
+ ksft_print_msg("[RUN] %s\n", __func__);
+
+ fixed_addr = mmap(NULL, ps, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (fixed_addr == MAP_FAILED) {
+ ksft_print_msg("Unable to reserve test address: %s\n",
+ strerror(errno));
+ ksft_test_result_skip("MAP_FIXED behavior\n");
+ return KSFT_SKIP;
+ }
+
+ if (munmap(fixed_addr, ps)) {
+ ksft_print_msg("Unable to release test address: %s\n",
+ strerror(errno));
+ ksft_test_result_fail("MAP_FIXED behavior\n");
+ return KSFT_FAIL;
+ }
+
+ ptr = mmap(fixed_addr, ps, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+
+#ifdef NOMMU
+ if (ptr == MAP_FAILED && (errno == ENODEV || errno == EINVAL)) {
+ ksft_print_msg("MAP_FIXED correctly rejected under nommu\n");
+ ksft_test_result_pass("MAP_FIXED behavior\n");
+ return KSFT_PASS;
+ }
+ if (ptr != MAP_FAILED) {
+ ksft_print_msg("MAP_FIXED unexpectedly allowed under nommu\n");
+ ksft_test_result_fail("MAP_FIXED behavior\n");
+ munmap(ptr, ps);
+ return KSFT_FAIL;
+ }
+ ksft_print_msg("MAP_FIXED failed under NOMMU: %s\n",
+ strerror(errno));
+ ksft_test_result_fail("MAP_FIXED behavior\n");
+ return KSFT_FAIL;
+#else
+ if (ptr != MAP_FAILED) {
+ ksft_print_msg("MAP_FIXED successfully allocated under MMU\n");
+ ksft_test_result_pass("MAP_FIXED behavior\n");
+ munmap(ptr, ps);
+ return KSFT_PASS;
+ }
+ ksft_print_msg("MAP_FIXED failed allocation under MMU\n");
+ ksft_test_result_fail("MAP_FIXED behavior\n");
+ return KSFT_FAIL;
+#endif
+}
+
+int main(int argc, char **argv)
+{
+ int i;
+
+ ps = sysconf(_SC_PAGESIZE);
+ ksft_print_header();
+ ksft_set_plan(ARRAY_SIZE(test_cases) + 1);
+
+#ifdef NOMMU
+ ksft_print_msg("Running strict MMAP test criteria under nommu architecture\n");
+#else
+ ksft_print_msg("Running MMAP test criteria under MMU architecture\n");
+#endif
+
+ test_map_fixed();
+ for (i = 0; i < (int)ARRAY_SIZE(test_cases); i++)
+ run_mapping_matrix_test(&test_cases[i]);
+
+ ksft_finished();
+}
diff --git a/tools/testing/selftests/nommu/nommu_mremap_test.c b/tools/testing/selftests/nommu/nommu_mremap_test.c
new file mode 100644
index 000000000000..7ccdf65b675f
--- /dev/null
+++ b/tools/testing/selftests/nommu/nommu_mremap_test.c
@@ -0,0 +1,366 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <sys/mman.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <errno.h>
+#include <string.h>
+#include <limits.h>
+#include "kselftest.h"
+
+#include <sys/vfs.h>
+#ifndef RAMFS_MAGIC
+#define RAMFS_MAGIC 0x858458f6
+#endif
+
+static size_t ps;
+
+static long get_fs_type(const char *path)
+{
+ struct statfs fs;
+
+ if (statfs(path, &fs) == 0)
+ return fs.f_type;
+
+ return 0;
+}
+
+static void munmap_shrink_test(void)
+{
+ void *addr;
+ int ret;
+
+ /* munmap shrink test */
+ for (int i = 0; i < 4; i++) {
+ addr = mmap(NULL, ps * 4, PROT_READ | PROT_WRITE,
+ MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
+ if (addr == MAP_FAILED) {
+ ksft_print_msg("mmap failed: %s(%d)\n", strerror(errno), errno);
+ ksft_test_result_fail("munmap shrink\n");
+ return;
+ }
+ ret = munmap((char *)addr + ps * i, ps);
+ if (ret != 0) {
+ ksft_print_msg("memory %p isn't unmapped at %p\n",
+ addr, (char *)addr + ps * i);
+ ksft_test_result_fail("munmap shrink\n");
+ return;
+ }
+
+ if (i == 0) {
+ if (munmap(addr + ps, ps * 3))
+ goto error;
+ } else if (i == 1) {
+ if (munmap(addr, ps) || munmap(addr + (ps * 2), ps * 2))
+ goto error;
+ } else if (i == 2) {
+ if (munmap(addr, ps * 2) || munmap(addr + (ps * 3), ps))
+ goto error;
+ } else if (i == 3) {
+ if (munmap(addr, ps * 3))
+ goto error;
+ }
+ }
+
+ ksft_test_result_pass("munmap shrink\n");
+ return;
+error:
+ for (int j = 0; j < 4; j++)
+ munmap((char *)addr + j * ps, ps);
+ ksft_print_msg("clean up failures\n");
+ ksft_test_result_fail("munmap shrink\n");
+}
+
+static size_t page_align(size_t len)
+{
+ return (len + ps - 1) / ps * ps;
+}
+
+static void mremap_shrink_test(void)
+{
+ void *addr, *addr2;
+ size_t current_len;
+ size_t old_len, new_len;
+ struct param {
+ size_t old;
+ size_t new;
+ } params[] = {
+ /* should not happen any shrink */
+ { .old = ps * 4 - 1, .new = ps * 4 - 2 },
+ /* should not happen any shrink */
+ { .old = ps * 4 - 1, .new = ps * 4 },
+ { .old = ps * 4, .new = ps * 2 },
+ /* should not happen any shrink */
+ { .old = ps * 2, .new = ps * 2 - 2 },
+ { .old = ps * 2 - 2, .new = ps * 1 },
+ };
+
+ /* mremap shrink test */
+ current_len = page_align(ps * 4 - 1);
+ addr = mmap(NULL, ps * 4 - 1, PROT_READ | PROT_WRITE,
+ MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
+ if (addr == MAP_FAILED) {
+ ksft_print_msg("mmap failed: %s(%d)\n", strerror(errno), errno);
+ ksft_test_result_fail("mremap shrink\n");
+ return;
+ }
+
+ for (int i = 0; i < ARRAY_SIZE(params); i++) {
+ old_len = params[i].old;
+ new_len = params[i].new;
+ current_len = page_align(new_len);
+ addr2 = mremap(addr, old_len, new_len, MREMAP_MAYMOVE);
+ if (addr2 == MAP_FAILED) {
+ ksft_print_msg("memory %p isn't remapped at %p\n", addr, addr2);
+ ksft_test_result_fail("mremap shrink\n");
+ munmap(addr, page_align(old_len));
+ return;
+ }
+
+ addr = addr2;
+ }
+
+ if (munmap(addr, current_len)) {
+ ksft_print_msg("cleanup failed: %s\n", strerror(errno));
+ ksft_test_result_fail("mremap shrink\n");
+ return;
+ }
+ ksft_test_result_pass("mremap shrink\n");
+}
+
+static int get_shared_writable_file_expected_error(const char *path)
+{
+ if (get_fs_type(path) == RAMFS_MAGIC)
+ return EPERM; /* ramfs failed */
+
+ return 0;
+}
+
+struct mremap_case_t {
+ const char *name;
+ const char *pathname;
+ int open_flags;
+ int mmap_prot;
+ int mmap_flags;
+ int exp_err;
+ int (*resolve_exp_err)(const char *path);
+ unsigned int old_pages;
+ unsigned int new_pages;
+};
+
+static struct mremap_case_t mremap_cases[] = {
+ {
+ .name = "anonymous shrink (r--)",
+ .pathname = NULL,
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ,
+ .mmap_flags = MAP_ANONYMOUS | MAP_PRIVATE,
+ .exp_err = 0,
+ .resolve_exp_err = 0,
+ },
+ {
+ .name = "shared file shrink (r--)",
+ .pathname = "/tmp/ksft.nommu-remap-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ,
+ .mmap_flags = MAP_SHARED,
+ .exp_err = 0,
+#ifdef NOMMU
+ .resolve_exp_err = get_shared_writable_file_expected_error,
+#else
+ .resolve_exp_err = 0,
+#endif
+ },
+ {
+ .name = "private file unchanged length (r-)",
+ .pathname = "/tmp/ksft.nommu-remap-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ,
+ .mmap_flags = MAP_PRIVATE,
+#ifdef NOMMU
+ .exp_err = EPERM,
+#else
+ .exp_err = 0,
+#endif
+ .resolve_exp_err = 0,
+ .old_pages = 4,
+ .new_pages = 4,
+ },
+ {
+ .name = "private file unchanged length (rw-)",
+ .pathname = "/tmp/ksft.nommu-remap-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ | PROT_WRITE,
+ .mmap_flags = MAP_PRIVATE,
+ .exp_err = 0,
+ .resolve_exp_err = 0,
+ .old_pages = 4,
+ .new_pages = 4,
+ },
+ {
+ .name = "private file growth (r-)",
+ .pathname = "/tmp/ksft.nommu-remap-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ,
+ .mmap_flags = MAP_PRIVATE,
+#ifdef NOMMU
+ .exp_err = EPERM,
+#else
+ .exp_err = 0,
+#endif
+ .resolve_exp_err = 0,
+ .old_pages = 4,
+ .new_pages = 8,
+ },
+ {
+ .name = "private file growth (rw-)",
+ .pathname = "/tmp/ksft.nommu-remap-XXXXXX",
+ .open_flags = O_CREAT | O_RDWR | O_EXCL,
+ .mmap_prot = PROT_READ | PROT_WRITE,
+ .mmap_flags = MAP_PRIVATE,
+#ifdef NOMMU
+ .exp_err = ENOMEM,
+#else
+ .exp_err = 0,
+#endif
+ .resolve_exp_err = 0,
+ .old_pages = 4,
+ .new_pages = 8,
+ },
+};
+
+static int run_mremap_test(struct mremap_case_t *tcase)
+{
+ int fd = -1;
+ void *addr, *addr2;
+ char pb[PATH_MAX];
+ const char *path = tcase->pathname;
+ int rc = KSFT_PASS;
+ int expected_error;
+ unsigned int old_pages = tcase->old_pages ?: 4;
+ unsigned int new_pages = tcase->new_pages ?: 2;
+ unsigned int file_pages = old_pages > new_pages ?
+ old_pages : new_pages;
+
+ ksft_print_msg("[RUN] Testing mremap: %s\n", tcase->name);
+
+ if (tcase->pathname && strstr(tcase->pathname, "XXXXXX")) {
+ strncpy(pb, tcase->pathname, sizeof(pb) - 1);
+ pb[sizeof(pb) - 1] = '\0';
+ fd = mkstemp(pb);
+ if (fd < 0) {
+ ksft_print_msg("Failed to setup file backing\n");
+ ksft_test_result_skip("%s\n", tcase->name);
+ return KSFT_SKIP;
+ }
+ if (ftruncate(fd, ps * file_pages) != 0) {
+ ksft_print_msg("Failed to setup file backing\n");
+ ksft_test_result_fail("%s\n", tcase->name);
+ close(fd);
+ unlink(pb);
+ return KSFT_FAIL;
+ }
+
+#ifdef NOMMU
+ if ((tcase->mmap_flags & MAP_SHARED) && get_fs_type(pb) != RAMFS_MAGIC) {
+ ksft_print_msg("Skip the test under non-ramfs filesystem (%s)\n",
+ pb);
+ ksft_test_result_skip("%s\n", tcase->name);
+ close(fd);
+ unlink(pb);
+ return KSFT_SKIP;
+ }
+#endif
+ path = pb;
+ } else if (tcase->pathname) {
+ fd = open(tcase->pathname, tcase->open_flags, 0600);
+ if (fd < 0) {
+ ksft_print_msg("Backing node not accessible\n");
+ ksft_test_result_skip("%s\n", tcase->name);
+ return KSFT_SKIP;
+ }
+
+#ifdef NOMMU
+ if ((tcase->mmap_flags & MAP_SHARED) &&
+ get_fs_type(tcase->pathname) != RAMFS_MAGIC) {
+ ksft_print_msg("Skip the test under non-ramfs filesystem (%s)\n",
+ tcase->pathname);
+ ksft_test_result_skip("%s\n", tcase->name);
+ close(fd);
+ return KSFT_SKIP;
+ }
+#endif
+ }
+
+ addr = mmap(NULL, ps * old_pages, tcase->mmap_prot,
+ tcase->mmap_flags, fd, 0);
+ if (addr == MAP_FAILED) {
+ ksft_print_msg("mmap mapping failed %s(%d)\n", strerror(errno), errno);
+ rc = KSFT_FAIL;
+ goto out;
+ }
+
+ expected_error = tcase->exp_err;
+ if (tcase->resolve_exp_err && fd >= 0)
+ expected_error = tcase->resolve_exp_err(path);
+
+ addr2 = mremap(addr, ps * old_pages, ps * new_pages,
+ MREMAP_MAYMOVE);
+
+ if (expected_error != 0) {
+ if (addr2 != MAP_FAILED) {
+ ksft_print_msg("Expected error %d, but mremap unexpectedly succeeded\n",
+ expected_error);
+ rc = KSFT_FAIL;
+ } else if (errno != expected_error) {
+ ksft_print_msg("Expected error %d, got %s(%d)\n",
+ expected_error, strerror(errno), errno);
+ rc = KSFT_FAIL;
+ } else {
+ ksft_print_msg("%s: Handled expected error path (errno=%d)\n",
+ tcase->name, expected_error);
+ }
+ } else if (addr2 == MAP_FAILED) {
+ ksft_print_msg("mremap shrink failed unexpectedly: %s\n",
+ strerror(errno));
+ rc = KSFT_FAIL;
+ } else {
+ ksft_print_msg("%s step successful\n", tcase->name);
+ }
+
+ /* clean up */
+ if (munmap(addr2 == MAP_FAILED ? addr : addr2,
+ addr2 == MAP_FAILED ? ps * old_pages : ps * new_pages)) {
+ ksft_print_msg("munmap failed: %s\n", strerror(errno));
+ rc = KSFT_FAIL;
+ }
+
+out:
+ if (fd >= 0) {
+ close(fd);
+ if (tcase->pathname && strstr(tcase->pathname, "XXXXXX"))
+ unlink(pb);
+ }
+
+ ksft_test_result_report(rc, "%s\n", tcase->name);
+ return rc;
+}
+
+int main(int argc, char **argv)
+{
+ int i;
+
+ ps = sysconf(_SC_PAGESIZE);
+ ksft_print_header();
+ ksft_set_plan(ARRAY_SIZE(mremap_cases) + 2);
+
+ munmap_shrink_test();
+ mremap_shrink_test();
+
+ for (i = 0; i < (int)ARRAY_SIZE(mremap_cases); i++)
+ run_mremap_test(&mremap_cases[i]);
+
+ ksft_finished();
+}
diff --git a/tools/testing/selftests/seccomp/seccomp_bpf.c b/tools/testing/selftests/seccomp/seccomp_bpf.c
index 891383a161d0..794336aaabb5 100644
--- a/tools/testing/selftests/seccomp/seccomp_bpf.c
+++ b/tools/testing/selftests/seccomp/seccomp_bpf.c
@@ -307,6 +307,10 @@ struct seccomp_notif_addfd_big {
#define SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV (1UL << 5)
#endif
+#ifndef SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV
+#define SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV (1UL << 6)
+#endif
+
#ifndef seccomp
int seccomp(unsigned int op, unsigned int flags, void *args)
{
@@ -4298,6 +4302,65 @@ TEST(user_notification_addfd)
close(memfd);
}
+TEST(user_notification_addfd_opath)
+{
+ struct seccomp_notif req = {};
+ struct seccomp_notif_addfd addfd = {};
+ struct seccomp_notif_resp resp = {};
+ pid_t pid;
+ int listener, pathfd, fd, status;
+
+ pathfd = open("/dev/null", O_PATH | O_CLOEXEC);
+ ASSERT_GE(pathfd, 0);
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+ listener = user_notif_syscall(__NR_getppid,
+ SECCOMP_FILTER_FLAG_NEW_LISTENER);
+ ASSERT_GE(listener, 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ int flags;
+ char byte;
+
+ close(pathfd);
+ fd = syscall(__NR_getppid);
+ if (fd < 0)
+ _exit(1);
+ flags = fcntl(fd, F_GETFL);
+ if (flags < 0 || !(flags & O_PATH))
+ _exit(2);
+ flags = fcntl(fd, F_GETFD);
+ if (flags < 0 || !(flags & FD_CLOEXEC))
+ _exit(3);
+ errno = 0;
+ if (read(fd, &byte, 1) != -1 || errno != EBADF)
+ _exit(4);
+ _exit(0);
+ }
+
+ ASSERT_EQ(ioctl(listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0);
+ addfd.id = req.id;
+ addfd.flags = SECCOMP_ADDFD_FLAG_SEND;
+ addfd.srcfd = pathfd;
+ addfd.newfd_flags = O_CLOEXEC;
+ fd = ioctl(listener, SECCOMP_IOCTL_NOTIF_ADDFD, &addfd);
+ if (fd < 0) {
+ resp.id = req.id;
+ resp.error = -errno;
+ TH_LOG("ADDFD failed with errno %d", -resp.error);
+ ASSERT_EQ(ioctl(listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0);
+ }
+
+ ASSERT_EQ(waitpid(pid, &status, 0), pid);
+ EXPECT_GE(fd, 0);
+ EXPECT_EQ(true, WIFEXITED(status));
+ if (WIFEXITED(status))
+ EXPECT_EQ(0, WEXITSTATUS(status));
+ close(pathfd);
+ close(listener);
+}
+
TEST(user_notification_addfd_rlimit)
{
pid_t pid;
@@ -4818,6 +4881,356 @@ static long get_proc_syscall(struct __test_metadata *_metadata, int pid)
return ret;
}
+
+static void notification_restart_handler(int sig)
+{
+ char c;
+ int saved_errno = errno;
+
+ if (write(handled, "s", 1) != 1 || read(handled, &c, 1) != 1)
+ _exit(1);
+ errno = saved_errno;
+}
+
+FIXTURE(notification_restart) {
+ int listener;
+ int sync[2];
+ pid_t pid;
+};
+
+FIXTURE_VARIANT(notification_restart) {
+ bool restart;
+ bool killable;
+};
+
+FIXTURE_VARIANT_ADD(notification_restart, neither) {
+ .restart = false, .killable = false,
+};
+FIXTURE_VARIANT_ADD(notification_restart, restart) {
+ .restart = true, .killable = false,
+};
+FIXTURE_VARIANT_ADD(notification_restart, killable) {
+ .restart = false, .killable = true,
+};
+FIXTURE_VARIANT_ADD(notification_restart, both) {
+ .restart = true, .killable = true,
+};
+
+FIXTURE_SETUP(notification_restart)
+{
+ unsigned int flags = SECCOMP_FILTER_FLAG_NEW_LISTENER;
+
+ self->pid = -1;
+ self->listener = -1;
+ self->sync[0] = self->sync[1] = -1;
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+ ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM, 0, self->sync), 0);
+ if (variant->restart)
+ flags |= SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV;
+ if (variant->killable)
+ flags |= SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV;
+ self->listener = user_notif_syscall(__NR_getppid, flags);
+ ASSERT_GE(self->listener, 0);
+}
+
+FIXTURE_TEARDOWN(notification_restart)
+{
+ if (self->pid > 0) {
+ kill(self->pid, SIGKILL);
+ waitpid(self->pid, NULL, 0);
+ }
+ close(self->listener);
+ close(self->sync[0]);
+ close(self->sync[1]);
+}
+
+static void notification_restart_child(struct __test_metadata *_metadata,
+ struct _test_data_notification_restart *self)
+{
+ struct sigaction action = { .sa_handler = notification_restart_handler };
+ long result[2];
+
+ self->pid = fork();
+ ASSERT_GE(self->pid, 0);
+ if (self->pid)
+ return;
+
+ close(self->listener);
+ close(self->sync[0]);
+ handled = self->sync[1];
+ if (sigemptyset(&action.sa_mask) || sigaction(SIGUSR1, &action, NULL))
+ _exit(1);
+ result[0] = syscall(__NR_getppid);
+ result[1] = errno;
+ if (write(handled, result, sizeof(result)) != sizeof(result))
+ _exit(1);
+ _exit(0);
+}
+
+static void notification_pending(struct __test_metadata *_metadata, int fd)
+{
+ struct pollfd pfd = { .fd = fd, .events = POLLIN };
+
+ ASSERT_EQ(poll(&pfd, 1, 5000), 1);
+ ASSERT_TRUE(pfd.revents & POLLIN);
+}
+
+static void notification_signal(struct __test_metadata *_metadata,
+ struct _test_data_notification_restart *self)
+{
+ struct pollfd pfd = { .fd = self->sync[0], .events = POLLIN };
+ char c;
+
+ ASSERT_EQ(kill(self->pid, SIGUSR1), 0);
+ ASSERT_EQ(poll(&pfd, 1, 5000), 1);
+ ASSERT_EQ(read(self->sync[0], &c, 1), 1);
+ ASSERT_EQ(c, 's');
+ /* The handler holds the task until the abandoned request is checked. */
+ pfd.fd = self->listener;
+ ASSERT_EQ(poll(&pfd, 1, 0), 0);
+ ASSERT_EQ(write(self->sync[0], "r", 1), 1);
+}
+
+static void notification_result(struct __test_metadata *_metadata,
+ struct _test_data_notification_restart *self,
+ long value, int error)
+{
+ long result[2];
+ int status;
+
+ ASSERT_EQ(read(self->sync[0], result, sizeof(result)), sizeof(result));
+ EXPECT_EQ(result[0], value);
+ if (value == -1)
+ EXPECT_EQ(result[1], error);
+ ASSERT_EQ(waitpid(self->pid, &status, 0), self->pid);
+ self->pid = -1;
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(WEXITSTATUS(status), 0);
+}
+
+TEST_F(notification_restart, before_receive)
+{
+ struct seccomp_notif req = {};
+ struct seccomp_notif_resp resp = {};
+ int i;
+
+ notification_restart_child(_metadata, self);
+ for (i = 0; i < 3; i++) {
+ notification_pending(_metadata, self->listener);
+ notification_signal(_metadata, self);
+ if (!variant->restart) {
+ notification_result(_metadata, self, -1, EINTR);
+ return;
+ }
+ }
+ notification_pending(_metadata, self->listener);
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0);
+ resp.id = req.id;
+ resp.flags = SECCOMP_USER_NOTIF_FLAG_CONTINUE;
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0);
+ notification_result(_metadata, self, getpid(), 0);
+}
+
+TEST_F(notification_restart, failed_receive)
+{
+ struct seccomp_notif_resp resp = {};
+ struct seccomp_notif req = {};
+ void *buf;
+
+ buf = mmap(NULL, sizeof(req), PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ ASSERT_NE(buf, MAP_FAILED);
+ notification_restart_child(_metadata, self);
+ notification_pending(_metadata, self->listener);
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, buf), -1);
+ ASSERT_EQ(errno, EFAULT);
+ ASSERT_EQ(munmap(buf, sizeof(req)), 0);
+ notification_signal(_metadata, self);
+ if (!variant->restart) {
+ notification_result(_metadata, self, -1, EINTR);
+ return;
+ }
+ notification_pending(_metadata, self->listener);
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0);
+ resp.id = req.id;
+ resp.error = -EAGAIN;
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0);
+ notification_result(_metadata, self, -1, EAGAIN);
+}
+
+TEST_F(notification_restart, after_receive)
+{
+ struct seccomp_notif req = {};
+ struct seccomp_notif_resp resp = {};
+ char c;
+
+ notification_restart_child(_metadata, self);
+ notification_pending(_metadata, self->listener);
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0);
+ if (!variant->killable) {
+ notification_signal(_metadata, self);
+ notification_result(_metadata, self, -1, EINTR);
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_ID_VALID, &req.id), -1);
+ EXPECT_EQ(errno, ENOENT);
+ return;
+ }
+ ASSERT_EQ(kill(self->pid, SIGUSR1), 0);
+ /* Either ordering of signal delivery and reply must preserve the response. */
+ resp.id = req.id;
+ resp.val = USER_NOTIF_MAGIC;
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0);
+ ASSERT_EQ(read(self->sync[0], &c, 1), 1);
+ ASSERT_EQ(c, 's');
+ ASSERT_EQ(write(self->sync[0], "r", 1), 1);
+ notification_result(_metadata, self, USER_NOTIF_MAGIC, 0);
+}
+
+TEST_F(notification_restart, fork_and_close)
+{
+ struct sigaction action = { .sa_handler = notification_restart_handler };
+ struct sock_filter filter[] = {
+ BPF_STMT(BPF_LD | BPF_W | BPF_ABS, offsetof(struct seccomp_data, nr)),
+#ifdef __NR_fork
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_fork, 0, 1),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF),
+#endif
+#ifdef __NR_clone
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_clone, 0, 1),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF),
+#endif
+#ifdef __NR_clone3
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_clone3, 0, 1),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF),
+#endif
+ BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_close, 0, 1),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF),
+ BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_ALLOW),
+ };
+ struct sock_fprog prog = { .len = ARRAY_SIZE(filter), .filter = filter };
+ char control[CMSG_SPACE(sizeof(int))] = {};
+ char c = 'f';
+ struct iovec iov = { .iov_base = &c, .iov_len = 1 };
+ struct msghdr msg = {
+ .msg_iov = &iov, .msg_iovlen = 1,
+ .msg_control = control, .msg_controllen = sizeof(control),
+ };
+ struct cmsghdr *cmsg;
+ unsigned int flags = SECCOMP_FILTER_FLAG_NEW_LISTENER;
+ int i, fd, listener, status;
+ long result[2] = {};
+ pid_t child;
+
+ if (variant->restart)
+ flags |= SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV;
+ if (variant->killable)
+ flags |= SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV;
+ ASSERT_EQ(close(self->listener), 0);
+ self->listener = -1;
+ self->pid = fork();
+ ASSERT_GE(self->pid, 0);
+ if (!self->pid) {
+ close(self->sync[0]);
+ handled = self->sync[1];
+ ASSERT_EQ(sigemptyset(&action.sa_mask), 0);
+ ASSERT_EQ(sigaction(SIGUSR1, &action, NULL), 0);
+ fd = open("/dev/null", O_RDONLY);
+ ASSERT_GE(fd, 0);
+ listener = seccomp(SECCOMP_SET_MODE_FILTER, flags, &prog);
+ ASSERT_GE(listener, 0);
+ cmsg = CMSG_FIRSTHDR(&msg);
+ cmsg->cmsg_level = SOL_SOCKET;
+ cmsg->cmsg_type = SCM_RIGHTS;
+ cmsg->cmsg_len = CMSG_LEN(sizeof(listener));
+ memcpy(CMSG_DATA(cmsg), &listener, sizeof(listener));
+ ASSERT_EQ(sendmsg(handled, &msg, 0), 1);
+
+ child = fork();
+ if (!child)
+ _exit(0);
+ if (variant->restart) {
+ ASSERT_GT(child, 0);
+ ASSERT_EQ(waitpid(child, &status, 0), child);
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(WEXITSTATUS(status), 0);
+ ASSERT_EQ(waitpid(-1, &status, WNOHANG), -1);
+ ASSERT_EQ(errno, ECHILD);
+ ASSERT_EQ(close(fd), 0);
+ ASSERT_EQ(fcntl(fd, F_GETFD), -1);
+ ASSERT_EQ(errno, EBADF);
+ } else {
+ ASSERT_EQ(child, -1);
+ ASSERT_EQ(errno, EINTR);
+ ASSERT_EQ(close(fd), -1);
+ ASSERT_EQ(errno, EINTR);
+ ASSERT_GE(fcntl(fd, F_GETFD), 0);
+ }
+
+ ASSERT_EQ(fork(), -1);
+ ASSERT_EQ(errno, EAGAIN);
+ ASSERT_EQ(write(handled, result, sizeof(result)), sizeof(result));
+ _exit(0);
+ }
+ ASSERT_EQ(recvmsg(self->sync[0], &msg, 0), 1);
+ ASSERT_FALSE(msg.msg_flags & MSG_CTRUNC);
+ cmsg = CMSG_FIRSTHDR(&msg);
+ ASSERT_NE(cmsg, NULL);
+ ASSERT_EQ(cmsg->cmsg_level, SOL_SOCKET);
+ ASSERT_EQ(cmsg->cmsg_type, SCM_RIGHTS);
+ ASSERT_EQ(cmsg->cmsg_len, CMSG_LEN(sizeof(listener)));
+ memcpy(&self->listener, CMSG_DATA(cmsg), sizeof(self->listener));
+
+ for (i = 0; i < 3; i++) {
+ struct seccomp_notif req = {};
+ struct seccomp_notif_resp resp = {};
+
+ notification_pending(_metadata, self->listener);
+ if (i < 2 || variant->restart) {
+ notification_signal(_metadata, self);
+ if (!variant->restart)
+ continue;
+ notification_pending(_metadata, self->listener);
+ }
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0);
+ EXPECT_EQ(req.pid, self->pid);
+ resp.id = req.id;
+ if (i == 2)
+ resp.error = -EAGAIN;
+ else
+ resp.flags = SECCOMP_USER_NOTIF_FLAG_CONTINUE;
+ ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0);
+ }
+ notification_result(_metadata, self, 0, 0);
+}
+
+TEST_F(notification_restart, fatal_signal)
+{
+ int status;
+
+ notification_restart_child(_metadata, self);
+ notification_pending(_metadata, self->listener);
+ ASSERT_EQ(kill(self->pid, SIGKILL), 0);
+ ASSERT_EQ(waitpid(self->pid, &status, 0), self->pid);
+ self->pid = -1;
+ ASSERT_TRUE(WIFSIGNALED(status));
+ EXPECT_EQ(WTERMSIG(status), SIGKILL);
+}
+
+TEST_F(notification_restart, listener_closed)
+{
+ notification_restart_child(_metadata, self);
+ notification_pending(_metadata, self->listener);
+ ASSERT_EQ(close(self->listener), 0);
+ self->listener = -1;
+ notification_result(_metadata, self, -1, ENOSYS);
+}
+
+TEST(user_notification_restart_requires_listener)
+{
+ ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0);
+ EXPECT_EQ(user_notif_syscall(__NR_getppid,
+ SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV), -1);
+ EXPECT_EQ(errno, EINVAL);
+}
+
/* Ensure non-fatal signals prior to receive are unmodified */
TEST(user_notification_wait_killable_pre_notification)
{
diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h
index a4e3d30b2fc1..e47be744a149 100644
--- a/tools/testing/vma/include/dup.h
+++ b/tools/testing/vma/include/dup.h
@@ -1358,23 +1358,23 @@ static inline int vfs_mmap_prepare(struct file *file, struct vm_area_desc *desc)
return file->f_op->mmap_prepare(desc);
}
-int mmap_prepare_validate(const struct vm_area_desc *prev_desc,
+int mmap_prepare_validate(const struct vm_area_desc *orig_desc,
const struct vm_area_desc *desc);
static inline int __compat_vma_mmap(struct vm_area_desc *desc,
struct vm_area_struct *vma)
{
- struct vm_area_desc prev_desc;
+ struct vm_area_desc orig_desc;
int err;
/* Derive state prior to mmap_prepare hook. */
- compat_set_desc_from_vma(&prev_desc, desc->file, vma);
+ compat_set_desc_from_vma(&orig_desc, desc->file, vma);
/* Perform any preparatory tasks for mmap action. */
err = mmap_action_prepare(desc);
if (err)
return err;
/* Check the caller did nothing crazy. */
- err = mmap_prepare_validate(&prev_desc, desc);
+ err = mmap_prepare_validate(&orig_desc, desc);
if (err)
return err;
/* Update the VMA from the descriptor. */
@@ -1657,14 +1657,14 @@ static inline bool file_is_dev_zero(const struct file *file)
return file && file->f_op == &zero_fops;
}
-static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags)
+static inline bool vma_flags_is_mm_managed(const vma_flags_t *flags)
{
- return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT);
+ return !vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT);
}
-static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma)
+static inline bool vma_is_mm_managed(const struct vm_area_struct *vma)
{
- return vma_flags_is_kernel_owned(&vma->flags);
+ return vma_flags_is_mm_managed(&vma->flags);
}
static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags)
@@ -1687,13 +1687,13 @@ static inline bool vma_flags_can_merge(const vma_flags_t *flags)
* VMA merging assumes that a VMA's flags and fields completely describe
* its state.
*
- * However, kernel-owned mappings may have established state upon mapping
- * not embodied in any attribute of the VMA.
+ * However, mappings which are not mm-managed may have established state
+ * upon mapping not embodied in any attribute of the VMA.
*
* Additionally, private (CoW) PFN maps encode the source PFN of the
* range in vma->vm_pgoff, which may otherwise cause spurious merges.
*/
- if (vma_flags_is_kernel_owned(flags))
+ if (!vma_flags_is_mm_managed(flags))
return false;
/* VMA explicitly marked as being unmergeable. */
if (vma_flags_is_fixed_mapping(flags))