diff options
Diffstat (limited to 'tools/testing')
75 files changed, 10951 insertions, 100 deletions
diff --git a/tools/testing/selftests/Makefile b/tools/testing/selftests/Makefile index 79a00e9ee46d..6e7a5c0658e3 100644 --- a/tools/testing/selftests/Makefile +++ b/tools/testing/selftests/Makefile @@ -45,6 +45,7 @@ TARGETS += filesystems/open_tree_ns TARGETS += filesystems/overlayfs TARGETS += filesystems/statmount TARGETS += filesystems/mount-notify +TARGETS += filesystems/mount_cycle TARGETS += filesystems/nsfs TARGETS += filesystems/fuse TARGETS += filesystems/move_mount @@ -97,6 +98,7 @@ TARGETS += net/packetdrill TARGETS += net/ppp TARGETS += net/rds TARGETS += net/tcp_ao +TARGETS += nfsd TARGETS += nolibc TARGETS += pci_endpoint TARGETS += pcie_bwctrl diff --git a/tools/testing/selftests/alsa/.gitignore b/tools/testing/selftests/alsa/.gitignore index 3dd8e1176b89..7b0e1e9ebf1b 100644 --- a/tools/testing/selftests/alsa/.gitignore +++ b/tools/testing/selftests/alsa/.gitignore @@ -1,3 +1,4 @@ +aloop-test global-timer mixer-test pcm-test diff --git a/tools/testing/selftests/alsa/Makefile b/tools/testing/selftests/alsa/Makefile index 8dab90ad22bb..afd64a679dc1 100644 --- a/tools/testing/selftests/alsa/Makefile +++ b/tools/testing/selftests/alsa/Makefile @@ -16,7 +16,7 @@ LDLIBS+=-lpthread OVERRIDE_TARGETS = 1 -TEST_GEN_PROGS := mixer-test pcm-test test-pcmtest-driver utimer-test +TEST_GEN_PROGS := aloop-test mixer-test pcm-test test-pcmtest-driver utimer-test TEST_GEN_PROGS_EXTENDED := libatest.so global-timer diff --git a/tools/testing/selftests/alsa/aloop-test.c b/tools/testing/selftests/alsa/aloop-test.c new file mode 100644 index 000000000000..a58b9fdbdc17 --- /dev/null +++ b/tools/testing/selftests/alsa/aloop-test.c @@ -0,0 +1,345 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Tests for the hw constraints between the two ends of an snd-aloop cable, + * with and without the "PCM Notify" control, and for the "PCM Slave" + * controls that report the playback side's parameters. + * + * Needs snd-aloop loaded with the default card id "Loopback". The tests use + * the cable between hw:Loopback,0,0 (playback) and hw:Loopback,1,0 (capture). + */ +#include <errno.h> +#include <stdbool.h> +#include <stdio.h> +#include <string.h> +#include <alsa/asoundlib.h> +#include "kselftest_harness.h" + +#define FRAMES 1024 +#define MAX_CHANNELS 4 + +struct stream_params { + snd_pcm_access_t access; + snd_pcm_format_t format; + unsigned int channels; + unsigned int rate; +}; + +/* The playback params differ from the capture params in every field. */ +static const struct stream_params capture_params = { + SND_PCM_ACCESS_RW_INTERLEAVED, SND_PCM_FORMAT_S16_LE, 2, 44100 +}; + +static const struct stream_params playback_params = { + SND_PCM_ACCESS_RW_NONINTERLEAVED, SND_PCM_FORMAT_S32_LE, 4, 96000 +}; + +struct probe_result { + unsigned int rate_min, rate_max; + unsigned int channels_min, channels_max; + bool format_ok; +}; + +/* Value events seen on the cable's controls */ +enum { + EV_ACTIVE = 1 << 0, + EV_FORMAT = 1 << 1, + EV_RATE = 1 << 2, + EV_CHANNELS = 1 << 3, + EV_ACCESS = 1 << 4, +}; + +FIXTURE(aloop) { + char play_name[32]; + char capt_name[32]; + snd_ctl_t *ctl; + bool saved_notify; + bool notify_saved; +}; + +/* The cable's controls are on the capture side device. */ +static void cable_ctl_id(snd_ctl_elem_value_t *value, const char *name) +{ + snd_ctl_elem_value_set_interface(value, SND_CTL_ELEM_IFACE_PCM); + snd_ctl_elem_value_set_name(value, name); + snd_ctl_elem_value_set_device(value, 1); + snd_ctl_elem_value_set_subdevice(value, 0); +} + +static long cable_ctl_get(snd_ctl_t *ctl, const char *name) +{ + snd_ctl_elem_value_t *value; + + snd_ctl_elem_value_alloca(&value); + cable_ctl_id(value, name); + if (snd_ctl_elem_read(ctl, value) < 0) + return -1; + if (!strcmp(name, "PCM Slave Access Mode")) + return snd_ctl_elem_value_get_enumerated(value, 0); + return snd_ctl_elem_value_get_integer(value, 0); +} + +static int set_notify(snd_ctl_t *ctl, bool on) +{ + snd_ctl_elem_value_t *value; + + snd_ctl_elem_value_alloca(&value); + cable_ctl_id(value, "PCM Notify"); + snd_ctl_elem_value_set_boolean(value, 0, on); + return snd_ctl_elem_write(ctl, value); +} + +/* Read all pending control events and return the EV_* bits for the cable's controls. */ +static unsigned int read_events(snd_ctl_t *ctl) +{ + static const struct { + const char *name; + unsigned int bit; + } names[] = { + { "PCM Slave Active", EV_ACTIVE }, + { "PCM Slave Format", EV_FORMAT }, + { "PCM Slave Rate", EV_RATE }, + { "PCM Slave Channels", EV_CHANNELS }, + { "PCM Slave Access Mode", EV_ACCESS }, + }; + snd_ctl_event_t *event; + unsigned int seen = 0; + int i; + + snd_ctl_event_alloca(&event); + while (snd_ctl_read(ctl, event) > 0) { + if (snd_ctl_event_get_type(event) != SND_CTL_EVENT_ELEM || + !(snd_ctl_event_elem_get_mask(event) & SND_CTL_EVENT_MASK_VALUE) || + snd_ctl_event_elem_get_device(event) != 1 || + snd_ctl_event_elem_get_subdevice(event) != 0) + continue; + for (i = 0; i < ARRAY_SIZE(names); i++) + if (!strcmp(snd_ctl_event_elem_get_name(event), names[i].name)) + seen |= names[i].bit; + } + return seen; +} + +/* Open and configure a stream. snd_pcm_hw_params() also prepares it. */ +static int open_pcm(snd_pcm_t **pcm, const char *name, snd_pcm_stream_t stream, + const struct stream_params *p) +{ + unsigned int buffer_time = 100000; + snd_pcm_hw_params_t *hw; + int err; + + snd_pcm_hw_params_alloca(&hw); + err = snd_pcm_open(pcm, name, stream, 0); + if (err < 0) + return err; + err = snd_pcm_hw_params_any(*pcm, hw); + if (err >= 0) + err = snd_pcm_hw_params_set_access(*pcm, hw, p->access); + if (err >= 0) + err = snd_pcm_hw_params_set_format(*pcm, hw, p->format); + if (err >= 0) + err = snd_pcm_hw_params_set_channels(*pcm, hw, p->channels); + if (err >= 0) + err = snd_pcm_hw_params_set_rate(*pcm, hw, p->rate, 0); + if (err >= 0) + err = snd_pcm_hw_params_set_buffer_time_near(*pcm, hw, &buffer_time, NULL); + if (err >= 0) + err = snd_pcm_hw_params(*pcm, hw); + if (err < 0) { + snd_pcm_close(*pcm); + *pcm = NULL; + } + return err; +} + +/* What a client probing the device sees, and whether it may use the given format */ +static int probe_pcm(const char *name, snd_pcm_stream_t stream, snd_pcm_format_t format, + struct probe_result *res) +{ + snd_pcm_hw_params_t *hw; + snd_pcm_t *pcm; + int err; + + snd_pcm_hw_params_alloca(&hw); + err = snd_pcm_open(&pcm, name, stream, 0); + if (err < 0) + return err; + err = snd_pcm_hw_params_any(pcm, hw); + if (err >= 0) + err = snd_pcm_hw_params_get_rate_min(hw, &res->rate_min, NULL); + if (err >= 0) + err = snd_pcm_hw_params_get_rate_max(hw, &res->rate_max, NULL); + if (err >= 0) + err = snd_pcm_hw_params_get_channels_min(hw, &res->channels_min); + if (err >= 0) + err = snd_pcm_hw_params_get_channels_max(hw, &res->channels_max); + if (err >= 0) + res->format_ok = !snd_pcm_hw_params_test_format(pcm, hw, format); + snd_pcm_close(pcm); + return err; +} + +static int start_playback(snd_pcm_t *pcm, const struct stream_params *p) +{ + static char silence[FRAMES * MAX_CHANNELS * 4]; + void *bufs[MAX_CHANNELS]; + snd_pcm_sframes_t written; + unsigned int i; + + if (p->access == SND_PCM_ACCESS_RW_NONINTERLEAVED) { + for (i = 0; i < p->channels; i++) + bufs[i] = silence + i * FRAMES * 4; + written = snd_pcm_writen(pcm, bufs, FRAMES); + } else { + written = snd_pcm_writei(pcm, silence, FRAMES); + } + if (written < 0) + return written; + if (snd_pcm_state(pcm) == SND_PCM_STATE_PREPARED) + return snd_pcm_start(pcm); + return 0; +} + +FIXTURE_SETUP(aloop) { + char ctl_name[32]; + snd_pcm_t *pcm; + int card, err; + + card = snd_card_get_index("Loopback"); + if (card < 0) + SKIP(return, "No Loopback card, snd-aloop is probably not loaded"); + + sprintf(ctl_name, "hw:%d", card); + sprintf(self->play_name, "hw:%d,0,0", card); + sprintf(self->capt_name, "hw:%d,1,0", card); + + err = snd_pcm_open(&pcm, self->capt_name, SND_PCM_STREAM_CAPTURE, SND_PCM_NONBLOCK); + if (err == -EBUSY) + SKIP(return, "%s is in use", self->capt_name); + ASSERT_EQ(err, 0); + snd_pcm_close(pcm); + err = snd_pcm_open(&pcm, self->play_name, SND_PCM_STREAM_PLAYBACK, SND_PCM_NONBLOCK); + if (err == -EBUSY) + SKIP(return, "%s is in use", self->play_name); + ASSERT_EQ(err, 0); + snd_pcm_close(pcm); + + ASSERT_EQ(snd_ctl_open(&self->ctl, ctl_name, SND_CTL_NONBLOCK), 0); + ASSERT_EQ(snd_ctl_subscribe_events(self->ctl, 1), 0); + self->saved_notify = cable_ctl_get(self->ctl, "PCM Notify"); + self->notify_saved = true; +} + +FIXTURE_TEARDOWN(aloop) { + if (self->notify_saved) + set_notify(self->ctl, self->saved_notify); + if (self->ctl) + snd_ctl_close(self->ctl); +} + +/* Without notify, a playback opened while a capture is set up is pinned to its parameters. */ +TEST_F(aloop, playback_constrained_without_notify) { + struct probe_result res; + snd_pcm_t *capt; + + ASSERT_EQ(set_notify(self->ctl, false), 0); + ASSERT_EQ(open_pcm(&capt, self->capt_name, SND_PCM_STREAM_CAPTURE, &capture_params), 0); + + ASSERT_EQ(probe_pcm(self->play_name, SND_PCM_STREAM_PLAYBACK, playback_params.format, + &res), 0); + EXPECT_EQ(res.rate_min, capture_params.rate); + EXPECT_EQ(res.rate_max, capture_params.rate); + EXPECT_EQ(res.channels_min, capture_params.channels); + EXPECT_EQ(res.channels_max, capture_params.channels); + EXPECT_FALSE(res.format_ok); + + snd_pcm_close(capt); +} + +/* With notify, the playback side is free to pick other parameters. */ +TEST_F(aloop, playback_unconstrained_with_notify) { + struct probe_result res; + snd_pcm_t *capt; + + ASSERT_EQ(set_notify(self->ctl, true), 0); + ASSERT_EQ(open_pcm(&capt, self->capt_name, SND_PCM_STREAM_CAPTURE, &capture_params), 0); + + ASSERT_EQ(probe_pcm(self->play_name, SND_PCM_STREAM_PLAYBACK, playback_params.format, + &res), 0); + EXPECT_LE(res.rate_min, capture_params.rate); + EXPECT_GE(res.rate_max, playback_params.rate); + EXPECT_LE(res.channels_min, capture_params.channels); + EXPECT_GE(res.channels_max, playback_params.channels); + EXPECT_TRUE(res.format_ok); + + snd_pcm_close(capt); +} + +/* + * With notify, starting a playback with different parameters stops the + * running capture, and the "PCM Slave" controls report the new parameters + * with a value event for each one that changed. + */ +TEST_F(aloop, params_change_stops_capture_with_notify) { + snd_pcm_t *capt, *play; + unsigned int events; + + /* + * The controls keep the last playback's parameters, and only notify on + * a change. Start a playback with the capture's parameters while no + * capture is open, so that every control changes below. + */ + ASSERT_EQ(open_pcm(&play, self->play_name, SND_PCM_STREAM_PLAYBACK, &capture_params), 0); + ASSERT_EQ(start_playback(play, &capture_params), 0); + snd_pcm_close(play); + + ASSERT_EQ(set_notify(self->ctl, true), 0); + ASSERT_EQ(open_pcm(&capt, self->capt_name, SND_PCM_STREAM_CAPTURE, &capture_params), 0); + ASSERT_EQ(snd_pcm_start(capt), 0); + ASSERT_EQ(snd_pcm_state(capt), SND_PCM_STATE_RUNNING); + EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Active"), 0); + read_events(self->ctl); + + ASSERT_EQ(open_pcm(&play, self->play_name, SND_PCM_STREAM_PLAYBACK, &playback_params), 0) + TH_LOG("Playback refused other parameters while the capture is running"); + ASSERT_EQ(start_playback(play, &playback_params), 0); + + /* loopback_check_format() stops the capture from the playback's start trigger. */ + EXPECT_NE(snd_pcm_state(capt), SND_PCM_STATE_RUNNING); + + EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Active"), 1); + EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Format"), playback_params.format); + EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Rate"), playback_params.rate); + EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Channels"), playback_params.channels); + EXPECT_EQ(cable_ctl_get(self->ctl, "PCM Slave Access Mode"), 1); + + events = read_events(self->ctl); + EXPECT_TRUE(events & EV_ACTIVE); + EXPECT_TRUE(events & EV_FORMAT); + EXPECT_TRUE(events & EV_RATE); + EXPECT_TRUE(events & EV_CHANNELS); + EXPECT_TRUE(events & EV_ACCESS); + + snd_pcm_close(play); + snd_pcm_close(capt); +} + +/* With notify, a capture opened second is still pinned to the playback's parameters. */ +TEST_F(aloop, capture_constrained_with_notify) { + struct probe_result res; + snd_pcm_t *play; + + ASSERT_EQ(set_notify(self->ctl, true), 0); + ASSERT_EQ(open_pcm(&play, self->play_name, SND_PCM_STREAM_PLAYBACK, &playback_params), 0); + + ASSERT_EQ(probe_pcm(self->capt_name, SND_PCM_STREAM_CAPTURE, capture_params.format, + &res), 0); + EXPECT_EQ(res.rate_min, playback_params.rate); + EXPECT_EQ(res.rate_max, playback_params.rate); + EXPECT_EQ(res.channels_min, playback_params.channels); + EXPECT_EQ(res.channels_max, playback_params.channels); + EXPECT_FALSE(res.format_ok); + + snd_pcm_close(play); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/alsa/mixer-test.c b/tools/testing/selftests/alsa/mixer-test.c index 53a72753bb08..966ef83354b9 100644 --- a/tools/testing/selftests/alsa/mixer-test.c +++ b/tools/testing/selftests/alsa/mixer-test.c @@ -28,7 +28,7 @@ #include "kselftest.h" #include "alsa-local.h" -#define TESTS_PER_CONTROL 7 +#define TESTS_PER_CONTROL 8 /* Suffixes of the SNDRV_CTL_NAME_IEC958() names, not exported to userspace */ #define IEC958_DEFAULT "Default" @@ -52,9 +52,14 @@ struct ctl_data { snd_ctl_elem_id_t *id; snd_ctl_elem_info_t *info; snd_ctl_elem_value_t *def_val; + snd_ctl_elem_value_t *snapshot; + bool snapshot_valid; + bool moved; int elem; int event_missing; int event_spurious; + int side_effects; + unsigned int ev_mask; struct card_data *card; struct ctl_data *next; }; @@ -159,6 +164,10 @@ static void find_controls(void) if (err < 0) ksft_exit_fail_msg("Out of memory\n"); + err = snd_ctl_elem_value_malloc(&ctl_data->snapshot); + if (err < 0) + ksft_exit_fail_msg("Out of memory\n"); + snd_ctl_elem_list_get_id(card_data->ctls, ctl, ctl_data->id); snd_ctl_elem_info_set_id(ctl_data->info, ctl_data->id); @@ -207,6 +216,20 @@ static void find_controls(void) snd_config_delete(config); } +/* The control on the same card that an event's numid refers to */ +static struct ctl_data *find_ctl_by_numid(struct card_data *card, + unsigned int numid) +{ + struct ctl_data *ctl; + + for (ctl = ctl_list; ctl != NULL; ctl = ctl->next) + if (ctl->card == card && + snd_ctl_elem_info_get_numid(ctl->info) == numid) + return ctl; + + return NULL; +} + /* * Block for up to timeout ms for an event, returns a negative value * on error, 0 for no event and 1 for an event. @@ -265,8 +288,21 @@ static int wait_for_event(struct ctl_data *ctl, int timeout) mask = snd_ctl_event_elem_get_mask(event); ev_id = snd_ctl_event_elem_get_numid(event); if (ev_id != snd_ctl_elem_info_get_numid(ctl->info)) { + struct ctl_data *other = find_ctl_by_numid(ctl->card, + ev_id); + + /* + * Remember that the driver announced this one. + * test_ctl_write_side_effects() uses that to tell a + * deliberate link from a silent register collision. + */ + if (other) + other->ev_mask |= mask; + ksft_print_msg("Event for unexpected ctl %s\n", snd_ctl_event_elem_get_name(event)); + /* The loop condition must not see the other control's mask */ + mask = 0; continue; } @@ -1147,6 +1183,202 @@ static void test_ctl_write_valid(struct ctl_data *ctl) ctl->card->card_name, ctl->elem); } +/* + * Build the smallest or the largest value the control offers. The smallest + * clears the control's register field and the largest sets its top bit, which + * is the bit a mask one bit too wide puts in its neighbour. + */ +static bool set_limit_value(struct ctl_data *ctl, snd_ctl_elem_value_t *val, + bool max) +{ + int i, count = snd_ctl_elem_info_get_count(ctl->info); + + snd_ctl_elem_value_set_id(val, ctl->id); + + switch (snd_ctl_elem_info_get_type(ctl->info)) { + case SND_CTL_ELEM_TYPE_BOOLEAN: + for (i = 0; i < count; i++) + snd_ctl_elem_value_set_boolean(val, i, max); + return true; + + case SND_CTL_ELEM_TYPE_INTEGER: + for (i = 0; i < count; i++) + snd_ctl_elem_value_set_integer(val, i, max ? + snd_ctl_elem_info_get_max(ctl->info) : + snd_ctl_elem_info_get_min(ctl->info)); + return true; + + case SND_CTL_ELEM_TYPE_INTEGER64: + for (i = 0; i < count; i++) + snd_ctl_elem_value_set_integer64(val, i, max ? + snd_ctl_elem_info_get_max64(ctl->info) : + snd_ctl_elem_info_get_min64(ctl->info)); + return true; + + case SND_CTL_ELEM_TYPE_ENUMERATED: + for (i = 0; i < count; i++) + snd_ctl_elem_value_set_enumerated(val, i, max ? + snd_ctl_elem_info_get_items(ctl->info) - 1 : 0); + return true; + + default: + /* Nothing sensible to write for the rest */ + return false; + } +} + +/* Note every control on the card that no longer reads as it did */ +static void find_moved_ctls(struct ctl_data *ctl, snd_ctl_elem_value_t *val) +{ + struct ctl_data *other; + int err; + + for (other = ctl_list; other != NULL; other = other->next) { + if (!other->snapshot_valid) + continue; + + /* + * The buffer is shared and compare() looks at all of it, so + * clear what the last control left in the slots this one + * does not use. + */ + snd_ctl_elem_value_clear(val); + snd_ctl_elem_value_set_id(val, other->id); + err = snd_ctl_elem_read(ctl->card->handle, val); + if (err < 0) { + ksft_print_msg("snd_ctl_elem_read() failed for %s: %s\n", + other->name, snd_strerror(err)); + continue; + } + + if (snd_ctl_elem_value_compare(other->snapshot, val)) + other->moved = true; + } +} + +/* + * Write one control and look for others on the same card that moved with it. + * A driver that links two controls on purpose tells userspace about both, so + * only an unannounced change is counted. That is what a control whose + * register mask covers bits belonging to its neighbour looks like from here. + */ +static void test_ctl_write_side_effects(struct ctl_data *ctl) +{ + struct ctl_data *other; + snd_ctl_elem_value_t *min_val, *max_val, *read_val; + int err; + + snd_ctl_elem_value_alloca(&min_val); + snd_ctl_elem_value_alloca(&max_val); + snd_ctl_elem_value_alloca(&read_val); + + /* Without a readable default there is nothing to put back */ + if (snd_ctl_elem_info_is_inactive(ctl->info) || + !snd_ctl_elem_info_is_writable(ctl->info) || + !snd_ctl_elem_info_is_readable(ctl->info) || + !set_limit_value(ctl, min_val, false) || + !set_limit_value(ctl, max_val, true)) { + ksft_test_result_skip("write_side_effects.%s.%d\n", + ctl->card->card_name, ctl->elem); + return; + } + + /* Drain first, a stale event would look like the driver announced it */ + drop_events(ctl); + + /* + * Record what the rest of the card reads as. A volatile control can + * move on its own so there is nothing to compare it against. + */ + for (other = ctl_list; other != NULL; other = other->next) { + other->snapshot_valid = false; + other->moved = false; + other->ev_mask = 0; + + if (other == ctl || other->card != ctl->card) + continue; + if (!snd_ctl_elem_info_is_readable(other->info) || + snd_ctl_elem_info_is_volatile(other->info)) + continue; + + snd_ctl_elem_value_clear(other->snapshot); + snd_ctl_elem_value_set_id(other->snapshot, other->id); + err = snd_ctl_elem_read(ctl->card->handle, other->snapshot); + if (err < 0) { + ksft_print_msg("snd_ctl_elem_read() failed for %s: %s\n", + other->name, snd_strerror(err)); + continue; + } + + other->snapshot_valid = true; + } + + /* + * Compare against the snapshot after each write, before anything is + * put back. Restoring the control we wrote goes through the same + * mask, so doing it first would hide the change we are looking for. + */ + err = snd_ctl_elem_write(ctl->card->handle, min_val); + if (err >= 0) { + drop_events(ctl); + find_moved_ctls(ctl, read_val); + err = snd_ctl_elem_write(ctl->card->handle, max_val); + } + if (err < 0) { + ksft_print_msg("snd_ctl_elem_write() failed for %s: %s\n", + ctl->name, snd_strerror(err)); + } else { + drop_events(ctl); + find_moved_ctls(ctl, read_val); + } + + for (other = ctl_list; other != NULL; other = other->next) { + if (!other->moved) + continue; + + if (other->ev_mask & SND_CTL_EVENT_MASK_VALUE) { + ksft_print_msg("Writing %s changed %s, the driver said so\n", + ctl->name, other->name); + } else { + ksft_print_msg("Writing %s silently changed %s\n", + ctl->name, other->name); + ctl->side_effects++; + } + } + + /* + * The control we wrote goes back first so its mask stops moving the + * rest. A plain write keeps this out of the event counters, they + * belong to the tests that check them. + */ + snd_ctl_elem_write(ctl->card->handle, ctl->def_val); + + for (other = ctl_list; other != NULL; other = other->next) { + if (!other->snapshot_valid || + !snd_ctl_elem_info_is_writable(other->info)) + continue; + + snd_ctl_elem_value_clear(read_val); + snd_ctl_elem_value_set_id(read_val, other->id); + if (snd_ctl_elem_read(ctl->card->handle, read_val) < 0) + continue; + + if (snd_ctl_elem_value_compare(other->snapshot, read_val)) + snd_ctl_elem_write(ctl->card->handle, other->snapshot); + } + + /* Our own restores queue events, the next test must not see them */ + drop_events(ctl); + + if (err < 0) + ksft_test_result_skip("write_side_effects.%s.%d\n", + ctl->card->card_name, ctl->elem); + else + ksft_test_result(!ctl->side_effects, + "write_side_effects.%s.%d\n", + ctl->card->card_name, ctl->elem); +} + static bool test_ctl_write_invalid_value(struct ctl_data *ctl, snd_ctl_elem_value_t *val) { @@ -1390,6 +1622,7 @@ int main(void) test_ctl_name(ctl); test_ctl_write_default(ctl); test_ctl_write_valid(ctl); + test_ctl_write_side_effects(ctl); test_ctl_write_invalid(ctl); test_ctl_event_missing(ctl); test_ctl_event_spurious(ctl); diff --git a/tools/testing/selftests/bpf/Makefile b/tools/testing/selftests/bpf/Makefile index afa589a27b15..a22be7efd1fa 100644 --- a/tools/testing/selftests/bpf/Makefile +++ b/tools/testing/selftests/bpf/Makefile @@ -422,6 +422,16 @@ $(LIBARENA_ASAN_SKEL): $(INCLUDE_DIR)/vmlinux.h $(BPFOBJ) $(LIBARENA_BPF_DEPS) +$(MAKE) -C libarena libarena_asan.skel.h $(LIBARENA_MAKE_ARGS) endif +# #![no_std] looks for compiler_builtins too. Nothing of it is used. +ifneq ($(RUST_CORE),) +$(RUST_CORE): $(RUST_CORE_SRC) + $(call msg,RUSTC,,$@) + $(Q)mkdir -p $(@D) + +$(Q)$(RUSTC_BPF) -A warnings --edition 2024 --crate-name core --out-dir $(@D) $< + +$(Q)echo '#![feature(compiler_builtins)] #![compiler_builtins] #![no_std]' | \ + $(RUSTC_BPF) -A warnings --crate-name compiler_builtins --out-dir $(@D) - +endif + # Generated test list headers define gen_tests_hdr @@ -457,7 +467,7 @@ RUNNER_PREREQS := $(INCLUDE_DIR)/vmlinux.h $(BPFOBJ) $(BPFTOOL) \ $(VERIFY_SIG_HDR) $(PRIVATE_KEY) $(VERIFICATION_CERT) \ $(LIBARENA_SKEL) $(LIBARENA_ASAN_SKEL) \ prog_tests/tests.h map_tests/tests.h \ - $(RUNNER_OBJS) + $(RUNNER_OBJS) $(RUST_CORE) # Runtime fixtures for each test_progs flavor. RUNNER_EXTRA_FILES := $(OUTPUT)/urandom_read \ diff --git a/tools/testing/selftests/bpf/Makefile.buildvars b/tools/testing/selftests/bpf/Makefile.buildvars index d2a0c0031b87..ff3476bc40a4 100644 --- a/tools/testing/selftests/bpf/Makefile.buildvars +++ b/tools/testing/selftests/bpf/Makefile.buildvars @@ -109,6 +109,31 @@ HOST_INCLUDE_DIR := $(INCLUDE_DIR) endif RESOLVE_BTFIDS := $(HOST_BUILD_DIR)/resolve_btfids/resolve_btfids +# Programs in Rust are built by upstream rustc. It has no prebuilt core for +# the bpf target, so it has to come with the source of core: +# rustup component add rust-src +# core is built as edition 2024, which it is since rustc 1.87. +# rustc emits LLVM bitcode and clang makes the object of it, so clang has to be +# 23 or newer and not older than LLVM of rustc. +# Otherwise RUST_CORE is empty and the tests are skipped. +RUSTC ?= rustc +RUST_CORE_SRC := $(wildcard $(shell $(RUSTC) --print sysroot 2>/dev/null)$\ + /lib/rustlib/src/rust/library/core/src/lib.rs) +ifneq ($(RUST_CORE_SRC),) +ifeq ($(shell { clang=$$(echo __clang_major__ | $(CLANG) -E -P -x c -) && \ + llvm=$$($(srctree)/scripts/rustc-llvm-version.sh $(RUSTC)) && \ + [ $$($(srctree)/scripts/rustc-version.sh $(RUSTC)) -ge 108700 ] && \ + [ $$clang -ge 23 ] && [ $$clang -ge $$((llvm / 10000)) ]; } \ + 2>/dev/null && echo y),y) +RUST_CORE := $(BUILD_DIR)/rust/libcore.rlib +endif +endif +# RUSTC_BOOTSTRAP=1 is to build core with a stable rustc, like the kernel does. +# panic=abort is a stop gap until panic=unwind is supported. +RUSTC_BPF = RUSTC_BOOTSTRAP=1 $(RUSTC) -O -C panic=abort --crate-type rlib \ + --target $(if $(IS_LITTLE_ENDIAN),bpfel,bpfeb)-unknown-none \ + -L $(dir $(RUST_CORE)) + DEFAULT_BPFTOOL := $(HOST_SCRATCH_DIR)/sbin/bpftool ifneq ($(CROSS_COMPILE),) CROSS_BPFTOOL := $(SCRATCH_DIR)/sbin/bpftool diff --git a/tools/testing/selftests/bpf/Makefile.skel b/tools/testing/selftests/bpf/Makefile.skel index 2e22bb901bf3..06e297a17156 100644 --- a/tools/testing/selftests/bpf/Makefile.skel +++ b/tools/testing/selftests/bpf/Makefile.skel @@ -20,7 +20,8 @@ ifneq ($(BPF_CC),) BPF_SRCS := $(notdir $(wildcard progs/*.c)) BPF_OBJS := $(patsubst %.c,$(RDIR)/%.bpf.o,$(BPF_SRCS)) -SKEL_BLACKLIST := btf__% test_pinning_invalid.c test_sk_assign.c +SKEL_BLACKLIST := btf__% test_pinning_invalid.c test_sk_assign.c \ + data_in_arena_extern.c data_in_arena_nomap.c LINKED_SKELS := test_static_linked.skel.h linked_funcs.skel.h \ linked_vars.skel.h linked_maps.skel.h linked_arena.skel.h \ @@ -137,4 +138,19 @@ $(LINKED_SKELS_H): $(RDIR)/%.skel.h: $$(addprefix $(RDIR)/,$$($$*.skel.h-deps)) $(BPFTOOL) $(GEN_SKEL) | $(RDIR) $(Q)$(cmd_bpf_link_skel) +# Programs in Rust, see Makefile.buildvars. No skeletons: the objects may be absent. +ifneq ($(RUST_CORE),) +ifeq ($(BPF_CC),$(CLANG)) +RUST_OBJS := $(patsubst progs/%.rs,$(RDIR)/%.bpf.o,$(wildcard progs/*.rs)) +BPF_OBJS += $(RUST_OBJS) + +$(RUST_OBJS): $(RDIR)/%.bpf.o: progs/%.rs $(RUST_CORE) | $(RDIR) + $(call msg,RUSTC,$(BINARY),$@) + +$(Q)$(RUSTC_BPF) --edition 2021 -C debuginfo=2 --emit=llvm-bc \ + $(patsubst -mcpu=%,-C target-cpu=%,$(filter -mcpu=%,$(BPF_CC_FLAGS))) \ + -o $(dir $(RUST_CORE))$(BINARY)-$*.bc $< && \ + $(BPF_CC) $(BPF_CC_FLAGS) -c $(dir $(RUST_CORE))$(BINARY)-$*.bc -o $@ $(call skip_on_fail,BPF) +endif +endif + endif # BPF_CC diff --git a/tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c b/tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c new file mode 100644 index 000000000000..2ac2e9a652fe --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/arena_scalar_blinded.c @@ -0,0 +1,21 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <test_progs.h> +#include "sysctl_helpers.h" +#include "verifier_arena_scalar.skel.h" + +/* The same tests with constants of the programs blinded */ +void serial_test_arena_scalar_blinded(void) +{ + const char *harden = "/proc/sys/net/core/bpf_jit_harden"; + char old[16] = {}; + + if (!is_jit_enabled()) { + test__skip(); + return; + } + if (sysctl_set_or_fail(harden, old, "2")) + return; + RUN_TESTS(verifier_arena_scalar); + sysctl_set_or_fail(harden, NULL, old); +} diff --git a/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c b/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c index 6dbd1487343c..122ecb7e98e2 100644 --- a/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c +++ b/tools/testing/selftests/bpf/prog_tests/bpf_qdisc.c @@ -11,6 +11,7 @@ #include "bpf_qdisc_fail__invalid_dynptr.skel.h" #include "bpf_qdisc_fail__invalid_dynptr_slice.skel.h" #include "bpf_qdisc_fail__invalid_dynptr_cross_frame.skel.h" +#include "bpf_qdisc_fail__invalid_dynptr_returned_slice.skel.h" #include "bpf_qdisc_fail__untrusted_write.skel.h" #include "bpf_qdisc_dynptr_use_after_invalidate_clone.skel.h" @@ -230,6 +231,7 @@ void test_ns_bpf_qdisc(void) test_incompl_ops(); RUN_TESTS(bpf_qdisc_fail__invalid_dynptr); RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_cross_frame); + RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_returned_slice); RUN_TESTS(bpf_qdisc_fail__invalid_dynptr_slice); RUN_TESTS(bpf_qdisc_fail__untrusted_write); RUN_TESTS(bpf_qdisc_dynptr_use_after_invalidate_clone); diff --git a/tools/testing/selftests/bpf/prog_tests/btf.c b/tools/testing/selftests/bpf/prog_tests/btf.c index df6ad38d287d..24ab62b2834a 100644 --- a/tools/testing/selftests/bpf/prog_tests/btf.c +++ b/tools/testing/selftests/bpf/prog_tests/btf.c @@ -424,7 +424,7 @@ static struct btf_raw_test raw_tests[] = { .err_str = "Invalid type", }, { - .descr = "global data test #8, invalid var size", + .descr = "global data test #8, var is smaller than its type", .raw_types = { /* int */ BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ @@ -457,8 +457,6 @@ static struct btf_raw_test raw_tests[] = { .key_type_id = 0, .value_type_id = 7, .max_entries = 1, - .btf_load_err = true, - .err_str = "Invalid size", }, { .descr = "global data test #9, invalid var size", @@ -498,7 +496,7 @@ static struct btf_raw_test raw_tests[] = { .err_str = "Invalid size", }, { - .descr = "global data test #10, invalid var size", + .descr = "global data test #10, section is smaller than map value", .raw_types = { /* int */ BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ @@ -531,8 +529,7 @@ static struct btf_raw_test raw_tests[] = { .key_type_id = 0, .value_type_id = 7, .max_entries = 1, - .btf_load_err = true, - .err_str = "Invalid size", + .map_create_err = true, }, { .descr = "global data test #11, multiple section members", @@ -1987,14 +1984,14 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "typedef (invalid name, invalid identifier)", + .descr = "typedef (invalid name, not printable)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPEDEF_ENC(NAME_TBD, 1), /* [2] */ BTF_END_RAW, }, - .str_sec = "\0__!int", - .str_sec_size = sizeof("\0__!int"), + .str_sec = "\0__\7int", + .str_sec_size = sizeof("\0__\7int"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "typedef_check_btf", .key_size = sizeof(int), @@ -2112,15 +2109,15 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "fwd type (invalid name, invalid identifier)", + .descr = "fwd type (invalid name, not printable)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_ENC(NAME_TBD, BTF_INFO_ENC(BTF_KIND_FWD, 0, 0), 0), /* [2] */ BTF_END_RAW, }, - .str_sec = "\0__!skb", - .str_sec_size = sizeof("\0__!skb"), + .str_sec = "\0__\7skb", + .str_sec_size = sizeof("\0__\7skb"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "fwd_type_check_btf", .key_size = sizeof(int), @@ -2175,7 +2172,7 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "struct type (invalid name, invalid identifier)", + .descr = "struct type (invalid name, not printable)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_ENC(NAME_TBD, @@ -2183,8 +2180,8 @@ static struct btf_raw_test raw_tests[] = { BTF_MEMBER_ENC(NAME_TBD, 1, 0), BTF_END_RAW, }, - .str_sec = "\0A!\0B", - .str_sec_size = sizeof("\0A!\0B"), + .str_sec = "\0A\7\0B", + .str_sec_size = sizeof("\0A\7\0B"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "struct_type_check_btf", .key_size = sizeof(int), @@ -2217,7 +2214,7 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "struct member (invalid name, invalid identifier)", + .descr = "struct member (invalid name, not printable)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_ENC(NAME_TBD, @@ -2225,8 +2222,8 @@ static struct btf_raw_test raw_tests[] = { BTF_MEMBER_ENC(NAME_TBD, 1, 0), BTF_END_RAW, }, - .str_sec = "\0A\0B*", - .str_sec_size = sizeof("\0A\0B*"), + .str_sec = "\0A\0B\7", + .str_sec_size = sizeof("\0A\0B\7"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "struct_type_check_btf", .key_size = sizeof(int), @@ -2260,7 +2257,7 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "enum type (invalid name, invalid identifier)", + .descr = "enum type (invalid name, not printable)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_ENC(NAME_TBD, @@ -2269,8 +2266,8 @@ static struct btf_raw_test raw_tests[] = { BTF_ENUM_ENC(NAME_TBD, 0), BTF_END_RAW, }, - .str_sec = "\0A!\0B", - .str_sec_size = sizeof("\0A!\0B"), + .str_sec = "\0A\7\0B", + .str_sec_size = sizeof("\0A\7\0B"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "enum_type_check_btf", .key_size = sizeof(int), @@ -2306,7 +2303,7 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "enum member (invalid name, invalid identifier)", + .descr = "enum member (invalid name, not printable)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_ENC(0, @@ -2315,8 +2312,8 @@ static struct btf_raw_test raw_tests[] = { BTF_ENUM_ENC(NAME_TBD, 0), BTF_END_RAW, }, - .str_sec = "\0A!", - .str_sec_size = sizeof("\0A!"), + .str_sec = "\0A\7", + .str_sec_size = sizeof("\0A\7"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "enum_type_check_btf", .key_size = sizeof(int), @@ -2625,14 +2622,14 @@ static struct btf_raw_test raw_tests[] = { .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_INT_ENC(0, 0, 0, 32, 4), /* [2] */ - /* void (*)(int a, unsigned int !!!) */ + /* void (*)(int a, unsigned int \7) */ BTF_FUNC_PROTO_ENC(0, 2), /* [3] */ BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1), BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 2), BTF_END_RAW, }, - .str_sec = "\0a\0!!!", - .str_sec_size = sizeof("\0a\0!!!"), + .str_sec = "\0a\0\7", + .str_sec_size = sizeof("\0a\0\7"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "func_proto_type_check_btf", .key_size = sizeof(int), @@ -2775,12 +2772,12 @@ static struct btf_raw_test raw_tests[] = { BTF_FUNC_PROTO_ENC(0, 2), /* [3] */ BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1), BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 2), - /* void !!!(int a, unsigned int b) */ + /* void \7(int a, unsigned int b) */ BTF_FUNC_ENC(NAME_TBD, 3), /* [4] */ BTF_END_RAW, }, - .str_sec = "\0a\0b\0!!!", - .str_sec_size = sizeof("\0a\0b\0!!!"), + .str_sec = "\0a\0b\0\7", + .str_sec_size = sizeof("\0a\0b\0\7"), .map_type = BPF_MAP_TYPE_ARRAY, .map_name = "func_type_check_btf", .key_size = sizeof(int), @@ -2793,7 +2790,7 @@ static struct btf_raw_test raw_tests[] = { }, { - .descr = "func (Some arg has no name)", + .descr = "func (Some arg of global func has no name)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ BTF_TYPE_INT_ENC(0, 0, 0, 32, 4), /* [2] */ @@ -2802,7 +2799,8 @@ static struct btf_raw_test raw_tests[] = { BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1), BTF_FUNC_PROTO_ARG_ENC(0, 2), /* void func(int a, unsigned int) */ - BTF_FUNC_ENC(NAME_TBD, 3), /* [4] */ + BTF_TYPE_ENC(NAME_TBD, /* [4] */ + BTF_INFO_ENC(BTF_KIND_FUNC, 0, BTF_FUNC_GLOBAL), 3), BTF_END_RAW, }, .str_sec = "\0a\0func", @@ -2819,6 +2817,57 @@ static struct btf_raw_test raw_tests[] = { }, { + .descr = "func (Some arg of static func has no name)", + .raw_types = { + /* int */ + BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ + /* unsigned int */ + BTF_TYPE_INT_ENC(0, 0, 0, 32, 4), /* [2] */ + /* void (*)(int a, unsigned int) */ + BTF_FUNC_PROTO_ENC(0, 2), /* [3] */ + BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1), + BTF_FUNC_PROTO_ARG_ENC(0, 2), + /* static void func(int a, unsigned int) */ + BTF_FUNC_ENC(NAME_TBD, 3), /* [4] */ + BTF_END_RAW, + }, + .str_sec = "\0a\0func", + .str_sec_size = sizeof("\0a\0func"), + .map_type = BPF_MAP_TYPE_ARRAY, + .map_name = "func_type_check_btf", + .key_size = sizeof(int), + .value_size = sizeof(int), + .key_type_id = 1, + .value_type_id = 1, + .max_entries = 4, +}, + +{ + .descr = "func (vararg of global func has no name)", + .raw_types = { + /* int */ + BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ + /* void (*)(int a, ...) */ + BTF_FUNC_PROTO_ENC(0, 2), /* [2] */ + BTF_FUNC_PROTO_ARG_ENC(NAME_TBD, 1), + BTF_FUNC_PROTO_ARG_ENC(0, 0), + /* void func(int a, ...) */ + BTF_TYPE_ENC(NAME_TBD, /* [3] */ + BTF_INFO_ENC(BTF_KIND_FUNC, 0, BTF_FUNC_GLOBAL), 2), + BTF_END_RAW, + }, + .str_sec = "\0a\0func", + .str_sec_size = sizeof("\0a\0func"), + .map_type = BPF_MAP_TYPE_ARRAY, + .map_name = "func_type_check_btf", + .key_size = sizeof(int), + .value_size = sizeof(int), + .key_type_id = 1, + .value_type_id = 1, + .max_entries = 4, +}, + +{ .descr = "func (Non zero vlen)", .raw_types = { BTF_TYPE_INT_ENC(0, BTF_INT_SIGNED, 0, 32, 4), /* [1] */ @@ -3585,15 +3634,28 @@ static struct btf_raw_test raw_tests[] = { .btf_load_err = true, }, { - .descr = "type name '?foo' is not ok", + .descr = "type name '?foo' is ok", .raw_types = { /* union ?foo; */ BTF_TYPE_ENC(1, BTF_INFO_ENC(BTF_KIND_FWD, 1, 0), 0), /* [1] */ BTF_END_RAW, }, BTF_STR_SEC("\0?foo"), - .err_str = "Invalid name", - .btf_load_err = true, +}, +{ + .descr = "names of Rust types and functions are ok", + .raw_types = { + BTF_TYPE_INT_ENC(NAME_NTH(1), 0, 0, 32, 4), /* [1] */ + BTF_STRUCT_ENC(NAME_NTH(2), 1, 4), /* [2] */ + BTF_MEMBER_ENC(NAME_NTH(3), 1, 0), + BTF_FWD_ENC(NAME_NTH(4), 0), /* [3] */ + BTF_TYPEDEF_ENC(NAME_NTH(5), 2), /* [4] */ + BTF_FUNC_PROTO_ENC(0, 1), /* [5] */ + BTF_FUNC_PROTO_ARG_ENC(NAME_NTH(6), 1), + BTF_FUNC_ENC(NAME_NTH(7), 5), /* [6] */ + BTF_END_RAW, + }, + BTF_STR_SEC("\0u32\0Option<&str>\0__0\0*const str\0{impl#9}<[u8; 4]>\0self\0fmt<str>"), }, { diff --git a/tools/testing/selftests/bpf/prog_tests/btf_rust.c b/tools/testing/selftests/bpf/prog_tests/btf_rust.c new file mode 100644 index 000000000000..daf777cadfda --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/btf_rust.c @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <test_progs.h> +#include <bpf/btf.h> +#include "bpftool_helpers.h" + +#define FUNC_NAME "write_fmt<scx_simple::BpfStream>" +#define KSYM_NAME "write_fmt_scx_simple__BpfStream_" + +/* The name of a function of Rust is a part of the name of the program in kallsyms */ +static void test_func_name(void) +{ + struct bpf_insn insns[] = { + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + LIBBPF_OPTS(bpf_prog_load_opts, opts); + int int_id, proto_id, prog_fd = -1, i; + struct bpf_func_info func_info = {}; + struct bpf_prog_info info = {}; + unsigned long long addr; + __u32 len = sizeof(info); + char sym[128], *p = sym; + struct btf *btf; + + btf = btf__new_empty(); + if (!ASSERT_OK_PTR(btf, "btf")) + return; + int_id = btf__add_int(btf, "i32", 4, BTF_INT_SIGNED); + ASSERT_GT(int_id, 0, "int"); + proto_id = btf__add_func_proto(btf, int_id); + ASSERT_GT(proto_id, 0, "proto"); + ASSERT_OK(btf__add_func_param(btf, "ctx", int_id), "param"); + func_info.type_id = btf__add_func(btf, FUNC_NAME, BTF_FUNC_STATIC, proto_id); + ASSERT_GT(func_info.type_id, 0, "func"); + if (!ASSERT_OK(btf__load_into_kernel(btf), "btf load")) + goto out; + + opts.prog_btf_fd = btf__fd(btf); + opts.func_info = &func_info; + opts.func_info_cnt = 1; + opts.func_info_rec_size = sizeof(func_info); + prog_fd = bpf_prog_load(BPF_PROG_TYPE_SOCKET_FILTER, NULL, "GPL", insns, + ARRAY_SIZE(insns), &opts); + if (!ASSERT_GE(prog_fd, 0, "prog load")) + goto out; + if (!ASSERT_OK(bpf_prog_get_info_by_fd(prog_fd, &info, &len), "prog info")) + goto out; + if (!info.jited_prog_len) { + test__skip(); + goto out; + } + + p += sprintf(p, "bpf_prog_"); + for (i = 0; i < BPF_TAG_SIZE; i++) + p += sprintf(p, "%02x", info.tag[i]); + sprintf(p, "_%s", KSYM_NAME); + ASSERT_OK(kallsyms_find(sym, &addr), sym); +out: + if (prog_fd >= 0) + close(prog_fd); + btf__free(btf); +} + +#define PIN_PATH "/sys/fs/bpf/btf_rust_piece" + +/* + * A piece of a static that LLVM split has the type of the whole static. + * It's the last variable in the section, so its type ends past the map value. + */ +static void test_piece(void) +{ + LIBBPF_OPTS(bpf_map_create_opts, opts); + int int_id, struct_id, var_id, piece_id, sec_id, map_fd = -1, key = 0; + __u32 value[2] = { 0x11111111, 0x22222222 }; + char line[256] = {}, out[1024] = {}; + struct btf *btf; + FILE *f = NULL; + + btf = btf__new_empty(); + if (!ASSERT_OK_PTR(btf, "btf")) + return; + int_id = btf__add_int(btf, "u32", 4, 0); + ASSERT_GT(int_id, 0, "int"); + struct_id = btf__add_struct(btf, "Whole", 8); + ASSERT_GT(struct_id, 0, "struct"); + ASSERT_OK(btf__add_field(btf, "a", int_id, 0, 0), "field"); + ASSERT_OK(btf__add_field(btf, "b", int_id, 32, 0), "field"); + var_id = btf__add_var(btf, "CNT", BTF_VAR_STATIC, int_id); + ASSERT_GT(var_id, 0, "var"); + piece_id = btf__add_var(btf, "WHOLE.1", BTF_VAR_STATIC, struct_id); + ASSERT_GT(piece_id, 0, "piece"); + sec_id = btf__add_datasec(btf, ".bss", sizeof(value)); + ASSERT_GT(sec_id, 0, "datasec"); + ASSERT_OK(btf__add_datasec_var_info(btf, var_id, 0, 4), "var info"); + ASSERT_OK(btf__add_datasec_var_info(btf, piece_id, 4, 4), "piece info"); + if (!ASSERT_OK(btf__load_into_kernel(btf), "btf load")) + goto out; + + opts.btf_fd = btf__fd(btf); + opts.btf_value_type_id = sec_id; + map_fd = bpf_map_create(BPF_MAP_TYPE_ARRAY, ".bss", sizeof(key), sizeof(value), 1, &opts); + if (!ASSERT_GE(map_fd, 0, "map create")) + goto out; + if (!ASSERT_OK(bpf_map_update_elem(map_fd, &key, value, 0), "map update")) + goto out; + + /* the variable is printed, the piece is not */ + unlink(PIN_PATH); + if (!ASSERT_OK(bpf_obj_pin(map_fd, PIN_PATH), "pin")) + goto out; + f = fopen(PIN_PATH, "r"); + if (!ASSERT_OK_PTR(f, "open")) + goto out; + while (fgets(line, sizeof(line), f) && line[0] == '#') + ; + ASSERT_HAS_SUBSTR(line, "286331153", "var"); + ASSERT_NULL(strstr(line, "572662306"), "piece"); + + /* the same for bpftool */ + if (!ASSERT_OK(get_bpftool_command_output("map dump pinned " PIN_PATH, out, sizeof(out)), + "bpftool")) + goto out; + ASSERT_HAS_SUBSTR(out, "CNT", "var"); + ASSERT_NULL(strstr(out, "WHOLE.1"), "piece"); +out: + if (f) + fclose(f); + unlink(PIN_PATH); + if (map_fd >= 0) + close(map_fd); + btf__free(btf); +} + +/* Serial: programs are not in kallsyms while another test sets bpf_jit_harden */ +void serial_test_btf_rust(void) +{ + if (test__start_subtest("func_name")) + test_func_name(); + if (test__start_subtest("piece")) + test_piece(); +} diff --git a/tools/testing/selftests/bpf/prog_tests/data_in_arena.c b/tools/testing/selftests/bpf/prog_tests/data_in_arena.c new file mode 100644 index 000000000000..cb1023507c01 --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/data_in_arena.c @@ -0,0 +1,216 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <test_progs.h> +#include "data_in_arena.skel.h" +#include "data_in_arena_decl.skel.h" +#include "data_in_arena_fail.skel.h" + +static int run_prog(struct bpf_program *prog) +{ + LIBBPF_OPTS(bpf_test_run_opts, topts); + + if (!ASSERT_OK(bpf_prog_test_run_opts(bpf_program__fd(prog), &topts), "test_run")) + return -1; + return topts.retval; +} + +static void run(struct data_in_arena *skel, int counter) +{ + int i; + + ASSERT_EQ(run_prog(skel->progs.use_data), counter + 7 + 1 + 2, "retval"); + ASSERT_EQ(skel->bss->sum, counter + 7 + 1 + 2, "sum"); + ASSERT_EQ(skel->data->counter, counter + 1, "counter"); + ASSERT_EQ(skel->data->pair[1], 5, "pair[1]"); + for (i = 0; i < 4; i++) + ASSERT_EQ(skel->bss->table[i], 10 * (i + 1) + i, "table"); +} + +static void test_in_arena(void) +{ + struct bpf_map_info info = {}; + struct data_in_arena *skel; + __u32 len = sizeof(info); + struct bpf_map *arena; + size_t sz; + + skel = data_in_arena__open(); + if (!ASSERT_OK_PTR(skel, "open")) + return; + + arena = bpf_object__find_map_by_name(skel->obj, "arena"); + if (!ASSERT_OK_PTR(arena, "arena")) + goto out; + ASSERT_EQ(bpf_map__type(arena), BPF_MAP_TYPE_ARENA, "arena type"); + ASSERT_EQ(bpf_map__max_entries(arena), 1, "arena pages"); + ASSERT_OK(bpf_map__set_max_entries(arena, 8), "arena resize"); + ASSERT_FALSE(bpf_map__autocreate(skel->maps.data), "data autocreate"); + ASSERT_FALSE(bpf_map__autocreate(skel->maps.bss), "bss autocreate"); + ASSERT_FALSE(bpf_map__autocreate(skel->maps.rodata), "rodata autocreate"); + ASSERT_EQ(bpf_map__set_autocreate(skel->maps.data, true), -EOPNOTSUPP, "set_autocreate"); + ASSERT_EQ(bpf_map__set_value_size(skel->maps.bss, 4096), -EOPNOTSUPP, "set_value_size"); + ASSERT_EQ(bpf_map__initial_value(skel->maps.data, &sz), skel->data, "initial_value"); + ASSERT_EQ(sz, sizeof(*skel->data), "initial_value size"); + + /* initial values are set the usual way */ + skel->data->counter = 100; + + if (!ASSERT_OK(data_in_arena__load(skel), "load")) + goto out; + /* there are no maps behind the sections */ + ASSERT_ERR(bpf_map_get_info_by_fd(bpf_map__fd(skel->maps.data), &info, &len), "data map"); + ASSERT_ERR(bpf_map_get_info_by_fd(bpf_map__fd(skel->maps.bss), &info, &len), "bss map"); + ASSERT_ERR(bpf_map_get_info_by_fd(bpf_map__fd(skel->maps.rodata), &info, &len), + "rodata map"); + run(skel, 100); + + /* pointers to data next to pointers to functions */ + ASSERT_EQ(run_prog(skel->progs.use_ops), 42 + 1 + 'e', "use_ops"); + ASSERT_EQ(skel->data->counter, 102, "counter"); + + /* alignment of sections and pointers to data in data */ + ASSERT_EQ((unsigned long)&skel->bss->aligned64 % 64, 0, "alignment"); + ASSERT_EQ(run_prog(skel->progs.use_ptrs), 0, "use_ptrs"); + ASSERT_EQ(skel->data->x, 43, "x"); + ASSERT_EQ(skel->bss->aligned64.v[7], 7, "aligned64"); + ASSERT_EQ(skel->data->px, &skel->data->x, "px"); + + /* format strings of bpf_printk() and BPF_SNPRINTF() */ + ASSERT_EQ(run_prog(skel->progs.use_printk), sizeof("43-7"), "use_printk"); + ASSERT_STREQ(skel->bss->out, "43-7", "out"); +out: + data_in_arena__destroy(skel); +} + +/* The object has an arena map and __arena variables */ +static void test_declared_arena(void) +{ + struct data_in_arena_decl *skel; + struct bpf_map *map; + int arenas = 0; + + skel = data_in_arena_decl__open(); + if (!ASSERT_OK_PTR(skel, "open")) + return; + bpf_object__for_each_map(map, skel->obj) + arenas += bpf_map__type(map) == BPF_MAP_TYPE_ARENA; + ASSERT_EQ(arenas, 1, "no second arena"); + skel->data->counter = 6; + if (!ASSERT_OK(data_in_arena_decl__load(skel), "load")) + goto out; + ASSERT_EQ(run_prog(skel->progs.use_data), 6 + 7 + 11, "retval"); + ASSERT_EQ(run_prog(skel->progs.use_data), 7 + 7 + 12, "retval"); + ASSERT_EQ(skel->bss->sum, 7 + 7 + 12, "sum"); + ASSERT_EQ(skel->data->counter, 8, "counter"); +out: + data_in_arena_decl__destroy(skel); +} + +/* A pointer in data that can't be made an address of arena fails the load */ +static void test_ptr_to_map(void) +{ + struct data_in_arena_fail *skel; + + skel = data_in_arena_fail__open(); + if (!ASSERT_OK_PTR(skel, "open")) + return; + ASSERT_ERR(data_in_arena_fail__load(skel), "load"); + data_in_arena_fail__destroy(skel); +} + +/* So does a pointer to a variable of the kernel. There is no skeleton: the open fails. */ +static void test_ptr_to_extern(void) +{ + struct bpf_object *obj; + + obj = bpf_object__open_file("./data_in_arena_extern.bpf.o", NULL); + if (!ASSERT_ERR_PTR(obj, "open")) + bpf_object__close(obj); +} + +/* __arena variables and no arena map. There is no skeleton: the open fails. */ +static void test_arena_var_no_map(void) +{ + struct bpf_object *obj; + + obj = bpf_object__open_file("./data_in_arena_nomap.bpf.o", NULL); + if (!ASSERT_ERR_PTR(obj, "open")) + bpf_object__close(obj); +} + +/* The address of an arena that libbpf doesn't create is not known. No pointers in data then. */ +static void test_not_my_arena(bool pin) +{ + LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = BPF_F_MMAPABLE); + struct data_in_arena *skel; + struct bpf_map *arena; + int fd = -1; + + skel = data_in_arena__open(); + if (!ASSERT_OK_PTR(skel, "open")) + return; + arena = bpf_object__find_map_by_name(skel->obj, "arena"); + if (!ASSERT_OK_PTR(arena, "arena")) + goto out; + if (pin) { + ASSERT_OK(bpf_map__set_pin_path(arena, "/sys/fs/bpf/data_in_arena"), "pin_path"); + } else { + fd = bpf_map_create(BPF_MAP_TYPE_ARENA, "arena", 0, 0, 1, &opts); + if (!ASSERT_GE(fd, 0, "map_create")) + goto out; + ASSERT_OK(bpf_map__reuse_fd(arena, fd), "reuse_fd"); + } + ASSERT_EQ(data_in_arena__load(skel), -ENOTSUP, "load"); +out: + if (fd >= 0) + close(fd); + data_in_arena__destroy(skel); +} + +/* The object is there if rustc and clang can build it, see Makefile.buildvars */ +static void test_rust(void) +{ + const char *file = "./data_in_arena_rust.bpf.o"; + struct bpf_object *obj; + + if (access(file, R_OK)) { + test__skip(); + return; + } + obj = bpf_object__open_file(file, NULL); + if (!ASSERT_OK_PTR(obj, "open")) + return; + if (!ASSERT_OK(bpf_object__load(obj), "load")) + goto out; + /* libbpf goes on without BTF when the kernel doesn't take it */ + ASSERT_GE(bpf_object__btf_fd(obj), 0, "btf_fd"); + ASSERT_EQ(run_prog(bpf_object__find_program_by_name(obj, "list_in_data")), + 100 + 20 + 3, "retval"); +out: + bpf_object__close(obj); +} + +void test_data_in_arena(void) +{ +#if !defined(__x86_64__) && !defined(__aarch64__) + /* other JITs don't take BPF_F_ARENA_SCALAR */ + test__skip(); + return; +#endif + if (test__start_subtest("arena")) + test_in_arena(); + if (test__start_subtest("declared_arena")) + test_declared_arena(); + if (test__start_subtest("ptr_to_map")) + test_ptr_to_map(); + if (test__start_subtest("ptr_to_extern")) + test_ptr_to_extern(); + if (test__start_subtest("arena_var_no_map")) + test_arena_var_no_map(); + if (test__start_subtest("pinned_arena")) + test_not_my_arena(true); + if (test__start_subtest("reused_arena")) + test_not_my_arena(false); + if (test__start_subtest("rust")) + test_rust(); +} diff --git a/tools/testing/selftests/bpf/prog_tests/verifier.c b/tools/testing/selftests/bpf/prog_tests/verifier.c index 8a6d341b754a..460ad10ddc02 100644 --- a/tools/testing/selftests/bpf/prog_tests/verifier.c +++ b/tools/testing/selftests/bpf/prog_tests/verifier.c @@ -12,6 +12,7 @@ #include "verifier_and.skel.h" #include "verifier_arena.skel.h" #include "verifier_arena_large.skel.h" +#include "verifier_arena_scalar.skel.h" #include "verifier_arena_globals1.skel.h" #include "verifier_arena_globals2.skel.h" #include "verifier_array_access.skel.h" @@ -198,6 +199,7 @@ void test_verifier_align(void) { RUN(verifier_align); } void test_verifier_and(void) { RUN(verifier_and); } void test_verifier_arena(void) { RUN(verifier_arena); } void test_verifier_arena_large(void) { RUN(verifier_arena_large); } +void test_verifier_arena_scalar(void) { RUN(verifier_arena_scalar); } void test_verifier_arena_globals1(void) { RUN(verifier_arena_globals1); } void test_verifier_arena_globals2(void) { RUN(verifier_arena_globals2); } void test_verifier_basic_stack(void) { RUN(verifier_basic_stack); } diff --git a/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c b/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c new file mode 100644 index 000000000000..8217f4c4c00c --- /dev/null +++ b/tools/testing/selftests/bpf/progs/bpf_qdisc_fail__invalid_dynptr_returned_slice.c @@ -0,0 +1,76 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <vmlinux.h> +#include "bpf_experimental.h" +#include "bpf_qdisc_common.h" +#include "bpf_misc.h" + +char _license[] SEC("license") = "GPL"; + +int proto; + +static __noinline struct ethhdr *slice_in_subprog(struct sk_buff *skb) +{ + struct bpf_dynptr ptr; + + bpf_dynptr_from_skb((struct __sk_buff *)skb, 0, &ptr); + return bpf_dynptr_slice(&ptr, 0, NULL, sizeof(struct ethhdr)); +} + +SEC("struct_ops") +__failure __msg("invalid mem access 'scalar'") +int BPF_PROG(invalid_dynptr_returned_slice, struct sk_buff *skb, + struct Qdisc *sch, struct bpf_sk_buff_ptr *to_free) +{ + struct ethhdr *hdr; + + hdr = slice_in_subprog(skb); + if (!hdr) { + bpf_qdisc_skb_drop(skb, to_free); + return NET_XMIT_DROP; + } + + /* this should fail */ + proto = hdr->h_proto; + + bpf_qdisc_skb_drop(skb, to_free); + + return NET_XMIT_DROP; +} + +SEC("struct_ops") +__auxiliary +struct sk_buff *BPF_PROG(bpf_qdisc_test_dequeue, struct Qdisc *sch) +{ + return NULL; +} + +SEC("struct_ops") +__auxiliary +int BPF_PROG(bpf_qdisc_test_init, struct Qdisc *sch, struct nlattr *opt, + struct netlink_ext_ack *extack) +{ + return 0; +} + +SEC("struct_ops") +__auxiliary +void BPF_PROG(bpf_qdisc_test_reset, struct Qdisc *sch) +{ +} + +SEC("struct_ops") +__auxiliary +void BPF_PROG(bpf_qdisc_test_destroy, struct Qdisc *sch) +{ +} + +SEC(".struct_ops") +struct Qdisc_ops test = { + .enqueue = (void *)invalid_dynptr_returned_slice, + .dequeue = (void *)bpf_qdisc_test_dequeue, + .init = (void *)bpf_qdisc_test_init, + .reset = (void *)bpf_qdisc_test_reset, + .destroy = (void *)bpf_qdisc_test_destroy, + .id = "bpf_qdisc_test", +}; diff --git a/tools/testing/selftests/bpf/progs/data_in_arena.c b/tools/testing/selftests/bpf/progs/data_in_arena.c new file mode 100644 index 000000000000..18f38103c0fc --- /dev/null +++ b/tools/testing/selftests/bpf/progs/data_in_arena.c @@ -0,0 +1,112 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <linux/bpf.h> +#include <bpf/bpf_helpers.h> + +/* global data of the object is in arena */ +char data_in_arena SEC(".arena.data"); + +int counter = 5; +long pair[2] = { 1, 2 }; +int x = 42; +long sum; +long table[4]; +struct { + long v[8]; +} aligned64 __attribute__((aligned(64))); +const volatile int ro = 7; +const volatile long ro_table[4] = { 10, 20, 30, 40 }; + +/* const strings stay in a map for helpers and kfuncs, a copy of them is in arena */ +const char hello[] SEC(".rodata.str.hello") = "hello"; + +/* pointers to data that are stored in data */ +int *px = &x; +const char *str = hello; +int *const volatile cpx SEC(".data.rel.ro") = &x; + +SEC("syscall") +int use_data(void *ctx) +{ + int i; + + for (i = 0; i < 4; i++) + table[i] = ro_table[i] + i; + sum = counter + ro + pair[0] + pair[1]; + counter++; + __sync_fetch_and_add(&pair[1], 3); + return sum; +} + +typedef int (*op_fn)(int); + +static __noinline int add1(int v) +{ + return v + 1; +} + +/* + * Pointers to functions and to data in read-only data of a program with callx. + * Volatile, so that the compiler doesn't replace the pointers with what + * they point to. + */ +static const volatile struct { + op_fn fn; + int *data; + const char *name; +} ops SEC(".data.rel.ro") = { add1, &x, hello }; + +SEC("syscall") +int use_ops(void *ctx) +{ + /* a program has the arena when its code refers to it */ + counter++; +#ifdef __clang__ + return ops.fn(*ops.data) + ops.name[1]; +#else + /* gcc doesn't support indirect calls */ + return add1(*ops.data) + ops.name[1]; +#endif +} + +SEC("syscall") +int use_ptrs(void *ctx) +{ + unsigned long addr = (unsigned long)&aligned64; + char local[4] = "abc"; + + /* hide the address from the compiler, it knows that '& 63' is 0 */ + asm volatile ("" : "+r"(addr)); + if (addr & 63) + return 1; + aligned64.v[7] = 7; + if (*px != 42) + return 2; + *px = 43; + if (x != 43 || *cpx != 43) + return 3; + if (str[0] != 'h' || str[4] != 'o' || str[5]) + return 4; + /* the literal is in a map */ + if (bpf_strncmp(local, sizeof(local), "abc")) + return 5; + return 0; +} + +char out[16]; + +/* format strings are in a map */ +SEC("syscall") +int use_printk(void *ctx) +{ + char buf[sizeof(out)]; + int i, n; + + bpf_printk("counter %d", counter); + n = BPF_SNPRINTF(buf, sizeof(buf), "%d-%d", x, ro); + for (i = 0; i < sizeof(out); i++) + out[i] = buf[i]; + return n; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_decl.c b/tools/testing/selftests/bpf/progs/data_in_arena_decl.c new file mode 100644 index 000000000000..01bc4fea8f0c --- /dev/null +++ b/tools/testing/selftests/bpf/progs/data_in_arena_decl.c @@ -0,0 +1,37 @@ +// SPDX-License-Identifier: GPL-2.0 + +#define BPF_NO_KFUNC_PROTOTYPES +#include <vmlinux.h> +#include <bpf/bpf_helpers.h> +#include "bpf_experimental.h" +#include <bpf_arena_common.h> + +struct { + __uint(type, BPF_MAP_TYPE_ARENA); + __uint(map_flags, BPF_F_MMAPABLE); + __uint(max_entries, 4); +} arena SEC(".maps"); + +/* global data of the object is in arena */ +char data_in_arena SEC(".arena.data"); + +int counter = 5; +long sum; +const volatile int ro = 7; + +#if defined(__BPF_FEATURE_ADDR_SPACE_CAST) +int __arena avar = 11; +#else +int avar = 11; +#endif + +SEC("syscall") +int use_data(void *ctx) +{ + sum = counter + ro + avar; + counter++; + avar++; + return sum; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_extern.c b/tools/testing/selftests/bpf/progs/data_in_arena_extern.c new file mode 100644 index 000000000000..dc1ad5ca3bdd --- /dev/null +++ b/tools/testing/selftests/bpf/progs/data_in_arena_extern.c @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <linux/bpf.h> +#include <bpf/bpf_helpers.h> + +/* global data of the object is in arena */ +char data_in_arena SEC(".arena.data"); + +extern const int bpf_prog_active __ksym; + +/* the variable of the kernel is not in arena */ +const void *kp = &bpf_prog_active; + +SEC("syscall") +int ptr_to_extern(void *ctx) +{ + return *(int *)kp; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_fail.c b/tools/testing/selftests/bpf/progs/data_in_arena_fail.c new file mode 100644 index 000000000000..ad8afe9afc96 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/data_in_arena_fail.c @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <linux/bpf.h> +#include <bpf/bpf_helpers.h> + +/* global data of the object is in arena */ +char data_in_arena SEC(".arena.data"); + +int x = 42; +/* the table is read-only data with pointers. It's not in arena. */ +int *const volatile tbl[1] SEC(".data.rel.ro") = { &x }; +int *const volatile *pp = tbl; + +SEC("syscall") +int ptr_to_map(void *ctx) +{ + return **pp; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_nomap.c b/tools/testing/selftests/bpf/progs/data_in_arena_nomap.c new file mode 100644 index 000000000000..ccf66321eb36 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/data_in_arena_nomap.c @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <vmlinux.h> +#include <bpf/bpf_helpers.h> +#include "bpf_arena_common.h" + +/* global data of the object is in arena */ +char data_in_arena SEC(".arena.data"); + +int counter = 5; +/* needs an arena map that is declared. The one that libbpf creates won't do. */ +int __arena_global avar = 11; + +SEC("syscall") +int arena_var(void *ctx) +{ + return avar + counter; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/data_in_arena_rust.rs b/tools/testing/selftests/bpf/progs/data_in_arena_rust.rs new file mode 100644 index 000000000000..3dd9d4208232 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/data_in_arena_rust.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: GPL-2.0 + +// Why .data, .bss and .rodata of a program in Rust are in arena. +// +// A reference in Rust is an address. It doesn't say what it points to and it +// can be stored in data. The list below has a node in each of the sections. +// The nodes are linked by references that are in the data: +// IN_BSS.next is stored by the program, +// IN_DATA.next is a relocation in .data against .rodata. +// sum() loads the references back and reads the three nodes with the same insn. +// +// When the sections are array maps libbpf skips the relocation in .data, and +// what sum() loads from a node is a number that can't be dereferenced: +// R1 invalid mem access 'scalar' +// In arena the address of a node is a number to begin with. + +#![no_std] +#![no_main] + +// Tell libbpf to keep .data, .bss and .rodata in arena. +#[used] +#[link_section = ".arena.data"] +static DATA_IN_ARENA: u8 = 0; + +#[used] +#[link_section = "license"] +static LICENSE: [u8; 4] = *b"GPL\0"; + +// panic=abort is a stop gap until panic=unwind is supported. +// Nothing here panics, so the handler is not a part of the program. +#[panic_handler] +fn panic(_info: &core::panic::PanicInfo) -> ! { + loop {} +} + +pub struct Node { + val: u32, + next: Option<&'static Node>, +} + +// no_mangle makes them visible outside, so LLVM can't fold the list into a constant. +#[no_mangle] +static IN_RODATA: Node = Node { val: 3, next: None }; +#[no_mangle] +static mut IN_DATA: Node = Node { + val: 20, + next: Some(&IN_RODATA), +}; +#[no_mangle] +static mut IN_BSS: Node = Node { val: 0, next: None }; + +#[inline(never)] +fn sum(mut node: Option<&Node>) -> u32 { + let mut sum = 0; + // The verifier wants a bound. + for _ in 0..8 { + let Some(n) = node else { break }; + sum += n.val; + node = n.next; + } + sum +} + +#[no_mangle] +#[link_section = "syscall"] +pub extern "C" fn list_in_data(_ctx: *mut u8) -> u32 { + unsafe { + let head = &mut *&raw mut IN_BSS; + head.val = 100; + head.next = Some(&*&raw const IN_DATA); + sum(Some(head)) + } +} diff --git a/tools/testing/selftests/bpf/progs/dynptr_fail.c b/tools/testing/selftests/bpf/progs/dynptr_fail.c index 148cf4417322..c2247b38c849 100644 --- a/tools/testing/selftests/bpf/progs/dynptr_fail.c +++ b/tools/testing/selftests/bpf/progs/dynptr_fail.c @@ -125,9 +125,9 @@ static int missing_release_callback_fn(__u32 index, void *data) return 0; } -/* Any dynptr initialized within a callback must have bpf_dynptr_put called */ +/* A callback cannot return with the last dynptr for a referenced resource. */ SEC("?raw_tp") -__failure __msg("Unreleased reference id") +__failure __msg("cannot overwrite referenced dynptr") int ringbuf_missing_release_callback(void *ctx) { bpf_loop(10, missing_release_callback_fn, NULL, 0); @@ -1895,6 +1895,101 @@ int clone_invalidate4(void *ctx) return 0; } +static __noinline void clone_slice_in_subprog(struct bpf_dynptr *ptr, int **data) +{ + struct bpf_dynptr clone; + + bpf_dynptr_clone(ptr, &clone); + *data = bpf_dynptr_data(&clone, 0, sizeof(val)); +} + +static __noinline void caller_slice_in_subprog(struct bpf_dynptr *ptr, int **data) +{ + struct bpf_dynptr clone; + + *data = bpf_dynptr_data(ptr, 0, sizeof(val)); + bpf_dynptr_clone(ptr, &clone); +} + +static __noinline void reserve_dynptr_in_subprog(void) +{ + struct bpf_dynptr ptr; + + bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr); +} + +/* A subprogram cannot lose the last dynptr that can release a resource. */ +SEC("?raw_tp") +__failure __msg("cannot overwrite referenced dynptr") +int referenced_dynptr_lost_on_subprog_return(void *ctx) +{ + reserve_dynptr_in_subprog(); + + return 0; +} + +/* + * Destroying a callee-local clone on return must not invalidate a slice whose + * source dynptr belongs to the caller. + */ +SEC("?raw_tp") +__success +int caller_dynptr_slice_across_subprog_valid(void *ctx) +{ + struct bpf_dynptr ptr; + int *data = NULL; + + bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr); + caller_slice_in_subprog(&ptr, &data); + if (data) + *data = 123; + bpf_ringbuf_submit_dynptr(&ptr, 0); + + return 0; +} + +/* + * A slice derived from a callee-local clone is invalid after the subprogram + * returns. + */ +SEC("?raw_tp") +__failure __msg("invalid mem access 'scalar'") +int callee_dynptr_slice_invalid_after_return(void *ctx) +{ + struct bpf_dynptr ptr; + int *data = NULL; + + bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr); + clone_slice_in_subprog(&ptr, &data); + if (data) + /* this should fail */ + *data = 123; + bpf_ringbuf_submit_dynptr(&ptr, 0); + + return 0; +} + +/* + * A slice from a caller-owned dynptr survives the subprogram return, but + * releasing the shared reservation must invalidate it. + */ +SEC("?raw_tp") +__failure __msg("invalid mem access 'scalar'") +int caller_dynptr_slice_release_after_subprog_invalid(void *ctx) +{ + struct bpf_dynptr ptr; + int *data = NULL; + + bpf_ringbuf_reserve_dynptr(&ringbuf, val, 0, &ptr); + caller_slice_in_subprog(&ptr, &data); + bpf_ringbuf_submit_dynptr(&ptr, 0); + if (data) + /* this should fail */ + *data = 123; + + return 0; +} + /* Invalidating a dynptr should invalidate any data slices * of its parent */ diff --git a/tools/testing/selftests/bpf/progs/test_siphash.h b/tools/testing/selftests/bpf/progs/test_siphash.h index 5d3a7ec36780..9a85670a9dea 100644 --- a/tools/testing/selftests/bpf/progs/test_siphash.h +++ b/tools/testing/selftests/bpf/progs/test_siphash.h @@ -22,7 +22,7 @@ static inline u64 rol64(u64 word, unsigned int shift) #define SIPHASH_CONST_2 0x6c7967656e657261ULL #define SIPHASH_CONST_3 0x7465646279746573ULL -/* lib/siphash.c */ +/* lib/crypto/siphash.c */ #define SIPROUND SIPHASH_PERMUTATION(v0, v1, v2, v3) #define PREAMBLE(len) \ diff --git a/tools/testing/selftests/bpf/progs/verifier_arena_scalar.c b/tools/testing/selftests/bpf/progs/verifier_arena_scalar.c new file mode 100644 index 000000000000..bebc37f501b7 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/verifier_arena_scalar.c @@ -0,0 +1,912 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <linux/bpf.h> +#include <bpf/bpf_helpers.h> +#include "../../../include/linux/filter.h" +#include "bpf_misc.h" + +void *bpf_arena_alloc_pages(void *map, void *addr, __u32 page_cnt, int node_id, + __u64 flags) __ksym; + +#ifdef __TARGET_ARCH_arm64 +#define ARENA_VM_START (1ull << 32) +#else +#define ARENA_VM_START (1ull << 44) +#endif + +struct { + __uint(type, BPF_MAP_TYPE_ARENA); + __uint(map_flags, BPF_F_MMAPABLE); + __uint(max_entries, 4); + __ulong(map_extra, ARENA_VM_START); +} arena SEC(".maps"); + +struct { + __uint(type, BPF_MAP_TYPE_HASH); + __uint(max_entries, 1); + __type(key, int); + __type(value, long long); +} hash SEC(".maps"); + +/* JITs that take BPF_F_ARENA_SCALAR */ +#define __arena_scalar __flag(BPF_F_ARENA_SCALAR) __arch_x86_64 __arch_arm64 + +/* BTF FUNC records are not generated for kfuncs referenced from inline assembly */ +void __kfunc_btf_root(void) +{ + bpf_arena_alloc_pages(0, 0, 0, 0, 0); +} + +/* Tests start with r6 = address of a new page as the user space sees it, a number */ + +SEC("syscall") +__arena_scalar +__description("arena_scalar: load and store of every size") +__success __retval(0) +__load_if_JITed() +__naked void ld_st_sizes(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r7 = r6; \ + r1 = 0x1122334455667788 ll; \ + *(u64 *)(r6 + 0) = r1; \ + *(u32 *)(r6 + 8) = r1; \ + *(u16 *)(r6 + 12) = r1; \ + *(u8 *)(r6 + 14) = r1; \ + *(u64 *)(r6 + 16) = 0x1234; \ + *(u32 *)(r6 + 24) = 0x5678; \ + *(u16 *)(r6 + 28) = 0x9a; \ + *(u8 *)(r6 + 30) = 0xbc; \ + r0 = 1; \ + r2 = *(u64 *)(r6 + 0); \ + if r2 != r1 goto 9f; \ + r0 = 2; \ + r2 = *(u32 *)(r6 + 8); \ + if r2 != 0x55667788 goto 9f; \ + r0 = 3; \ + r2 = *(u16 *)(r6 + 12); \ + if r2 != 0x7788 goto 9f; \ + r0 = 4; \ + r2 = *(u8 *)(r6 + 14); \ + if r2 != 0x88 goto 9f; \ + r0 = 5; \ + r2 = *(u64 *)(r6 + 16); \ + if r2 != 0x1234 goto 9f; \ + r0 = 6; \ + r2 = *(u32 *)(r6 + 24); \ + if r2 != 0x5678 goto 9f; \ + r0 = 7; \ + r2 = *(u16 *)(r6 + 28); \ + if r2 != 0x9a goto 9f; \ + r0 = 8; \ + r2 = *(u8 *)(r6 + 30); \ + if r2 != 0xbc goto 9f; \ + /* the address is what it was */ \ + r0 = 9; \ + if r6 != r7 goto 9f; \ + r0 = 10; \ + r7 >>= 32; \ + if r7 == 0 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: load into the register that holds the address") +__success __retval(0) +__load_if_JITed() +__naked void ld_into_base(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r1 = 77; \ + *(u64 *)(r6 + 0) = r1; \ + r6 = *(u64 *)(r6 + 0); \ + r0 = 1; \ + if r6 != 77 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: sign extending load") +__success __retval(0) +__load_if_JITed() +__naked void ldsx(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r7 = r6; \ + *(u64 *)(r6 + 0) = 0x80; \ + .8byte %[ldsx_insn]; /* r2 = *(s8 *)(r6 + 0) */ \ + r0 = 1; \ + if r2 != -128 goto 9f; \ + r0 = 2; \ + if r6 != r7 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(ldsx_insn, BPF_RAW_INSN(BPF_LDX | BPF_MEMSX | BPF_B, + BPF_REG_2, BPF_REG_6, 0, 0)) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: address loaded from arena") +__success __retval(0) +__load_if_JITed() +__naked void ptr_chase(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + /* page[0] = &page[64]; page[64] = 5; */ \ + r1 = r6; \ + r1 += 64; \ + *(u64 *)(r6 + 0) = r1; \ + *(u64 *)(r1 + 0) = 5; \ + r2 = *(u64 *)(r6 + 0); \ + r0 = 1; \ + if r2 != r1 goto 9f; \ + r3 = *(u64 *)(r2 + 0); \ + r0 = 2; \ + if r3 != 5 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: atomics") +__success __retval(0) +__load_if_JITed() +__naked void atomics(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r7 = r6; \ + *(u64 *)(r6 + 8) = 1; \ + r1 = 2; \ + lock *(u64 *)(r6 + 8) += r1; \ + r1 = 4; \ + .8byte %[fetch_add_insn]; /* r1 = atomic_fetch_add((u64 *)(r6 + 8), r1) */ \ + r0 = 1; \ + if r1 != 3 goto 9f; \ + r1 = 8; \ + .8byte %[xchg_insn]; /* r1 = xchg_64(r6 + 8, r1) */ \ + r0 = 2; \ + if r1 != 7 goto 9f; \ + r0 = 8; \ + r1 = 16; \ + .8byte %[cmpxchg_insn]; /* r0 = cmpxchg_64(r6 + 8, r0, r1) */ \ + r2 = r0; \ + r0 = 3; \ + if r2 != 8 goto 9f; \ + r2 = *(u64 *)(r6 + 8); \ + r0 = 4; \ + if r2 != 16 goto 9f; \ + r0 = 5; \ + if r6 != r7 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(fetch_add_insn, BPF_ATOMIC_OP(BPF_DW, BPF_ADD | BPF_FETCH, + BPF_REG_6, BPF_REG_1, 8)), + __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_6, BPF_REG_1, 8)), + __imm_insn(cmpxchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_CMPXCHG, BPF_REG_6, BPF_REG_1, 8)) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: cmpxchg through r0") +__success __retval(0) +__load_if_JITed() +__naked void cmpxchg_r0(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + /* The address is in r0. The value at the address is not equal to it. */ \ + *(u64 *)(r6 + 0) = 3; \ + r0 = r6; \ + r1 = 5; \ + .8byte %[cmpxchg_insn]; /* r0 = cmpxchg_64(r0 + 0, r0, r1) */ \ + r2 = r0; \ + r0 = 1; \ + if r2 != 3 goto 9f; \ + r2 = *(u64 *)(r6 + 0); \ + r0 = 2; \ + if r2 != 3 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(cmpxchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_CMPXCHG, BPF_REG_0, BPF_REG_1, 0)) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: xchg into the register that holds the address") +__success __retval(0) +__load_if_JITed() +__naked void xchg_into_base(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + *(u64 *)(r6 + 0) = 3; \ + r1 = r6; \ + .8byte %[xchg_insn]; /* r1 = xchg_64(r1 + 0, r1) */ \ + r0 = 1; \ + if r1 != 3 goto 9f; \ + r2 = *(u64 *)(r6 + 0); \ + r0 = 2; \ + if r2 != r6 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_1, BPF_REG_1, 0)) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: xchg into the register that holds a pointer to stack") +__failure __msg("misaligned access off (0x0; 0xffffffffffffffff)+0 size 8") +__naked void xchg_into_stack_ptr(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r1 = 0; \ + *(u64 *)(r10 - 8) = r1; \ + r1 = r10; \ + r1 += -8; \ + .8byte %[xchg_insn]; /* r1 = xchg_64(r1 + 0, r1) */ \ + r0 = 0; \ + exit; \ +" : + : __imm_addr(arena), + __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_1, BPF_REG_1, 0)) + : __clobber_all); +} + +#ifdef CAN_USE_LOAD_ACQ_STORE_REL + +SEC("syscall") +__arena_scalar +__description("arena_scalar: load-acquire and store-release") +__success __retval(0) +__load_if_JITed() +__naked void load_acq_store_rel(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r7 = r6; \ + r1 = 0x1234; \ + .8byte %[store_release_insn]; /* store_release((u64 *)(r6 + 8), r1) */ \ + .8byte %[load_acquire_insn]; /* r2 = load_acquire((u64 *)(r6 + 8)) */ \ + r0 = 1; \ + if r2 != 0x1234 goto 9f; \ + .8byte %[load_acquire8_insn]; /* w2 = load_acquire((u8 *)(r6 + 8)) */ \ + r0 = 2; \ + if r2 != 0x34 goto 9f; \ + r0 = 3; \ + if r6 != r7 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(store_release_insn, + BPF_ATOMIC_OP(BPF_DW, BPF_STORE_REL, BPF_REG_6, BPF_REG_1, 8)), + __imm_insn(load_acquire_insn, + BPF_ATOMIC_OP(BPF_DW, BPF_LOAD_ACQ, BPF_REG_2, BPF_REG_6, 8)), + __imm_insn(load_acquire8_insn, + BPF_ATOMIC_OP(BPF_B, BPF_LOAD_ACQ, BPF_REG_2, BPF_REG_6, 8)) + : __clobber_all); +} + +#endif /* CAN_USE_LOAD_ACQ_STORE_REL */ + +SEC("syscall") +__arena_scalar +__description("arena_scalar: number that is not an address in arena") +__success __retval(0) +__load_if_JITed() +__naked void not_in_arena(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + /* nothing is allocated: all loads read 0, stores are dropped */ \ + r6 = 0xdeadbeef00000000 ll; \ + r0 = 1; \ + r2 = *(u64 *)(r6 + 0); \ + if r2 != 0 goto 9f; \ + r0 = 2; \ + r2 = *(u8 *)(r6 - 32768); \ + if r2 != 0 goto 9f; \ + r6 = 0x12345678ffffffff ll; \ + r0 = 3; \ + r2 = *(u64 *)(r6 + 32760); \ + if r2 != 0 goto 9f; \ + *(u64 *)(r6 + 32760) = 1; \ + *(u8 *)(r6 + 32767) = r2; \ + r1 = 1; \ + lock *(u64 *)(r6 + 32760) += r1; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: store through the address of the stack as a number") +__success __retval(0) +__load_if_JITed() +__naked void stack_addr_as_number(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + *(u64 *)(r10 - 8) = 5; \ + r6 = r10; \ + r6 |= 0; \ + /* r6 is a number now. The store goes to arena, not to the stack. */ \ + *(u64 *)(r6 - 8) = 7; \ + r1 = 9; \ + *(u64 *)(r6 - 8) = r1; \ + lock *(u64 *)(r6 - 8) += r1; \ + r2 = *(u64 *)(r10 - 8); \ + r0 = 1; \ + if r2 != 5 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: 64-bit math on the address stays 64-bit") +__success __retval(0) +__load_if_JITed() +__naked void alu64_after_access(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r2 = *(u64 *)(r6 + 0); \ + r7 = r6; \ + r7 += 8; \ + r7 -= r6; \ + r0 = 1; \ + if r7 != 8 goto 9f; \ + r7 = r6; \ + r7 >>= 32; \ + r0 = 2; \ + if r7 == 0 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: st of an immediate changes no register") +__success __retval(0) +__load_if_JITed() +__naked void st_keeps_regs(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r0 = r6; \ + r1 = 0x1111; \ + r2 = 0x2222; \ + *(u64 *)(r0 + 0) = 5; \ + r3 = r0; \ + r0 = 1; \ + if r3 != r6 goto 9f; \ + r0 = 2; \ + if r1 != 0x1111 goto 9f; \ + r0 = 3; \ + if r2 != 0x2222 goto 9f; \ + r0 = 0x3333; \ + r1 = r6; \ + *(u32 *)(r1 + 8) = -7; \ + r3 = r0; \ + r0 = 4; \ + if r3 != 0x3333 goto 9f; \ + r0 = 5; \ + if r1 != r6 goto 9f; \ + r0 = 6; \ + r3 = *(u64 *)(r6 + 0); \ + if r3 != 5 goto 9f; \ + r0 = 7; \ + r3 = *(u64 *)(r6 + 8); \ + r4 = 0xfffffff9 ll; \ + if r3 != r4 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: access through a number with garbage in the upper half") +__success __retval(0) +__load_if_JITed() +__naked void st_value_garbage(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r1 = 0xdeadbeef00001000 ll; \ + *(u64 *)(r6 + 2048) = r1; \ + r7 = *(u64 *)(r6 + 2048); \ + r8 = *(u64 *)(r6 + 2048); \ + r1 = 3; \ + *(u64 *)(r8 + 0) = 1; \ + r0 = 1; \ + if r8 != r7 goto 9f; \ + *(u8 *)(r8 + 1) = 1; \ + r0 = 2; \ + if r8 != r7 goto 9f; \ + *(u32 *)(r8 + 4) = r1; \ + r0 = 3; \ + if r8 != r7 goto 9f; \ + r2 = *(u16 *)(r8 + 2); \ + r0 = 4; \ + if r8 != r7 goto 9f; \ + r0 = 5; \ + if r1 != 3 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: access through a number without the upper half") +__success __retval(0) +__load_if_JITed() +__naked void st_value_small(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r1 = 0x1000 ll; \ + *(u64 *)(r6 + 2048) = r1; \ + r7 = *(u64 *)(r6 + 2048); \ + r8 = *(u64 *)(r6 + 2048); \ + r1 = 3; \ + *(u64 *)(r8 + 0) = 1; \ + r0 = 1; \ + if r8 != r7 goto 9f; \ + *(u8 *)(r8 + 1) = 1; \ + r0 = 2; \ + if r8 != r7 goto 9f; \ + *(u32 *)(r8 + 4) = r1; \ + r0 = 3; \ + if r8 != r7 goto 9f; \ + r2 = *(u16 *)(r8 + 2); \ + r0 = 4; \ + if r8 != r7 goto 9f; \ + r0 = 5; \ + if r1 != 3 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: atomics through a number that is not an address in arena") +__success __retval(0) +__load_if_JITed() +__naked void atomic_value(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r1 = 0xdeadbeef00001000 ll; \ + *(u64 *)(r6 + 2048) = r1; \ + r7 = *(u64 *)(r6 + 2048); \ + r8 = *(u64 *)(r6 + 2048); \ + r1 = 3; \ + lock *(u64 *)(r8 + 0) += r1; \ + r0 = 1; \ + if r8 != r7 goto 9f; \ + .8byte %[fetch_add_insn]; \ + r0 = 2; \ + if r8 != r7 goto 9f; \ + .8byte %[xchg_insn]; \ + r0 = 3; \ + if r8 != r7 goto 9f; \ + r0 = 0; \ + .8byte %[cmpxchg_insn]; \ + r0 = 4; \ + if r8 != r7 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(fetch_add_insn, BPF_ATOMIC_OP(BPF_DW, BPF_ADD | BPF_FETCH, + BPF_REG_8, BPF_REG_1, 0)), + __imm_insn(xchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_XCHG, BPF_REG_8, BPF_REG_1, 0)), + __imm_insn(cmpxchg_insn, BPF_ATOMIC_OP(BPF_DW, BPF_CMPXCHG, BPF_REG_8, BPF_REG_1, 0)) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: number and pointer to arena at the same insn") +__success __retval(0) +__load_if_JITed() +__naked void mixed_number_arena(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + *(u64 *)(r6 + 8) = 0x1234; \ + /* a number that the verifier does not know, 0 at run time */ \ + r7 = *(u64 *)(r6 + 16); \ + r8 = r6; \ + if r7 != 0 goto 1f; \ + .8byte %[cast_kern_insn]; \ +1: *(u8 *)(r8 + 0) = 1; \ + *(u8 *)(r8 + 1) = r7; \ + r2 = *(u8 *)(r8 + 1); \ + lock *(u64 *)(r8 + 24) += r7; \ + r0 = 0; \ + if r7 != 0 goto 9f; \ + /* pointer to arena only: JIT adds all 64 bits of r8 to the base */ \ + r0 = 2; \ + r2 = *(u64 *)(r8 + 8); \ + if r2 != 0x1234 goto 9f; \ + r0 = 3; \ + r2 = *(u8 *)(r6 + 0); \ + if r2 != 1 goto 9f; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm_insn(cast_kern_insn, BPF_RAW_INSN(BPF_ALU64 | BPF_MOV | BPF_X, + BPF_REG_8, BPF_REG_8, 1, 1)) + : __clobber_all); +} + +SEC("syscall") +__description("arena_scalar: no flag, no access through a number") +__failure __msg("R6 invalid mem access 'scalar'") +__naked void no_flag(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r6 = 0x100000000000 ll; \ + r0 = *(u64 *)(r6 + 0); \ + exit; \ +" : + : __imm_addr(arena) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: no arena, no access through a number") +__failure __msg("R6 invalid mem access 'scalar'") +__naked void no_arena(void) +{ + asm volatile (" \ + r6 = 0x100000000000 ll; \ + r0 = *(u64 *)(r6 + 0); \ + exit; \ +" ::: __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: pointer that is NULL is an address in arena") +__success __retval(0) +__load_if_JITed() +__naked void null_ptr(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r1 = 0; \ + *(u32 *)(r10 - 4) = r1; \ + r2 = r10; \ + r2 += -4; \ + r1 = %[hash] ll; \ + call %[bpf_map_lookup_elem]; \ + r1 = r0; \ + r0 = 1; \ + if r1 != 0 goto 9f; \ + /* nothing is allocated: the load reads 0 */ \ + r0 = *(u64 *)(r1 + 0); \ +9: exit; \ +" : + : __imm_addr(arena), + __imm_addr(hash), + __imm(bpf_map_lookup_elem) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: pointer that is NULL with an offset is an address in arena") +__success __retval(0) +__load_if_JITed() +__naked void null_ptr_off(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r1 = 0; \ + *(u32 *)(r10 - 4) = r1; \ + r2 = r10; \ + r2 += -4; \ + r1 = %[hash] ll; \ + call %[bpf_map_lookup_elem]; \ + r1 = r0; \ + r0 = 1; \ + if r1 != 0 goto 9f; \ + r1 += 8; \ + *(u64 *)(r1 + 0) = 5; \ + r0 = *(u64 *)(r1 + 0); \ +9: exit; \ +" : + : __imm_addr(arena), + __imm_addr(hash), + __imm(bpf_map_lookup_elem) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: number that is less than a page is an address in arena") +__success __retval(0) +__load_if_JITed() +__naked void small_number(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + call %[bpf_get_prandom_u32]; \ + r1 = r0; \ + r1 &= 0xfff; \ + r0 = *(u64 *)(r1 + 0); \ + exit; \ +" : + : __imm_addr(arena), + __imm(bpf_get_prandom_u32) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: number is not a pointer for a helper") +__failure __msg("R2 type=scalar expected=") +__naked void helper_arg(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + r1 = %[hash] ll; \ + r2 = r6; \ + call %[bpf_map_lookup_elem]; \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm_addr(hash), + __imm(bpf_arena_alloc_pages), + __imm(bpf_map_lookup_elem) + : __clobber_all); +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: number and pointer to stack at the same insn") +__failure __msg("same insn cannot be used with different pointers") +__load_if_JITed() +__naked void mixed_number_stack(void) +{ + asm volatile (" \ + r1 = %[arena] ll; \ + r2 = 0; \ + r3 = 1; \ + r4 = -1; \ + r5 = 0; \ + call %[bpf_arena_alloc_pages]; \ + r6 = r0; \ + r0 = 100; \ + if r6 == 0 goto 9f; \ + *(u64 *)(r10 - 8) = 0; \ + call %[bpf_get_prandom_u32]; \ + if w0 != 0 goto 1f; \ + r6 = r10; \ + r6 += -8; \ +1: r0 = *(u64 *)(r6 + 0); \ + r0 = 0; \ +9: exit; \ +" : + : __imm_addr(arena), + __imm(bpf_arena_alloc_pages), + __imm(bpf_get_prandom_u32) + : __clobber_all); +} + +static int st_cb(__u64 idx, void *ctx) +{ + volatile long *p = *(volatile long **)ctx; + + p[idx] = 7; + return 0; +} + +SEC("syscall") +__arena_scalar +__description("arena_scalar: store through a number in a callback") +__success __retval(0) +__load_if_JITed() +int st_in_callback(void *unused) +{ + volatile long *p = bpf_arena_alloc_pages(&arena, NULL, 1, -1, 0); + + if (!p) + return 100; + bpf_loop(4, st_cb, &p, 0); + return p[0] + p[3] + p[4] - 14; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c b/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c index 4d8273c258d5..31663338866d 100644 --- a/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c +++ b/tools/testing/selftests/bpf/progs/verifier_value_illegal_alu.c @@ -22,8 +22,8 @@ struct { SEC("socket") __description("map element value illegal alu op, 1") -__failure __msg("R0 bitwise operator &= on pointer") -__failure_unpriv +__failure __msg("R0 invalid mem access 'scalar'") +__failure_unpriv __msg_unpriv("R0 bitwise operator &= on pointer") __naked void value_illegal_alu_op_1(void) { asm volatile (" \ @@ -70,8 +70,8 @@ l0_%=: exit; \ SEC("socket") __description("map element value illegal alu op, 3") -__failure __msg("R0 pointer arithmetic with /= operator") -__failure_unpriv +__failure __msg("R0 invalid mem access 'scalar'") +__failure_unpriv __msg_unpriv("R0 pointer arithmetic with /= operator") __naked void value_illegal_alu_op_3(void) { asm volatile (" \ @@ -165,6 +165,171 @@ __naked void map_ptr_illegal_alu_op(void) : __clobber_all); } +SEC("socket") +__description("tag in the low bit of a pointer, and, shift") +__success __retval(0) +__failure_unpriv __msg_unpriv("R1 bitwise operator &= on pointer") +__naked void ptr_tag_and_shift(void) +{ + asm volatile (" \ + r2 = r10; \ + r2 += -8; \ + r1 = 0; \ + *(u64*)(r2 + 0) = r1; \ + r1 = %[map_hash_48b] ll; \ + call %[bpf_map_lookup_elem]; \ + if r0 == 0 goto l0_%=; \ + r1 = r0; \ + r1 &= 1; \ + r2 = r0; \ + r2 >>= 1; \ + r3 = r0; \ + r3 |= 1; \ + r3 ^= 1; \ + r0 = *(u32*)(r0 + 0); \ + r0 = 0; \ +l0_%=: exit; \ +" : + : __imm(bpf_map_lookup_elem), + __imm_addr(map_hash_48b) + : __clobber_all); +} + +SEC("socket") +__description("tag in the low bit of a pointer, CAP_BPF without CAP_PERFMON") +__success __retval(0) +__failure_unpriv __msg_unpriv("R1 bitwise operator &= on pointer") +__caps_unpriv(CAP_BPF) +__naked void ptr_tag_cap_bpf(void) +{ + asm volatile (" \ + r2 = r10; \ + r2 += -8; \ + r1 = 0; \ + *(u64*)(r2 + 0) = r1; \ + r1 = %[map_hash_48b] ll; \ + call %[bpf_map_lookup_elem]; \ + if r0 == 0 goto l0_%=; \ + r1 = r0; \ + r1 &= 1; \ + r0 = 0; \ +l0_%=: exit; \ +" : + : __imm(bpf_map_lookup_elem), + __imm_addr(map_hash_48b) + : __clobber_all); +} + +SEC("socket") +__description("number op= pointer") +__success __retval(0) +__failure_unpriv __msg_unpriv("R1 pointer arithmetic with *= operator") +__naked void number_mul_ptr(void) +{ + asm volatile (" \ + r2 = r10; \ + r2 += -8; \ + r1 = 0; \ + *(u64*)(r2 + 0) = r1; \ + r1 = %[map_hash_48b] ll; \ + call %[bpf_map_lookup_elem]; \ + if r0 == 0 goto l0_%=; \ + r1 = 7; \ + r1 *= r0; \ + r0 = 0; \ +l0_%=: exit; \ +" : + : __imm(bpf_map_lookup_elem), + __imm_addr(map_hash_48b) + : __clobber_all); +} + +SEC("socket") +__description("pointer with the tag cleared is a number") +__failure __msg("R0 invalid mem access 'scalar'") +__failure_unpriv __msg_unpriv("R0 bitwise operator |= on pointer") +__naked void ptr_tag_cleared_deref(void) +{ + asm volatile (" \ + r2 = r10; \ + r2 += -8; \ + r1 = 0; \ + *(u64*)(r2 + 0) = r1; \ + r1 = %[map_hash_48b] ll; \ + call %[bpf_map_lookup_elem]; \ + if r0 == 0 goto l0_%=; \ + r0 |= 1; \ + r0 ^= 1; \ + r0 = *(u32*)(r0 + 0); \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm(bpf_map_lookup_elem), + __imm_addr(map_hash_48b) + : __clobber_all); +} + +SEC("socket") +__description("shift of a pointer that may be NULL") +__failure __msg("R0 pointer arithmetic on map_value_or_null prohibited, null-check it first") +__failure_unpriv +__naked void ptr_or_null_shift(void) +{ + asm volatile (" \ + r2 = r10; \ + r2 += -8; \ + r1 = 0; \ + *(u64*)(r2 + 0) = r1; \ + r1 = %[map_hash_48b] ll; \ + call %[bpf_map_lookup_elem]; \ + r0 >>= 1; \ + r0 = 0; \ + exit; \ +" : + : __imm(bpf_map_lookup_elem), + __imm_addr(map_hash_48b) + : __clobber_all); +} + +SEC("socket") +__description("and of a pointer to map") +__failure __msg("R0 pointer arithmetic on map_ptr prohibited") +__failure_unpriv +__naked void map_ptr_and(void) +{ + asm volatile (" \ + r0 = %[map_hash_48b] ll; \ + r0 &= 1; \ + r0 = 0; \ + exit; \ +" : + : __imm_addr(map_hash_48b) + : __clobber_all); +} + +SEC("socket") +__description("32-bit and of a pointer") +__success __retval(0) +__failure_unpriv __msg_unpriv("R0 32-bit pointer arithmetic prohibited") +__naked void ptr_and32(void) +{ + asm volatile (" \ + r2 = r10; \ + r2 += -8; \ + r1 = 0; \ + *(u64*)(r2 + 0) = r1; \ + r1 = %[map_hash_48b] ll; \ + call %[bpf_map_lookup_elem]; \ + if r0 == 0 goto l0_%=; \ + w0 &= 1; \ +l0_%=: r0 = 0; \ + exit; \ +" : + : __imm(bpf_map_lookup_elem), + __imm_addr(map_hash_48b) + : __clobber_all); +} + SEC("flow_dissector") __description("flow_keys illegal alu op with variable offset") __failure __msg("R7 pointer arithmetic on flow_keys prohibited") diff --git a/tools/testing/selftests/bpf/test_loader.c b/tools/testing/selftests/bpf/test_loader.c index 25eeb1c1248b..a89890cd56d8 100644 --- a/tools/testing/selftests/bpf/test_loader.c +++ b/tools/testing/selftests/bpf/test_loader.c @@ -580,6 +580,8 @@ static int parse_test_spec(struct test_loader *tester, update_flags(&spec->prog_flags, BPF_F_XDP_HAS_FRAGS, clear); } else if (strcmp(val, "BPF_F_TEST_REG_INVARIANTS") == 0) { update_flags(&spec->prog_flags, BPF_F_TEST_REG_INVARIANTS, clear); + } else if (strcmp(val, "BPF_F_ARENA_SCALAR") == 0) { + update_flags(&spec->prog_flags, BPF_F_ARENA_SCALAR, clear); } else /* assume numeric value */ { err = parse_int(val, &flags, "test prog flags"); if (err) diff --git a/tools/testing/selftests/drivers/net/psp.py b/tools/testing/selftests/drivers/net/psp.py index 5a81f40cac7d..473500901879 100755 --- a/tools/testing/selftests/drivers/net/psp.py +++ b/tools/testing/selftests/drivers/net/psp.py @@ -11,6 +11,8 @@ import struct import termios import time +from contextlib import contextmanager + from lib.py import defer from lib.py import ksft_run, ksft_exit, ksft_pr from lib.py import ksft_true, ksft_eq, ksft_ne, ksft_gt, ksft_raises @@ -58,6 +60,17 @@ def _make_psp_conn(cfg, version=0, ipver=None): return s +@contextmanager +def _make_lo_conn(): + # After tx-assoc, the client's egress is dropped, since lo has no + # psp_dev, so its FIN never reaches the server. Closing the server + # resets the unaccepted child, and the client accepts the cleartext + # RST because it hasn't received any PSP traffic yet. + with socket.create_server(("localhost", 0)) as srv, \ + socket.create_connection(srv.getsockname()[:2]) as s: + yield s + + def _close_conn(cfg, s): _send_with_ack(cfg, b'data close\0') s.close() @@ -200,20 +213,18 @@ def dev_rotate_spi(cfg): _init_psp_dev(cfg) top_a = top_b = 0 - with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s: + with _make_lo_conn() as s: assoc_a = cfg.pspnl.rx_assoc({"version": 0, "dev-id": cfg.psp_dev_id, "sock-fd": s.fileno()}) top_a = assoc_a['rx-key']['spi'] >> 31 - s.close() rot = cfg.pspnl.key_rotate({"id": cfg.psp_dev_id}) - with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s: + with _make_lo_conn() as s: ksft_eq(rot['id'], cfg.psp_dev_id) assoc_b = cfg.pspnl.rx_assoc({"version": 0, "dev-id": cfg.psp_dev_id, "sock-fd": s.fileno()}) top_b = assoc_b['rx-key']['spi'] >> 31 - s.close() ksft_ne(top_a, top_b) @@ -221,7 +232,7 @@ def assoc_basic(cfg): """ Test creating associations """ _init_psp_dev(cfg) - with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s: + with _make_lo_conn() as s: assoc = cfg.pspnl.rx_assoc({"version": 0, "dev-id": cfg.psp_dev_id, "sock-fd": s.fileno()}) @@ -234,7 +245,6 @@ def assoc_basic(cfg): "tx-key": assoc['rx-key'], "sock-fd": s.fileno()}) ksft_eq(len(assoc), 0) - s.close() def assoc_bad_dev(cfg): @@ -309,6 +319,32 @@ def assoc_sk_only_unconn(cfg): ksft_eq(the_exception.nl_msg.error, -errno.EINVAL) +def assoc_rx_unconnected(cfg): + """ Test that an Rx assoc is rejected on an unconnected socket """ + _init_psp_dev(cfg) + + with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s: + with ksft_raises(NlError) as cm: + cfg.pspnl.rx_assoc({"version": 0, + "dev-id": cfg.psp_dev_id, + "sock-fd": s.fileno()}) + ksft_eq(cm.exception.nl_msg.error, -errno.ENOTCONN) + ksft_eq(cm.exception.nl_msg.extack['bad-attr'], ".sock-fd") + + +def assoc_rx_listener(cfg): + """ Test that an Rx assoc is rejected on a listening socket """ + _init_psp_dev(cfg) + + with socket.create_server(("localhost", 0)) as s: + with ksft_raises(NlError) as cm: + cfg.pspnl.rx_assoc({"version": 0, + "dev-id": cfg.psp_dev_id, + "sock-fd": s.fileno()}) + ksft_eq(cm.exception.nl_msg.error, -errno.ENOTCONN) + ksft_eq(cm.exception.nl_msg.extack['bad-attr'], ".sock-fd") + + def assoc_version_mismatch(cfg): """ Test creating associations where Rx and Tx PSP versions do not match """ _init_psp_dev(cfg) @@ -320,7 +356,7 @@ def assoc_version_mismatch(cfg): # Translate versions to integers versions = [cfg.pspnl.consts["version"].entries[v].value for v in versions] - with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s: + with _make_lo_conn() as s: rx = cfg.pspnl.rx_assoc({"version": versions[0], "dev-id": cfg.psp_dev_id, "sock-fd": s.fileno()}) @@ -393,7 +429,7 @@ def assoc_twice(cfg): return assoc - with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s: + with _make_lo_conn() as s: assoc = rx_assoc_check(s) tx = cfg.pspnl.tx_assoc({"dev-id": cfg.psp_dev_id, "version": 0, @@ -402,7 +438,7 @@ def assoc_twice(cfg): ksft_eq(len(tx), 0) # Use the same Tx assoc second time - with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s2: + with _make_lo_conn() as s2: rx_assoc_check(s2) tx = cfg.pspnl.tx_assoc({"dev-id": cfg.psp_dev_id, "version": 0, @@ -410,8 +446,6 @@ def assoc_twice(cfg): "sock-fd": s2.fileno()}) ksft_eq(len(tx), 0) - s.close() - def _data_basic_send(cfg, version, ipver): """ Test basic data send """ diff --git a/tools/testing/selftests/filesystems/.gitignore b/tools/testing/selftests/filesystems/.gitignore index 57f5bbdbedff..cf79000e4092 100644 --- a/tools/testing/selftests/filesystems/.gitignore +++ b/tools/testing/selftests/filesystems/.gitignore @@ -6,3 +6,4 @@ anon_inode_test kernfs_test idmapped_tmpfile ustat_test +rw_hint_test diff --git a/tools/testing/selftests/filesystems/Makefile b/tools/testing/selftests/filesystems/Makefile index bc4bfb677589..fbd5c27505ef 100644 --- a/tools/testing/selftests/filesystems/Makefile +++ b/tools/testing/selftests/filesystems/Makefile @@ -1,7 +1,7 @@ # SPDX-License-Identifier: GPL-2.0 CFLAGS += $(KHDR_INCLUDES) -TEST_GEN_PROGS := devpts_pts anon_inode_test kernfs_test fclog ustat_test +TEST_GEN_PROGS := devpts_pts anon_inode_test kernfs_test fclog ustat_test rw_hint_test TEST_GEN_PROGS += idmapped_tmpfile TEST_GEN_PROGS_EXTENDED := dnotify_test diff --git a/tools/testing/selftests/filesystems/empty_mntns/.gitignore b/tools/testing/selftests/filesystems/empty_mntns/.gitignore index 99f89d329db2..27bcbcaeb1d1 100644 --- a/tools/testing/selftests/filesystems/empty_mntns/.gitignore +++ b/tools/testing/selftests/filesystems/empty_mntns/.gitignore @@ -2,3 +2,6 @@ clone3_empty_mntns_test empty_mntns_test overmount_chroot_test +internal_sb_reconfigure_test +nullfs_atime_test +root_readdir_test diff --git a/tools/testing/selftests/filesystems/empty_mntns/Makefile b/tools/testing/selftests/filesystems/empty_mntns/Makefile index 22e3fb915e81..22af2164c54e 100644 --- a/tools/testing/selftests/filesystems/empty_mntns/Makefile +++ b/tools/testing/selftests/filesystems/empty_mntns/Makefile @@ -4,9 +4,14 @@ CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) $(TOOLS_INCLUDES) LDLIBS += -lcap TEST_GEN_PROGS := empty_mntns_test overmount_chroot_test clone3_empty_mntns_test +TEST_GEN_PROGS += internal_sb_reconfigure_test nullfs_atime_test root_readdir_test + +LOCAL_HDRS += ../readdir_hold.h include ../../lib.mk $(OUTPUT)/empty_mntns_test: ../utils.c $(OUTPUT)/overmount_chroot_test: ../utils.c $(OUTPUT)/clone3_empty_mntns_test: ../utils.c +$(OUTPUT)/internal_sb_reconfigure_test: ../utils.c +$(OUTPUT)/root_readdir_test: LDLIBS += -pthread diff --git a/tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c b/tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c new file mode 100644 index 000000000000..cb645d1e5a9a --- /dev/null +++ b/tools/testing/selftests/filesystems/empty_mntns/internal_sb_reconfigure_test.c @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +/* + * The root of an empty mount namespace is a nullfs mount. Its superblock is + * kernel-internal and shared by every mount namespace. It can't be + * reconfigured, neither through fspick() nor through mount(MS_REMOUNT) nor + * through umount() of the root which remounts it read-only. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <string.h> +#include <sys/mount.h> +#include <sys/statfs.h> +#include <sys/statvfs.h> +#include <sys/syscall.h> +#include <sys/vfs.h> +#include <sys/wait.h> +#include <unistd.h> + +#include "../utils.h" +#include "../wrappers.h" +#include "empty_mntns.h" +#include "kselftest_harness.h" + +#ifndef __NR_fspick +#define __NR_fspick 433 +#endif + +static int sys_fspick(int dfd, const char *path, unsigned int flags) +{ + return syscall(__NR_fspick, dfd, path, flags); +} + +/* Child exit codes. */ +enum { + CHILD_OK, + CHILD_USERNS, /* could not create the user namespace */ + CHILD_UNSHARE, /* could not create the empty mount namespace */ + CHILD_FSPICK, /* fspick() of the root was not refused with EINVAL */ + CHILD_REMOUNT, /* mount(MS_REMOUNT) was not refused with EINVAL */ + CHILD_UMOUNT, /* umount() of the root succeeded */ + CHILD_STATFS, /* statfs() of the root failed */ + CHILD_RDONLY, /* the root ended up read-only */ +}; + +static int empty_mntns_child(void) +{ + struct statfs st; + + if (enter_userns()) + return CHILD_USERNS; + if (unshare(UNSHARE_EMPTY_MNTNS)) + return CHILD_UNSHARE; + + if (sys_fspick(AT_FDCWD, "/", 0) >= 0 || errno != EINVAL) + return CHILD_FSPICK; + if (!mount(NULL, "/", NULL, MS_REMOUNT | MS_RDONLY, NULL) || + errno != EINVAL) + return CHILD_REMOUNT; + if (!umount2("/", 0)) + return CHILD_UMOUNT; + if (statfs("/", &st)) + return CHILD_STATFS; + if (st.f_flags & ST_RDONLY) + return CHILD_RDONLY; + return CHILD_OK; +} + +FIXTURE(internal_sb_reconfigure) {}; + +FIXTURE_SETUP(internal_sb_reconfigure) +{ + pid_t pid; + int status; + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + if (enter_userns()) + _exit(1); + if (unshare(UNSHARE_EMPTY_MNTNS)) + _exit(1); + _exit(0); + } + ASSERT_EQ(waitpid(pid, &status, 0), pid); + if (!WIFEXITED(status) || WEXITSTATUS(status)) + SKIP(return, "UNSHARE_EMPTY_MNTNS not supported"); +} + +FIXTURE_TEARDOWN(internal_sb_reconfigure) {} + +TEST_F(internal_sb_reconfigure, nullfs_root) +{ + pid_t pid; + int status; + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(empty_mntns_child()); + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFEXITED(status)); + ASSERT_EQ(WEXITSTATUS(status), CHILD_OK); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c b/tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c new file mode 100644 index 000000000000..51e34f3c4f54 --- /dev/null +++ b/tools/testing/selftests/filesystems/empty_mntns/nullfs_atime_test.c @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * The root of every empty mount namespace is the same nullfs inode. A read + * of it by one user must not change the access time another user sees. + */ +#define _GNU_SOURCE +#include <dirent.h> +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/stat.h> +#include <sys/wait.h> + +#include "../../kselftest_harness.h" + +#ifndef UNSHARE_EMPTY_MNTNS +#define UNSHARE_EMPTY_MNTNS 0x00100000 +#endif + +enum { + CHILD_OK, + CHILD_UNSUPPORTED, + CHILD_SETUP, + CHILD_CHANGED, +}; + +static int wait_byte(int fd) +{ + char c; + + return read(fd, &c, 1) == 1 ? 0 : -1; +} + +static int send_byte(int fd) +{ + return write(fd, "x", 1) == 1 ? 0 : -1; +} + +static int empty_mntns(void) +{ + if (!unshare(UNSHARE_EMPTY_MNTNS)) + return 0; + return errno == EINVAL ? CHILD_UNSUPPORTED : CHILD_SETUP; +} + +/* the watcher: stats its root before and after the reader read its own */ +static int watcher(int to_reader, int from_reader) +{ + struct stat before, after; + int ret; + + ret = empty_mntns(); + if (ret) + return ret; + if (stat("/", &before)) + return CHILD_SETUP; + if (send_byte(to_reader) || wait_byte(from_reader)) + return CHILD_SETUP; + if (stat("/", &after)) + return CHILD_SETUP; + if (before.st_atim.tv_sec != after.st_atim.tv_sec || + before.st_atim.tv_nsec != after.st_atim.tv_nsec) + return CHILD_CHANGED; + return CHILD_OK; +} + +/* the reader: lists its own root, which is the same inode */ +static int reader(int to_watcher, int from_watcher) +{ + struct dirent *de; + DIR *d; + int ret; + + ret = empty_mntns(); + if (ret) + return ret; + if (wait_byte(from_watcher)) + return CHILD_SETUP; + d = opendir("/"); + if (!d) + return CHILD_SETUP; + while ((de = readdir(d))) + ; + closedir(d); + return send_byte(to_watcher) ? CHILD_SETUP : CHILD_OK; +} + +static int wait_child(pid_t pid) +{ + int status; + + if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status)) + return -1; + return WEXITSTATUS(status); +} + +TEST(empty_mntns_root_atime) +{ + int to_reader[2], to_watcher[2], w, r; + pid_t watcher_pid, reader_pid; + + if (geteuid()) + SKIP(return, "test requires root"); + ASSERT_EQ(pipe(to_reader), 0); + ASSERT_EQ(pipe(to_watcher), 0); + + watcher_pid = fork(); + ASSERT_GE(watcher_pid, 0); + if (watcher_pid == 0) + _exit(watcher(to_reader[1], to_watcher[0])); + reader_pid = fork(); + ASSERT_GE(reader_pid, 0); + if (reader_pid == 0) + _exit(reader(to_watcher[1], to_reader[0])); + + r = wait_child(reader_pid); + w = wait_child(watcher_pid); + if (r == CHILD_UNSUPPORTED || w == CHILD_UNSUPPORTED) + SKIP(return, "UNSHARE_EMPTY_MNTNS not supported"); + EXPECT_EQ(r, CHILD_OK); + EXPECT_EQ(w, CHILD_OK) + TH_LOG("the access time of the root changed while this namespace did nothing"); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c b/tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c new file mode 100644 index 000000000000..0ff544247027 --- /dev/null +++ b/tools/testing/selftests/filesystems/empty_mntns/root_readdir_test.c @@ -0,0 +1,45 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Reading the root of an empty mount namespace holds nothing that anybody + * else waits for: the directory never has an entry, so a readdir stuck in + * the page fault of its buffer stalls neither a create nor a lookup in it. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> + +#include "../../kselftest_harness.h" +#include "../readdir_hold.h" +#include "empty_mntns.h" + +TEST(readdir_blocks_nobody) +{ + struct readdir_hold hold; + bool stalled; + int dfd; + + if (geteuid()) + SKIP(return, "test requires root"); + if (readdir_hold_init(&hold)) + SKIP(return, "test requires userfaultfd"); + if (unshare(UNSHARE_EMPTY_MNTNS)) { + readdir_hold_destroy(&hold); + if (errno == EINVAL) + SKIP(return, "UNSHARE_EMPTY_MNTNS not supported"); + ASSERT_TRUE(false) + TH_LOG("unshare(UNSHARE_EMPTY_MNTNS): %m"); + } + dfd = open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(dfd, 0); + ASSERT_EQ(readdir_hold_check(&hold, dfd, &stalled), 0); + /* the lookup came back while the readdir was stuck in its fault */ + EXPECT_FALSE(stalled); + close(dfd); + readdir_hold_destroy(&hold); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/.gitignore b/tools/testing/selftests/filesystems/mount_cycle/.gitignore new file mode 100644 index 000000000000..30f7dd071bcb --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/.gitignore @@ -0,0 +1,8 @@ +# SPDX-License-Identifier: GPL-2.0-only +loop_cycle_test +unmounted_tree_test +overmount_reparent_test +mount_cover_test +locked_handle_test +nsfs_rbind_loop_test +overmount_ns_file_test diff --git a/tools/testing/selftests/filesystems/mount_cycle/Makefile b/tools/testing/selftests/filesystems/mount_cycle/Makefile new file mode 100644 index 000000000000..fbe8c5e19d42 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/Makefile @@ -0,0 +1,12 @@ +# SPDX-License-Identifier: GPL-2.0 +TEST_GEN_PROGS := loop_cycle_test unmounted_tree_test overmount_reparent_test mount_cover_test locked_handle_test +TEST_GEN_PROGS += nsfs_rbind_loop_test overmount_ns_file_test + +CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) + +LOCAL_HDRS += ../readdir_hold.h + +include ../../lib.mk + +$(OUTPUT)/locked_handle_test: LDLIBS += -pthread +$(OUTPUT)/mount_cover_test: LDLIBS += -pthread diff --git a/tools/testing/selftests/filesystems/mount_cycle/config b/tools/testing/selftests/filesystems/mount_cycle/config new file mode 100644 index 000000000000..3bd5ce46af74 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/config @@ -0,0 +1,39 @@ +CONFIG_USER_NS=y +CONFIG_TMPFS=y +CONFIG_BLK_DEV_LOOP=y +CONFIG_VFAT_FS=y +CONFIG_MSDOS_FS=y +CONFIG_NLS_CODEPAGE_437=y +CONFIG_NLS_ISO8859_1=y +CONFIG_MINIX_FS=y +CONFIG_AUTOFS_FS=y +CONFIG_ZRAM=y +CONFIG_ZRAM_WRITEBACK=y +CONFIG_KEYS=y +CONFIG_ECRYPT_FS=y +CONFIG_BINFMT_MISC=y +CONFIG_FUSE_FS=y +CONFIG_FUSE_PASSTHROUGH=y +CONFIG_BLK_DEV_ZONED=y +CONFIG_BLK_DEV_ZONED_LOOP=y +CONFIG_CONFIGFS_FS=y +CONFIG_USB_SUPPORT=y +CONFIG_USB=y +CONFIG_USB_STORAGE=y +CONFIG_SCSI=y +CONFIG_BLK_DEV_SD=y +CONFIG_USB_GADGET=y +CONFIG_USB_DUMMY_HCD=y +CONFIG_USB_CONFIGFS=y +CONFIG_USB_CONFIGFS_MASS_STORAGE=y +CONFIG_MD=y +CONFIG_BLK_DEV_MD=y +CONFIG_MD_RAID1=y +CONFIG_MD_BITMAP=y +CONFIG_MD_BITMAP_FILE=y +CONFIG_INOTIFY_USER=y +CONFIG_DNOTIFY=y +CONFIG_FANOTIFY=y +CONFIG_FILE_LOCKING=y +CONFIG_CRYPTO_AES=y +CONFIG_USERFAULTFD=y diff --git a/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c b/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c new file mode 100644 index 000000000000..b21f43471e49 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/locked_handle_test.c @@ -0,0 +1,377 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * may_decode_fh() refuses a file handle below a mount with locked children + * to root in a user namespace. That has to hold while the mount is lazily + * unmounted and its children, the locked ones too, are taken off it. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <pthread.h> +#include <sched.h> +#include <stdatomic.h> +#include <stdbool.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mount.h> +#include <sys/prctl.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> + +#include "../../kselftest_harness.h" + +#ifndef OPEN_TREE_CLONE +#define OPEN_TREE_CLONE 1 +#endif +#ifndef OPEN_TREE_CLOEXEC +#define OPEN_TREE_CLOEXEC O_CLOEXEC +#endif +#ifndef AT_RECURSIVE +#define AT_RECURSIVE 0x8000 +#endif + +#define DIR_LEN 64 +#define PATH_LEN 128 + +#define ROUNDS 100 /* lazy umounts raced per test */ +#define RACERS 3 /* threads in open_by_handle_at() */ +#define EXTRA_MOUNTS 64 /* make umount_tree() hold mount_lock longer */ +#define MAX_DELAY_US 3000 /* before the umount */ +#define TAIL_US 2000 /* after it */ + +struct handle { + struct file_handle fh; + unsigned char buf[MAX_HANDLE_SZ]; +}; + +/* exit codes of the child */ +enum { + CHILD_OK, + CHILD_SETUP, + CHILD_DECODED, /* decoded past a locked child */ + CHILD_MOUNTED, /* not refused while mounted */ + CHILD_UNMOUNTED, /* not refused once unmounted */ + CHILD_ALLOWED, /* refused where nothing is locked */ + CHILD_NOUSERNS, /* no user namespace to be had */ +}; + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +/* Become root in a new user namespace with a private mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + prctl(PR_SET_DUMPABLE, 1); + /* EINVAL: no USER_NS, ENOSPC: user.max_user_namespaces is 0, EPERM: an LSM */ + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return errno == EINVAL || errno == ENOSPC || errno == EPERM ? + CHILD_NOUSERNS : CHILD_SETUP; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return CHILD_SETUP; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return CHILD_SETUP; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return CHILD_SETUP; + if (setgid(0) || setuid(0)) + return CHILD_SETUP; + if (mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL)) + return CHILD_SETUP; + return CHILD_OK; +} + +static int get_handle(const char *path, struct handle *h) +{ + int mntid; + + h->fh.handle_bytes = MAX_HANDLE_SZ; + return name_to_handle_at(AT_FDCWD, path, &h->fh, &mntid, 0); +} + +static int decode(int dfd, struct handle *h) +{ + return open_by_handle_at(dfd, &h->fh, O_RDONLY | O_DIRECTORY | O_CLOEXEC); +} + +struct race { + int dfd; + struct handle *h; + atomic_int go; + atomic_int stop; + atomic_int won; /* the first decoded descriptor */ +}; + +static void *racer(void *arg) +{ + struct race *r = arg; + + while (!atomic_load(&r->go)) + ; + while (!atomic_load(&r->stop)) { + int none = -1; + int fd; + + fd = decode(r->dfd, r->h); + if (fd < 0) + continue; + if (!atomic_compare_exchange_strong(&r->won, &none, fd)) + close(fd); + } + return NULL; +} + +/* stop the @n racers started so far and wait for them, they spin on @r */ +static void stop_racers(struct race *r, pthread_t *th, int n) +{ + int i; + + atomic_store(&r->stop, 1); + atomic_store(&r->go, 1); /* one still waiting for the start sees stop next */ + for (i = 0; i < n; i++) + pthread_join(th[i], NULL); +} + +/* + * Bind @h recursively at @w, or take a detached copy of it, and race + * open_by_handle_at() of @inner against the lazy umount. The decoded + * descriptor, if there was one, is left in @won. + */ +static int race_round(const char *h, const char *w, struct handle *inner, + bool dissolve, int *won) +{ + struct race r = { .h = inner, .won = -1 }; + pthread_t th[RACERS]; + int treefd = -1, fd, i; + + if (dissolve) { + treefd = syscall(__NR_open_tree, AT_FDCWD, h, + OPEN_TREE_CLONE | AT_RECURSIVE | OPEN_TREE_CLOEXEC); + if (treefd < 0) + return CHILD_SETUP; + /* an ordinary descriptor on the copy, treefd's close dissolves it */ + r.dfd = openat(treefd, ".", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + } else { + if (mount(h, w, NULL, MS_BIND | MS_REC, NULL)) + return CHILD_SETUP; + r.dfd = open(w, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + } + if (r.dfd < 0) + return CHILD_SETUP; + + fd = decode(r.dfd, inner); + if (fd >= 0 || errno != EPERM) { + if (fd >= 0) + close(fd); + return CHILD_MOUNTED; + } + + for (i = 0; i < RACERS; i++) { + if (pthread_create(&th[i], NULL, racer, &r)) { + stop_racers(&r, th, i); + return CHILD_SETUP; + } + } + atomic_store(&r.go, 1); + usleep(rand() % MAX_DELAY_US); + if (dissolve) + close(treefd); + else if (umount2(w, MNT_DETACH)) { + stop_racers(&r, th, RACERS); + return CHILD_SETUP; + } + usleep(TAIL_US); + stop_racers(&r, th, RACERS); + + fd = decode(r.dfd, inner); + if (fd >= 0) { + close(fd); + return CHILD_UNMOUNTED; + } + close(r.dfd); + *won = atomic_load(&r.won); + return CHILD_OK; +} + +static int race_child(const char *h, const char *w, struct handle *inner, + bool dissolve) +{ + int ret, won, i; + + srand(getpid()); + ret = enter_userns(); + if (ret) + return ret; + for (i = 0; i < ROUNDS; i++) { + ret = race_round(h, w, inner, dissolve, &won); + if (ret) + return ret; + if (won >= 0) + return CHILD_DECODED; + } + return CHILD_OK; +} + +/* a bind without locked children decodes while mounted and not after */ +static int allowed_child(const char *plain, const char *w, struct handle *h) +{ + int dfd, fd, ret; + + ret = enter_userns(); + if (ret) + return ret; + if (mount(plain, w, NULL, MS_BIND | MS_REC, NULL)) + return CHILD_SETUP; + dfd = open(w, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + if (dfd < 0) + return CHILD_SETUP; + fd = decode(dfd, h); + if (fd < 0) + return CHILD_ALLOWED; + close(fd); + if (umount2(w, MNT_DETACH)) + return CHILD_SETUP; + fd = decode(dfd, h); + if (fd >= 0) { + close(fd); + return CHILD_UNMOUNTED; + } + return errno == EPERM ? CHILD_OK : CHILD_UNMOUNTED; +} + +FIXTURE(locked_handle) { + char base[DIR_LEN]; + char h[PATH_LEN]; /* the tree with the covered directory */ + char plain[PATH_LEN]; /* a tree with no mount in it */ + char w[PATH_LEN]; /* where the child binds either */ + struct handle inner; /* h/top/secret/inner, covered */ + struct handle uncovered; /* plain/inner */ +}; + +FIXTURE_SETUP(locked_handle) +{ + char p[PATH_LEN], e[PATH_LEN]; + int i; + + if (geteuid()) + SKIP(return, "test requires root"); + + snprintf(self->base, sizeof(self->base), "/tmp/locked_handle.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0); + + snprintf(self->h, sizeof(self->h), "%s/h", self->base); + snprintf(self->plain, sizeof(self->plain), "%s/plain", self->base); + snprintf(self->w, sizeof(self->w), "%s/w", self->base); + ASSERT_EQ(mkdir(self->h, 0755), 0); + ASSERT_EQ(mkdir(self->plain, 0755), 0); + ASSERT_EQ(mkdir(self->w, 0755), 0); + snprintf(p, sizeof(p), "%s/plain/inner", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + ASSERT_EQ(get_handle(p, &self->uncovered), 0); + + /* the handle is for this filesystem */ + ASSERT_EQ(mount("tmpfs", self->h, "tmpfs", 0, NULL), 0); + snprintf(p, sizeof(p), "%s/h/top", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + snprintf(p, sizeof(p), "%s/h/top/secret", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + snprintf(p, sizeof(p), "%s/h/top/secret/inner", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + ASSERT_EQ(get_handle(p, &self->inner), 0); + snprintf(e, sizeof(e), "%s/h/empty", self->base); + ASSERT_EQ(mkdir(e, 0755), 0); + + /* cover it, and some more so the umount takes longer */ + snprintf(p, sizeof(p), "%s/h/top/secret", self->base); + ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, NULL), 0); + for (i = 0; i < EXTRA_MOUNTS; i++) { + snprintf(p, sizeof(p), "%s/h/c%d", self->base, i); + ASSERT_EQ(mkdir(p, 0755), 0); + ASSERT_EQ(mount(e, p, NULL, MS_BIND, NULL), 0); + } +} + +FIXTURE_TEARDOWN(locked_handle) +{ + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +static int wait_child(pid_t pid) +{ + int status; + + if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status)) + return -1; + return WEXITSTATUS(status); +} + +TEST_F(locked_handle, lazy_umount) +{ + pid_t pid; + int ret; + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(race_child(self->h, self->w, &self->inner, false)); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + if (ret == CHILD_NOUSERNS) + SKIP(return, "no user namespaces"); + EXPECT_EQ(ret, CHILD_OK); +} + +TEST_F(locked_handle, dissolve) +{ + pid_t pid; + int ret; + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(race_child(self->h, self->w, &self->inner, true)); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + if (ret == CHILD_NOUSERNS) + SKIP(return, "no user namespaces"); + EXPECT_EQ(ret, CHILD_OK); +} + +TEST_F(locked_handle, allowed_use) +{ + pid_t pid; + int ret; + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(allowed_child(self->plain, self->w, &self->uncovered)); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + if (ret == CHILD_NOUSERNS) + SKIP(return, "no user namespaces"); + EXPECT_EQ(ret, CHILD_OK); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c b/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c new file mode 100644 index 000000000000..6b4f5304c2e4 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/loop_cycle_test.c @@ -0,0 +1,1542 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A mount that another namespace's rmdir detached, or that went down with + * the detached tree it was in when the tree's last fd was closed, keeps + * its submounts connected, and a connected submount is put by its + * parent's final mntput(). A submount whose filesystem keeps a file open + * on the parent then holds the parent's count above zero for good: + * nothing in userspace refers to either mount any more and nothing can + * release them. A loop device is the simplest such filesystem, its + * backing file sits on the parent. + */ +#define _GNU_SOURCE +#include <dirent.h> +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/ioctl.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <stdbool.h> +#include <sys/wait.h> +#include <unistd.h> +#include <linux/fuse.h> +#include <linux/keyctl.h> +#include <linux/major.h> +#include <linux/raid/md_u.h> +#include <linux/raid/md_p.h> +#include <linux/loop.h> +#include <linux/magic.h> +#include <sys/syscall.h> +#include <sys/sysmacros.h> +#include <sys/uio.h> + +#include "../wrappers.h" +#include "../../kselftest_harness.h" + +#define IMAGE_SIZE (1440 * 1024) +#define SECTOR 512 + +/* + * The tmpfs of the fixture. A directory of its own from mkdtemp(), so that + * the test needs no writable root directory and two instances don't take + * each other's mounts down. + */ +static char base[64]; + +/* The configfs directory of the gadget, named after the test process. */ +static char gadget[96]; + +/* A path below the base directory. Eight of them can be in use at a time. */ +static const char *at(const char *rel) +{ + static char buf[8][PATH_MAX]; + static unsigned int next; + char *p = buf[next++ % 8]; + + snprintf(p, PATH_MAX, "%s/%s", base, rel); + return p; +} + +/* A path below the gadget's configfs directory. */ +static const char *gat(const char *rel) +{ + static char buf[4][PATH_MAX]; + static unsigned int next; + char *p = buf[next++ % 4]; + + snprintf(p, PATH_MAX, "%s%s", gadget, rel); + return p; +} + +/* The loop devices this process has bound, to release them on every path. */ +#define MAX_LOOPS 4 +static int bound[MAX_LOOPS]; +static int nr_bound; + +/* What a child tells its parent: the loop devices it has bound, -1 for none. */ +struct report { + int n[2]; +}; + +/* A blank FAT12 floppy image: boot sector, two FATs, an empty root directory. */ +static int write_fat12(int fd) +{ + struct stat st; + unsigned char sector[SECTOR] = { + 0xeb, 0x3c, 0x90, 'M', 'S', 'W', 'I', 'N', '4', '.', '1', + [11] = 0x00, 0x02, /* bytes per sector: 512 */ + [13] = 1, /* sectors per cluster */ + [14] = 1, 0, /* reserved sectors */ + [16] = 2, /* FATs */ + [17] = 0xe0, 0x00, /* root directory entries: 224 */ + [19] = 0x40, 0x0b, /* total sectors: 2880 */ + [21] = 0xf0, /* media descriptor */ + [22] = 9, 0, /* sectors per FAT */ + [24] = 18, 0, /* sectors per track */ + [26] = 2, 0, /* heads */ + [38] = 0x29, /* extended boot signature */ + [39] = 0x12, 0x34, 0x56, 0x78, + [43] = 'N', 'O', ' ', 'N', 'A', 'M', 'E', ' ', ' ', ' ', ' ', + [54] = 'F', 'A', 'T', '1', '2', ' ', ' ', ' ', + [510] = 0x55, 0xaa, + }; + unsigned char fat[SECTOR] = { 0xf0, 0xff, 0xff }; + + if (pwrite(fd, sector, SECTOR, 0) != SECTOR) + return -1; + /* the first FAT and the second one, one sector each is enough */ + if (pwrite(fd, fat, SECTOR, 1 * SECTOR) != SECTOR || + pwrite(fd, fat, SECTOR, 10 * SECTOR) != SECTOR) + return -1; + if (fstat(fd, &st) || S_ISBLK(st.st_mode)) + return 0; + return ftruncate(fd, IMAGE_SIZE); +} + +/* + * A blank FAT12 image with 4 KiB sectors and @sectors of them, for a device + * with 4 KiB logical blocks (zram) or for a bigger image than a floppy. + */ +static int write_fat12_4k(int fd, unsigned int sectors) +{ + unsigned int fat_sectors = (sectors * 3 / 2 + 4095) / 4096; + unsigned char sector[4096] = { + 0xeb, 0x3c, 0x90, 'M', 'S', 'W', 'I', 'N', '4', '.', '1', + [11] = 0x00, 0x10, /* bytes per sector: 4096 */ + [13] = 1, /* sectors per cluster */ + [14] = 1, 0, /* reserved sectors */ + [16] = 2, /* FATs */ + [17] = 128, 0, /* root directory entries: one sector */ + [19] = sectors & 0xff, sectors >> 8, + [21] = 0xf8, /* media descriptor */ + [22] = fat_sectors, 0, + [24] = 63, 0, /* sectors per track */ + [26] = 255, 0, /* heads */ + [38] = 0x29, /* extended boot signature */ + [39] = 0x12, 0x34, 0x56, 0x78, + [43] = 'N', 'O', ' ', 'N', 'A', 'M', 'E', ' ', ' ', ' ', ' ', + [54] = 'F', 'A', 'T', '1', '2', ' ', ' ', ' ', + [510] = 0x55, 0xaa, + }; + unsigned char fat[4096] = { 0xf8, 0xff, 0xff }; + struct stat st; + + if (pwrite(fd, sector, sizeof(sector), 0) != sizeof(sector)) + return -1; + if (pwrite(fd, fat, sizeof(fat), 1 * 4096) != sizeof(fat) || + pwrite(fd, fat, sizeof(fat), (1 + fat_sectors) * 4096) != sizeof(fat)) + return -1; + if (fstat(fd, &st) || S_ISBLK(st.st_mode)) + return 0; + return ftruncate(fd, (off_t)sectors * 4096); +} + +#define MINIX_BLOCK 1024 +#define MINIX_BLOCKS 4096 /* a 4 MiB image */ +#define MINIX_INODES 512 +#define MINIX_ITABLE (MINIX_INODES * 32 / MINIX_BLOCK) +#define MINIX_FIRSTDATA (2 + 1 + 1 + MINIX_ITABLE) /* boot, super, imap, zmap, inodes */ + +/* + * A blank minix v1 image, for the holders that need a FIFO or a device + * node on the dying mount, which vfat can't hold. Superblock in block 1, + * one block each for the inode and zone bitmaps, the inode table, and + * the root directory in the first data zone. + */ +static int write_minix(int fd) +{ + struct { + __u16 s_ninodes, s_nzones, s_imap_blocks, s_zmap_blocks; + __u16 s_firstdatazone, s_log_zone_size; + __u32 s_max_size; + __u16 s_magic, s_state; + } sb = { + .s_ninodes = MINIX_INODES, + .s_nzones = MINIX_BLOCKS, + .s_imap_blocks = 1, + .s_zmap_blocks = 1, + .s_firstdatazone = MINIX_FIRSTDATA, + .s_max_size = (7 + 512 + 512 * 512) * MINIX_BLOCK, + .s_magic = MINIX_SUPER_MAGIC, + .s_state = 1, /* MINIX_VALID_FS */ + }; + struct { + __u16 i_mode, i_uid; + __u32 i_size, i_time; + __u8 i_gid, i_nlinks; + __u16 i_zone[9]; + } root = { + .i_mode = S_IFDIR | 0755, + .i_size = 2 * 16, + .i_nlinks = 2, + .i_zone = { MINIX_FIRSTDATA }, + }; + unsigned char imap[MINIX_BLOCK], zmap[MINIX_BLOCK], dir[MINIX_BLOCK] = {}; + int i; + + /* bit 0 is reserved in both maps, the root inode and its zone are in use */ + memset(imap, 0xff, sizeof(imap)); + for (i = 2; i <= MINIX_INODES; i++) + imap[i / 8] &= ~(1 << (i % 8)); + memset(zmap, 0xff, sizeof(zmap)); + for (i = 2; i <= MINIX_BLOCKS - MINIX_FIRSTDATA; i++) + zmap[i / 8] &= ~(1 << (i % 8)); + dir[0] = 1; + dir[2] = '.'; + dir[16] = 1; + dir[18] = '.'; + dir[19] = '.'; + + if (pwrite(fd, &sb, sizeof(sb), 1 * MINIX_BLOCK) != sizeof(sb) || + pwrite(fd, imap, sizeof(imap), 2 * MINIX_BLOCK) != sizeof(imap) || + pwrite(fd, zmap, sizeof(zmap), 3 * MINIX_BLOCK) != sizeof(zmap) || + pwrite(fd, &root, sizeof(root), 4 * MINIX_BLOCK) != sizeof(root) || + pwrite(fd, dir, sizeof(dir), MINIX_FIRSTDATA * MINIX_BLOCK) != sizeof(dir)) + return -1; + return ftruncate(fd, (off_t)MINIX_BLOCKS * MINIX_BLOCK); +} + +static int read_sysfs(const char *path, char *buf, size_t size) +{ + ssize_t n; + int fd; + + fd = open(path, O_RDONLY); + if (fd < 0) + return -1; + n = read(fd, buf, size - 1); + close(fd); + if (n < 0) + return -1; + buf[n] = '\0'; + return 0; +} + +/* Is @dev the mount source of this line of mountinfo? loop1 is not loop10. */ +static bool line_has_source(const char *line, const char *dev) +{ + const char *sep, *src, *end; + + sep = strstr(line, " - "); + if (!sep) + return false; + src = strchr(sep + 3, ' '); /* skip the filesystem type */ + if (!src) + return false; + src++; + end = strchr(src, ' '); + if (!end) + return false; + return (size_t)(end - src) == strlen(dev) && !strncmp(src, dev, end - src); +} + +/* + * Does a mount namespace of a process that /proc shows have a mount of @dev? + * The mounts under test are in no namespace at this point on any kernel, so + * this only checks that the test got as far as it thinks. + */ +static bool mounted_anywhere(const char *dev) +{ + char path[PATH_MAX], line[4096]; + struct dirent *de; + bool found = false; + DIR *proc; + FILE *f; + + proc = opendir("/proc"); + if (!proc) + return false; + while (!found && (de = readdir(proc))) { + if (de->d_name[0] < '0' || de->d_name[0] > '9') + continue; + snprintf(path, sizeof(path), "/proc/%s/mountinfo", de->d_name); + f = fopen(path, "re"); + if (!f) + continue; + while (fgets(line, sizeof(line), f)) { + if (line_has_source(line, dev)) { + found = true; + break; + } + } + fclose(f); + } + closedir(proc); + return found; +} + +/* Wait up to @ms milliseconds for the loop device to give up its backing file. */ +static bool loop_released(const char *sysfs, int ms) +{ + char buf[PATH_MAX]; + + for (; ms > 0; ms -= 100) { + if (read_sysfs(sysfs, buf, sizeof(buf)) < 0) + return errno == ENOENT; + usleep(100000); + } + return read_sysfs(sysfs, buf, sizeof(buf)) < 0 && errno == ENOENT; +} + +/* Tell the loop device to give its file up, now or when its last user is gone. */ +static void loop_clear(int n) +{ + char dev[32]; + int lfd; + + if (n < 0) + return; + snprintf(dev, sizeof(dev), "/dev/loop%d", n); + lfd = open(dev, O_RDWR); + if (lfd < 0) + return; + ioctl(lfd, LOOP_CLR_FD); + close(lfd); +} + +FIXTURE(loop_cycle) { + char dev[32]; /* the loop device the child set up */ + char sysfs[64]; /* its backing_file attribute */ + int loops[MAX_LOOPS]; /* every loop device the case has bound */ + int nr_loops; + bool gadget; /* the gadget's configfs directory exists */ +}; + +static void remember_loop(FIXTURE_DATA(loop_cycle) *self, int n) +{ + if (n >= 0 && self->nr_loops < MAX_LOOPS) + self->loops[self->nr_loops++] = n; +} + +FIXTURE_SETUP(loop_cycle) +{ + if (geteuid() != 0) + SKIP(return, "test requires CAP_SYS_ADMIN"); + if (access("/dev/loop-control", R_OK | W_OK)) + SKIP(return, "test requires loop devices"); + + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + + snprintf(base, sizeof(base), "/tmp/loop_cycle.XXXXXX"); + ASSERT_NE(mkdtemp(base), NULL); + /* a failed assertion in the setup does not run the teardown */ + ASSERT_EQ(mount("tmpfs", base, "tmpfs", 0, NULL), 0) + rmdir(base); + snprintf(gadget, sizeof(gadget), + "/sys/kernel/config/usb_gadget/kselftest_cycle_%d", getpid()); + self->dev[0] = '\0'; + self->nr_loops = 0; + self->gadget = false; +} + +static int mount_configfs(void); +static void gadget_remove(void); + +FIXTURE_TEARDOWN(loop_cycle) +{ + if (self->gadget) + gadget_remove(); + /* whatever a failed assertion has left bound */ + for (int i = 0; i < self->nr_loops; i++) + loop_clear(self->loops[i]); + umount2(base, MNT_DETACH); + rmdir(base); +} + +/* Child exit codes. */ +enum { + CHILD_OK, + CHILD_NS, /* could not set up the namespace or the tmpfs */ + CHILD_IMAGE, /* could not write the image */ + CHILD_LOOP, /* could not set up the loop device */ + CHILD_MOUNT, /* could not mount it (vfat and msdos both refused) */ + CHILD_PIPE, /* the parent went away */ + CHILD_HOLDER, /* could not set the holder up below the mount */ + CHILD_SKIP, /* the kernel lacks what the holder needs */ + CHILD_NOFS, /* the kernel lacks the filesystem of the image */ +}; + +/* Bind the loop device that LOOP_CTL_GET_FREE names to @ifd; the device number. */ +static int loop_bind_free(int ifd) +{ + int cfd, lfd, n; + char dev[32]; + + cfd = open("/dev/loop-control", O_RDWR); + if (cfd < 0) + return -1; + n = ioctl(cfd, LOOP_CTL_GET_FREE); + close(cfd); + if (n < 0) + return -1; + snprintf(dev, sizeof(dev), "/dev/loop%d", n); + lfd = open(dev, O_RDWR); + if (lfd < 0) + return -1; + if (ioctl(lfd, LOOP_SET_FD, ifd)) + n = -1; + close(lfd); + return n; +} + +/* + * Bind a free loop device to the open image @ifd; the device number. The + * device is free when LOOP_CTL_GET_FREE names it and may be somebody else's + * a moment later, so try again when it is busy. + */ +static int loop_bind(int ifd) +{ + int n = -1; + + for (int i = 0; i < 64 && n < 0; i++) { + n = loop_bind_free(ifd); + if (n < 0 && errno != EBUSY) + return -1; + } + if (n >= 0 && nr_bound < MAX_LOOPS) + bound[nr_bound++] = n; + return n; +} + +/* Mount a FAT image, as vfat or as msdos; -1 with ENODEV if the kernel has neither. */ +static int mount_fat(const char *dev, const char *mp) +{ + int err; + + if (!mount(dev, mp, "vfat", 0, NULL)) + return 0; + err = errno; + if (!mount(dev, mp, "msdos", 0, NULL)) + return 0; + if (err != ENODEV) + errno = err; + return -1; +} + +/* + * A child that gives up has to take down what it has set up: nobody else + * knows about it. Unmount, then tell the loop devices to let go. The mounts + * are released after this process is gone and the devices follow them. + */ +static int child_fails(int ret) +{ + umount2(at("vol"), MNT_DETACH); + umount2(at("vol2"), MNT_DETACH); + umount2(at("p"), MNT_DETACH); + umount2(at("img"), MNT_DETACH); + for (int i = 0; i < nr_bound; i++) + loop_clear(bound[i]); + return ret; +} + +/* Write an image to @img, bind a loop device to it and mount that at @mp; the device number. */ +static int loop_mount(const char *img, const char *mp) +{ + char dev[32]; + int ifd, n; + + ifd = open(img, O_RDWR | O_CREAT | O_EXCL, 0600); + if (ifd < 0 || write_fat12(ifd)) + return -CHILD_IMAGE; + n = loop_bind(ifd); + close(ifd); /* the loop device holds the file from now on */ + if (n < 0) + return -CHILD_LOOP; + snprintf(dev, sizeof(dev), "/dev/loop%d", n); + if (mkdir(mp, 0755)) + return -CHILD_MOUNT; + if (mount_fat(dev, mp)) + return errno == ENODEV ? -CHILD_NOFS : -CHILD_MOUNT; + return n; +} + +/* + * Two tmpfs mounts, each carrying the image of the loop mount below the + * other: the loop mount below vol has its image on vol2 and the other way + * round. + */ +static int crossed_child(int to_parent, int from_parent) +{ + struct report r; + char c; + + if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL)) + return CHILD_NS; + if (mkdir(at("vol"), 0755) || mount("tmpfs", at("vol"), "tmpfs", 0, NULL) || + mkdir(at("vol2"), 0755) || mount("tmpfs", at("vol2"), "tmpfs", 0, NULL)) + return child_fails(CHILD_NS); + r.n[0] = loop_mount(at("vol2/img"), at("vol/mnt")); + if (r.n[0] < 0) + return child_fails(-r.n[0]); + r.n[1] = loop_mount(at("vol/img"), at("vol2/mnt")); + if (r.n[1] < 0) + return child_fails(-r.n[1]); + if (write(to_parent, &r, sizeof(r)) != sizeof(r)) + return child_fails(CHILD_PIPE); + if (read(from_parent, &c, 1) != 1) + return child_fails(CHILD_PIPE); + return CHILD_OK; +} + +/* + * In its own mount namespace the child mounts a tmpfs on vol, + * puts a filesystem image on it, binds a loop device to the image and + * mounts that loop device below. The loop device's backing file is a + * reference on the mount the image is on, held by the loop device, held + * by the mounted filesystem, held by the mount below. + */ +static int loop_child(int to_parent, int from_parent) +{ + struct report r = { { -1, -1 } }; + char c; + + if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL)) + return CHILD_NS; + if (mkdir(at("vol"), 0755) || mount("tmpfs", at("vol"), "tmpfs", 0, NULL)) + return child_fails(CHILD_NS); + r.n[0] = loop_mount(at("vol/img"), at("vol/mnt")); + if (r.n[0] < 0) + return child_fails(-r.n[0]); + + if (write(to_parent, &r, sizeof(r)) != sizeof(r)) + return child_fails(CHILD_PIPE); + /* keep the namespace alive while the parent removes the directory */ + if (read(from_parent, &c, 1) != 1) + return child_fails(CHILD_PIPE); + return CHILD_OK; +} + +/* + * The child did not report. Skip if the kernel lacks something, fail with + * what the child said otherwise. A child that a signal killed has failed. + */ +#define CHILD_GAVE_UP(pid) do { \ + int __status; \ + \ + ASSERT_EQ(waitpid(pid, &__status, 0), pid); \ + if (WIFEXITED(__status) && WEXITSTATUS(__status) == CHILD_NOFS) \ + SKIP(return, "test requires the filesystem of the image (FAT or minix)"); \ + if (WIFEXITED(__status) && WEXITSTATUS(__status) == CHILD_SKIP) \ + SKIP(return, "the kernel lacks what this holder needs"); \ + ASSERT_TRUE(false) \ + TH_LOG("child failed to set up: %s %d", \ + WIFEXITED(__status) ? "exit status" : "signal", \ + WIFEXITED(__status) ? WEXITSTATUS(__status) : WTERMSIG(__status)); \ +} while (0) + +/* The child has to leave by itself and with nothing to complain about. */ +#define CHILD_LEFT(pid) do { \ + int __status; \ + \ + ASSERT_EQ(waitpid(pid, &__status, 0), pid); \ + ASSERT_TRUE(WIFEXITED(__status)); \ + ASSERT_EQ(WEXITSTATUS(__status), CHILD_OK); \ +} while (0) + +/* + * An exclusive open of the device fails while a filesystem holds it, and + * the only filesystem that ever did is the one mounted below the dead + * mount. Give a release in flight a moment. Then the device must clear + * right away rather than only be marked for autoclear. + */ +static void assert_loop_released(struct __test_metadata *_metadata, + FIXTURE_DATA(loop_cycle) *self) +{ + int lfd; + + for (int i = 0; i < 20; i++) { + lfd = open(self->dev, O_RDONLY | O_EXCL); + if (lfd >= 0) + break; + usleep(100000); + } + EXPECT_GE(lfd, 0) + TH_LOG("%s is still held by the loop mount below the dead mount: nothing refers to either mount and nothing can release them", + self->dev); + if (lfd >= 0) + close(lfd); + + lfd = open(self->dev, O_RDWR); + ASSERT_GE(lfd, 0); + ASSERT_EQ(ioctl(lfd, LOOP_CLR_FD), 0); + close(lfd); + ASSERT_TRUE(loop_released(self->sysfs, 5000)) + TH_LOG("%s kept its backing file after LOOP_CLR_FD: the filesystem on it is still mounted somewhere nobody can reach", + self->dev); + /* somebody else may bind it from now on, so the teardown leaves it alone */ + for (int i = 0; i < self->nr_loops; i++) { + char dev[32]; + + snprintf(dev, sizeof(dev), "/dev/loop%d", self->loops[i]); + if (!strcmp(dev, self->dev)) + self->loops[i] = -1; + } +} + +/* + * rmdir of vol from here, where it is not a mountpoint, detaches + * the child's tmpfs with the loop mount connected below it. Once the child + * is gone nothing refers to either mount. The loop device must then be + * free to give up its backing file, which only happens when the mounted + * filesystem below the detached tmpfs has been released. + */ +TEST_F(loop_cycle, detached_loop_mount_released) +{ + int to_parent[2], to_child[2]; + struct report r = { { -1, -1 } }; + char buf[PATH_MAX]; + pid_t pid; + + ASSERT_EQ(pipe(to_parent), 0); + ASSERT_EQ(pipe(to_child), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + close(to_parent[0]); + close(to_child[1]); + _exit(loop_child(to_parent[1], to_child[0])); + } + close(to_parent[1]); + close(to_child[0]); + + if (read(to_parent[0], &r, sizeof(r)) != sizeof(r)) + CHILD_GAVE_UP(pid); + remember_loop(self, r.n[0]); + snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", r.n[0]); + snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", r.n[0]); + ASSERT_EQ(read_sysfs(self->sysfs, buf, sizeof(buf)), 0); + ASSERT_NE(strstr(buf, "/vol/img"), NULL); + + /* not a mountpoint in this namespace, so the directory can go */ + ASSERT_EQ(rmdir(at("vol")), 0); + + /* the child leaves: its namespace and every reference it held are gone */ + ASSERT_EQ(write(to_child[1], "", 1), 1); + CHILD_LEFT(pid); + close(to_parent[0]); + close(to_child[1]); + + /* nothing can reach the two mounts any more */ + ASSERT_EQ(access(at("vol"), F_OK), -1); + ASSERT_FALSE(mounted_anywhere(self->dev)); + + assert_loop_released(_metadata, self); +} + +/* + * The same two mounts in a detached tree: a clone of vol from + * open_tree(), the image opened through the clone so that the loop device + * holds the clone, and the loop mount moved below the clone. The last + * close of the tree's fd dissolves the tree with the loop mount left + * connected below the dead clone, and nothing refers to either afterwards. + */ +TEST_F(loop_cycle, dissolved_tree_loop_mount_released) +{ + int tfd, ifd, fsfd, mfd, n; + char buf[PATH_MAX]; + + fsfd = sys_fsopen("vfat", 0); + if (fsfd < 0) + fsfd = sys_fsopen("msdos", 0); + if (fsfd < 0) + SKIP(return, "test requires a FAT filesystem"); + + ASSERT_EQ(mkdir(at("vol"), 0755), 0); + ASSERT_EQ(mount("tmpfs", at("vol"), "tmpfs", 0, NULL), 0); + tfd = sys_open_tree(AT_FDCWD, at("vol"), OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + ASSERT_GE(tfd, 0); + + /* the image, opened through the clone: the loop device holds the clone */ + ifd = openat(tfd, "img", O_RDWR | O_CREAT | O_EXCL, 0600); + ASSERT_GE(ifd, 0); + ASSERT_EQ(write_fat12(ifd), 0); + n = loop_bind(ifd); + close(ifd); + ASSERT_GE(n, 0); + remember_loop(self, n); + snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", n); + snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", n); + + /* the loop mount, moved below the clone */ + ASSERT_EQ(mkdirat(tfd, "mnt", 0755), 0); + ASSERT_EQ(sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", self->dev, 0), 0); + ASSERT_EQ(sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0), 0); + mfd = sys_fsmount(fsfd, 0, 0); + ASSERT_GE(mfd, 0); + close(fsfd); + ASSERT_EQ(sys_move_mount(mfd, "", tfd, "mnt", MOVE_MOUNT_F_EMPTY_PATH), 0); + close(mfd); + + ASSERT_EQ(read_sysfs(self->sysfs, buf, sizeof(buf)), 0); + ASSERT_NE(strstr(buf, "img"), NULL); + + /* the last fd of the tree: both mounts die, the loop mount connected */ + close(tfd); + + /* nothing can reach the two mounts any more */ + ASSERT_FALSE(mounted_anywhere(self->dev)); + + assert_loop_released(_metadata, self); +} + +/* + * The cycle in two steps: rmdir of vol leaves the loop mount below it + * connected while its image's mount, vol2, is alive; then rmdir of vol2 + * takes that one with the loop mount whose image is on the dead vol. Each + * dead mount now owns a loop mount whose filesystem pins the other. + */ +TEST_F(loop_cycle, crossed_images_released) +{ + int to_parent[2], to_child[2]; + struct report r = { { -1, -1 } }; + char sysfs[2][64], dev[2][32]; + pid_t pid; + + ASSERT_EQ(pipe(to_parent), 0); + ASSERT_EQ(pipe(to_child), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + close(to_parent[0]); + close(to_child[1]); + _exit(crossed_child(to_parent[1], to_child[0])); + } + close(to_parent[1]); + close(to_child[0]); + + if (read(to_parent[0], &r, sizeof(r)) != sizeof(r)) + CHILD_GAVE_UP(pid); + for (int i = 0; i < 2; i++) { + remember_loop(self, r.n[i]); + snprintf(dev[i], sizeof(dev[i]), "/dev/loop%d", r.n[i]); + snprintf(sysfs[i], sizeof(sysfs[i]), "/sys/block/loop%d/loop/backing_file", r.n[i]); + } + + /* step one: the mount with the first loop mount below it goes */ + ASSERT_EQ(rmdir(at("vol")), 0); + + /* step two: the other one, with the loop mount whose image is on the first */ + ASSERT_EQ(rmdir(at("vol2")), 0); + + /* the child leaves: its namespace and every reference it held are gone */ + ASSERT_EQ(write(to_child[1], "", 1), 1); + CHILD_LEFT(pid); + close(to_parent[0]); + close(to_child[1]); + + /* nothing can reach the four mounts any more */ + for (int i = 0; i < 2; i++) { + ASSERT_FALSE(mounted_anywhere(dev[i])); + strcpy(self->dev, dev[i]); + strcpy(self->sysfs, sysfs[i]); + assert_loop_released(_metadata, self); + } +} + +/* + * The holders. Each keeps a file or a path on a mount P for as long as + * its own filesystem or device lives, and each has that filesystem or + * device mounted at C below P. Once another namespace's rmdir has + * detached P with C connected below it and the child is gone, P is owned + * by nobody, C by P, and P is kept by whatever the holder still holds. + * + * P is a loop mount so that its death can be observed: the loop device + * gives its backing file up when P's superblock goes. The image sits on + * a tmpfs next to P, not above it, so the loop device's own reference is + * not part of the picture. + */ +enum holder { + HOLDER_AUTOFS, /* a FIFO on P as the daemon's pipe */ + HOLDER_ZRAM, /* a device node on P as the writeback device */ + HOLDER_ECRYPTFS, /* a directory on P as the lower directory */ + HOLDER_BINFMT_MISC, /* an executable on P as an 'F' interpreter */ + HOLDER_FUSE, /* a file on P as a passthrough backing file */ + HOLDER_ZLOOP, /* a directory on P for the zone files */ + HOLDER_GADGET, /* a file on P as a mass storage LUN, over dummy_hcd */ + HOLDER_MD, /* a file on P as an array's bitmap file */ +}; + +#define HOLDER_IMG at("img/p.img") +#define HOLDER_MNT at("p") +#define HOLDER_BELOW at("p/c") + +static int write_file(const char *path, const char *s) +{ + int fd = open(path, O_WRONLY); + ssize_t n; + + if (fd < 0) + return -1; + n = write(fd, s, strlen(s)); + close(fd); + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +/* Bind a free loop device to @img; the device number. */ +static int loop_attach(const char *img) +{ + int ifd, n; + + ifd = open(img, O_RDWR); + if (ifd < 0) + return -1; + n = loop_bind(ifd); + close(ifd); + return n; +} + +static int holder_autofs(void) +{ + char opts[64]; + int pfd; + + if (mkfifo(at("p/pipe"), 0600)) + return CHILD_HOLDER; + pfd = open(at("p/pipe"), O_RDWR); + if (pfd < 0) + return CHILD_HOLDER; + snprintf(opts, sizeof(opts), "fd=%d,minproto=5,maxproto=5", pfd); + if (mount("autofs", HOLDER_BELOW, "autofs", 0, opts)) + return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER; + close(pfd); /* the mount keeps its own */ + return CHILD_OK; +} + +/* + * zram's writeback device has to be a block device node, and that node has + * to be on P. Point it at a second loop device. + */ +static void zram_reset(void); + +static int holder_zram(void) +{ + char dev[32], buf[64]; + struct stat st; + int fd, n; + + if (access("/sys/block/zram0/backing_dev", W_OK)) + return CHILD_SKIP; + /* somebody else's device, leave it alone */ + if (read_sysfs("/sys/block/zram0/initstate", buf, sizeof(buf)) || buf[0] != '0') + return CHILD_SKIP; + fd = open(at("img/wb.img"), O_RDWR | O_CREAT | O_EXCL, 0600); + if (fd < 0 || ftruncate(fd, IMAGE_SIZE)) + return CHILD_HOLDER; + close(fd); + n = loop_attach(at("img/wb.img")); + if (n < 0) + return CHILD_HOLDER; + /* the minor is not the number of the device when loop has partitions */ + snprintf(dev, sizeof(dev), "/dev/loop%d", n); + if (stat(dev, &st) || mknod(at("p/wbdev"), S_IFBLK | 0600, st.st_rdev)) + return CHILD_HOLDER; + if (write_file("/sys/block/zram0/backing_dev", at("p/wbdev"))) + return CHILD_HOLDER; + snprintf(buf, sizeof(buf), "%d", 4 * 1024 * 1024); + if (write_file("/sys/block/zram0/disksize", buf)) + goto undo; + fd = open("/dev/zram0", O_RDWR); + if (fd < 0) + goto undo; + n = write_fat12_4k(fd, 1024); /* zram has 4 KiB blocks */ + close(fd); /* the reset in undo is refused while the device is open */ + if (n) + goto undo; + if (mount_fat("/dev/zram0", HOLDER_BELOW)) { + n = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER; + zram_reset(); + return n; + } + return CHILD_OK; +undo: + zram_reset(); + return CHILD_HOLDER; +} + +/* Is @alg in /proc/crypto? A module is listed once a request has loaded it. */ +static bool crypto_has(const char *alg) +{ + char line[256], name[64]; + bool found = false; + FILE *f; + + f = fopen("/proc/crypto", "r"); + if (!f) + return true; /* no way to tell, assume it is */ + while (!found && fgets(line, sizeof(line), f)) + found = sscanf(line, "name : %63s", name) == 1 && !strcmp(name, alg); + fclose(f); + return found; +} + +/* The kernel's auth token layout, which userspace has to match byte for byte. */ +struct ecryptfs_auth_tok { + __u16 version; + __u16 token_type; + __u32 flags; + struct { + __u32 flags, encrypted_key_size, decrypted_key_size; + __u8 encrypted_key[512], decrypted_key[64]; + } session_key; + __u8 reserved[32]; + struct { + __u32 password_bytes; + __s32 hash_algo; + __u32 hash_iterations, session_key_encryption_key_bytes, flags; + __u8 session_key_encryption_key[64]; + __u8 signature[17]; + __u8 salt[8]; + } password; +} __attribute__((packed)); + +#define ECRYPTFS_SIG "0123456789abcdef" + +static int holder_ecryptfs(void) +{ + struct ecryptfs_auth_tok tok = { + .version = 0x0004, + .token_type = 0, /* ECRYPTFS_PASSWORD */ + .password.session_key_encryption_key_bytes = 16, + .password.flags = 0x02, /* ECRYPTFS_SESSION_KEY_ENCRYPTION_KEY_SET */ + .password.signature = ECRYPTFS_SIG, + }; + + /* a session keyring of this child's own, so that the key goes with it */ + if (syscall(__NR_keyctl, KEYCTL_JOIN_SESSION_KEYRING, NULL) < 0) + return errno == ENOSYS ? CHILD_SKIP : CHILD_HOLDER; + if (syscall(__NR_add_key, "user", ECRYPTFS_SIG, &tok, sizeof(tok), + KEY_SPEC_SESSION_KEYRING) < 0) + return CHILD_HOLDER; + if (mkdir(at("p/lower"), 0755)) + return CHILD_HOLDER; + if (mount(at("p/lower"), HOLDER_BELOW, "ecryptfs", 0, + "ecryptfs_sig=" ECRYPTFS_SIG ",ecryptfs_cipher=aes,ecryptfs_key_bytes=16")) { + /* EINVAL without the cipher; a module is loaded by the attempt */ + if (errno == ENODEV || (errno == EINVAL && !crypto_has("aes"))) + return CHILD_SKIP; + return CHILD_HOLDER; + } + return CHILD_OK; +} + +/* + * binfmt_misc instances are per user namespace, so the mount below P is + * made from a new one, which gets a copy of P. + */ +static int holder_binfmt_misc(void) +{ + static const char interp[] = "#!/bin/true\n"; + char reg[PATH_MAX]; + int out; + + /* 'F' opens the interpreter at registration, nothing runs it here */ + out = open(at("p/interp"), O_WRONLY | O_CREAT | O_EXCL, 0755); + if (out < 0 || write(out, interp, sizeof(interp) - 1) != sizeof(interp) - 1) + return CHILD_HOLDER; + close(out); + + /* ENOSPC: user.max_user_namespaces is 0, EPERM: an LSM says no */ + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) { + if (errno == EINVAL || errno == ENOSPC || errno == EPERM) + return CHILD_SKIP; + return CHILD_HOLDER; + } + if (write_file("/proc/self/setgroups", "deny") || + write_file("/proc/self/uid_map", "0 0 1") || + write_file("/proc/self/gid_map", "0 0 1")) + return CHILD_HOLDER; + if (mount("binfmt_misc", HOLDER_BELOW, "binfmt_misc", 0, NULL)) + return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER; + snprintf(reg, sizeof(reg), ":cycle:E::cyc::%s:F", at("p/interp")); + if (write_file(at("p/c/register"), reg)) + return CHILD_HOLDER; + return CHILD_OK; +} + +/* + * A fuse server that only ever answers FUSE_INIT, with passthrough on, and + * then registers a file on P as a backing file. The registration alone + * makes the fuse superblock hold the file. + */ +static int holder_fuse(void) +{ + struct fuse_backing_map map = {}; + struct fuse_in_header *ih; + struct fuse_init_out init = { + .major = FUSE_KERNEL_VERSION, + .minor = FUSE_KERNEL_MINOR_VERSION, + .flags = FUSE_INIT_EXT, + .flags2 = FUSE_PASSTHROUGH >> 32, + .max_write = 4096, + .max_stack_depth = 1, + }; + struct fuse_out_header oh = { .len = sizeof(oh) + sizeof(init) }; + struct iovec iov[2] = { { &oh, sizeof(oh) }, { &init, sizeof(init) } }; + char opts[64], buf[8192]; + int ffd, bfd; + ssize_t n; + + bfd = open(at("p/backing"), O_RDWR | O_CREAT | O_EXCL, 0600); + if (bfd < 0) + return CHILD_HOLDER; + ffd = open("/dev/fuse", O_RDWR); + if (ffd < 0) + return CHILD_SKIP; + snprintf(opts, sizeof(opts), "fd=%d,rootmode=40000,user_id=0,group_id=0", ffd); + if (mount("fuse", HOLDER_BELOW, "fuse", 0, opts)) + return errno == ENODEV ? CHILD_SKIP : CHILD_HOLDER; + + n = read(ffd, buf, sizeof(buf)); + ih = (void *)buf; + if (n < (ssize_t)sizeof(*ih) || ih->opcode != FUSE_INIT) + return CHILD_HOLDER; + oh.unique = ih->unique; + if (writev(ffd, iov, 2) != (ssize_t)oh.len) + return CHILD_HOLDER; + + map.fd = bfd; + if (ioctl(ffd, FUSE_DEV_IOC_BACKING_OPEN, &map) < 0) { + /* EOPNOTSUPP: no FUSE_PASSTHROUGH, ENOTTY: no such ioctl */ + if (errno == EPERM || errno == EOPNOTSUPP || errno == ENOTTY) + return CHILD_SKIP; + return CHILD_HOLDER; + } + close(bfd); /* the connection keeps its own */ + return CHILD_OK; /* ffd stays open until the child exits */ +} + +/* Wait for a device node the kernel is about to create. */ +static int open_when_there(const char *dev, int flags, int ms) +{ + int fd; + + for (; ms > 0; ms -= 100) { + fd = open(dev, flags); + if (fd >= 0) + return fd; + usleep(100000); + } + return -1; +} + +/* + * zloop keeps every zone file open. One conventional zone is enough for a + * FAT image, the sequential one stays empty. + */ +static int holder_zloop(void) +{ + char cmd[PATH_MAX]; + int fd, ret = CHILD_HOLDER; + + if (access("/dev/zloop-control", W_OK)) + return CHILD_SKIP; + /* somebody else's device, leave it alone */ + if (!access("/sys/block/zloop0", F_OK)) + return CHILD_SKIP; + if (mkdir(at("p/zl"), 0755) || mkdir(at("p/zl/0"), 0755)) + return CHILD_HOLDER; + snprintf(cmd, sizeof(cmd), + "add id=0,capacity_mb=8,zone_size_mb=4,conv_zones=1,base_dir=%s", + at("p/zl")); + if (write_file("/dev/zloop-control", cmd)) + return CHILD_HOLDER; + fd = open_when_there("/dev/zloop0", O_RDWR, 5000); + if (fd < 0 || write_fat12_4k(fd, 1024)) /* 4 KiB blocks, like P */ + goto undo; + close(fd); + fd = -1; + if (mount_fat("/dev/zloop0", HOLDER_BELOW)) { + ret = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER; + goto undo; + } + return CHILD_OK; +undo: + if (fd >= 0) + close(fd); + write_file("/dev/zloop-control", "remove id=0"); + return ret; +} + +static int mount_configfs(void) +{ + if (access("/sys/kernel/config", F_OK)) + return -1; + if (mount("configfs", "/sys/kernel/config", "configfs", 0, NULL) && errno != EBUSY) + return -1; + return 0; +} + +/* Take the gadget out of configfs again, in the reverse order of its creation. */ +static void gadget_remove(void) +{ + if (mount_configfs()) + return; + write_file(gat("/functions/mass_storage.0/lun.0/file"), "\n"); + write_file(gat("/UDC"), "\n"); + unlink(gat("/configs/c.1/mass_storage.0")); + rmdir(gat("/configs/c.1/strings/0x409")); + rmdir(gat("/configs/c.1")); + rmdir(gat("/functions/mass_storage.0")); + rmdir(gat("/strings/0x409")); + rmdir(gat("")); +} + +/* The disk usb-storage created for the gadget, by the LUN's inquiry string. */ +static int find_gadget_disk(char *dev, size_t len, int ms) +{ + char path[PATH_MAX], model[64]; + struct dirent *de; + DIR *d; + + for (; ms > 0; ms -= 100, usleep(100000)) { + d = opendir("/sys/block"); + if (!d) + return -1; + while ((de = readdir(d))) { + if (strncmp(de->d_name, "sd", 2)) + continue; + snprintf(path, sizeof(path), "/sys/block/%s/device/model", de->d_name); + if (read_sysfs(path, model, sizeof(model)) || + strncmp(model, "File-Stor Gadget", 16)) + continue; + snprintf(dev, len, "/dev/%s", de->d_name); + closedir(d); + return 0; + } + closedir(d); + } + return -1; +} + +/* + * A mass storage gadget bound to the dummy UDC, so that this kernel is + * also the USB host that sees the LUN as a SCSI disk. sd locks the + * medium on open, which is what keeps the LUN's file from being ejected. + */ +#define GADGET_STEP(x) do { \ + if (x) { \ + fprintf(stderr, "gadget: %s failed: %s\n", #x, strerror(errno)); \ + gadget_remove(); \ + return CHILD_HOLDER; \ + } \ +} while (0) + +static int holder_gadget(void) +{ + char dev[PATH_MAX]; + int fd; + + if (mount_configfs()) + return CHILD_SKIP; + if (access("/sys/kernel/config/usb_gadget", F_OK) || + access("/sys/class/udc/dummy_udc.0", F_OK)) + return CHILD_SKIP; + + fd = open(at("p/lun.img"), O_RDWR | O_CREAT | O_EXCL, 0600); + if (fd < 0 || write_fat12(fd)) + return CHILD_HOLDER; + close(fd); + + GADGET_STEP(mkdir(gat(""), 0755)); + GADGET_STEP(write_file(gat("/idVendor"), "0x1d6b")); + GADGET_STEP(write_file(gat("/idProduct"), "0x0104")); + GADGET_STEP(mkdir(gat("/strings/0x409"), 0755)); + GADGET_STEP(write_file(gat("/strings/0x409/serialnumber"), "1")); + GADGET_STEP(write_file(gat("/strings/0x409/manufacturer"), "kselftest")); + GADGET_STEP(write_file(gat("/strings/0x409/product"), "cycle")); + GADGET_STEP(mkdir(gat("/configs/c.1"), 0755)); + GADGET_STEP(mkdir(gat("/configs/c.1/strings/0x409"), 0755)); + GADGET_STEP(write_file(gat("/configs/c.1/strings/0x409/configuration"), "c")); + GADGET_STEP(mkdir(gat("/functions/mass_storage.0"), 0755)); + GADGET_STEP(write_file(gat("/functions/mass_storage.0/lun.0/removable"), "1")); + GADGET_STEP(write_file(gat("/functions/mass_storage.0/lun.0/file"), at("p/lun.img"))); + GADGET_STEP(symlink(gat("/functions/mass_storage.0"), gat("/configs/c.1/mass_storage.0"))); + /* EBUSY: the controller is somebody else's, leave it alone */ + if (write_file(gat("/UDC"), "dummy_udc.0")) { + fd = errno == EBUSY ? CHILD_SKIP : CHILD_HOLDER; + gadget_remove(); + return fd; + } + + /* usb-storage waits a second before it scans the device */ + if (find_gadget_disk(dev, sizeof(dev), 15000)) { + /* without usb-storage and sd the LUN never shows up as a disk */ + fd = (access("/sys/bus/usb/drivers/usb-storage", F_OK) || + access("/sys/bus/scsi/drivers/sd", F_OK)) ? CHILD_SKIP : CHILD_HOLDER; + gadget_remove(); + return fd; + } + fd = open_when_there(dev, O_RDONLY, 5000); + GADGET_STEP(fd < 0); + close(fd); + if (mount_fat(dev, HOLDER_BELOW)) { + fd = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER; + gadget_remove(); + return fd; + } + return CHILD_OK; +} + +#define BITMAP_MAGIC 0x6d746962 + +/* + * A RAID1 of one loop device, not persistent, with its bitmap in a file + * on P. The bitmap file needs a superblock the kernel accepts; sync_size, + * uuid and events are not looked at for a non-persistent array. + */ +static int holder_md(void) +{ + struct { + __u32 magic, version; + __u8 uuid[16]; + __u64 events, events_cleared, sync_size; + __u32 state, chunksize, daemon_sleep, write_behind; + } bsb = { + .magic = BITMAP_MAGIC, + .version = 4, + .chunksize = 64 * 1024, + .daemon_sleep = 5, + }; + mdu_array_info_t info = { + .level = 1, + .raid_disks = 1, + .size = 8 * 1024, /* KiB */ + .not_persistent = 1, + }; + mdu_disk_info_t disk = { + .major = 7, + .state = (1 << MD_DISK_ACTIVE) | (1 << MD_DISK_SYNC), + }; + int fd, mdfd, bfd, n, ret = CHILD_HOLDER; + char dev[32], buf[4096]; + struct stat st; + + fd = open(at("img/md.img"), O_RDWR | O_CREAT | O_EXCL, 0600); + if (fd < 0 || ftruncate(fd, 8 * 1024 * 1024)) + return CHILD_HOLDER; + close(fd); + n = loop_attach(at("img/md.img")); + if (n < 0) + return CHILD_HOLDER; + /* the minor is not the number of the device when loop has partitions */ + snprintf(dev, sizeof(dev), "/dev/loop%d", n); + if (stat(dev, &st)) + return CHILD_HOLDER; + disk.major = major(st.st_rdev); + disk.minor = minor(st.st_rdev); + + bfd = open(at("p/bitmap"), O_RDWR | O_CREAT | O_EXCL, 0600); + if (bfd < 0 || write(bfd, &bsb, sizeof(bsb)) != sizeof(bsb) || ftruncate(bfd, 4096)) + return CHILD_HOLDER; + + if (read_sysfs("/proc/mdstat", buf, sizeof(buf))) { + fprintf(stderr, "md: no /proc/mdstat\n"); + return CHILD_SKIP; + } + if (!strstr(buf, "[raid1]")) { + fprintf(stderr, "md: no raid1 personality: %s\n", buf); + return CHILD_SKIP; + } + /* a node of our own: /dev may not have one and is not ours to change */ + if (mknod(at("img/md0"), S_IFBLK | 0600, makedev(MD_MAJOR, 0))) + return CHILD_HOLDER; + mdfd = open(at("img/md0"), O_RDWR); + if (mdfd < 0) + return CHILD_SKIP; + /* somebody else's array: SET_ARRAY_INFO would say EINVAL, not EBUSY */ + if (!read_sysfs("/sys/block/md0/md/array_state", buf, sizeof(buf)) && + strncmp(buf, "clear", 5)) { + fprintf(stderr, "md: md0 is in use: %s", buf); + close(mdfd); + return CHILD_SKIP; + } + /* the bitmap ops are only installed once a bitmap type is chosen */ + write_file("/sys/block/md0/md/bitmap_type", "bitmap"); + /* EBUSY: somebody else's array, leave it alone */ + if (ioctl(mdfd, SET_ARRAY_INFO, &info)) + return errno == EBUSY ? CHILD_SKIP : CHILD_HOLDER; + /* the array is ours from here on and has to be stopped on every path */ + if (ioctl(mdfd, ADD_NEW_DISK, &disk)) + goto undo; + /* attach the bitmap file before the array runs, the way mdadm does */ + if (ioctl(mdfd, SET_BITMAP_FILE, bfd)) { + fprintf(stderr, "md: SET_BITMAP_FILE: %s\n", strerror(errno)); + if (errno == EINVAL) + ret = CHILD_SKIP; + goto undo; + } + close(bfd); /* the array keeps its own */ + if (ioctl(mdfd, RUN_ARRAY, NULL)) { + fprintf(stderr, "md: RUN_ARRAY: %s\n", strerror(errno)); + goto undo; + } + if (write_fat12(mdfd)) + goto undo; + close(mdfd); + if (mount_fat(at("img/md0"), HOLDER_BELOW)) { + ret = errno == ENODEV ? CHILD_NOFS : CHILD_HOLDER; + mdfd = open(at("img/md0"), O_RDWR); + goto undo; + } + return CHILD_OK; +undo: + if (mdfd >= 0) { + ioctl(mdfd, STOP_ARRAY); + close(mdfd); + } + return ret; +} + +/* + * A device keeps its file for as long as it is configured, so once nothing + * below the dead mount is left the device has to be told to let go. With + * the cycle unbroken C's filesystem still holds the device and every one + * of these refuses. + */ +static int holder_let_go_once(enum holder holder) +{ + int fd, ret; + + switch (holder) { + case HOLDER_ZRAM: + return write_file("/sys/block/zram0/reset", "1"); + case HOLDER_ZLOOP: + return write_file("/dev/zloop-control", "remove id=0"); + case HOLDER_GADGET: + if (mount_configfs()) + return -1; + /* a zero-length write is a no-op for configfs; a newline ejects */ + return write_file(gat("/functions/mass_storage.0/lun.0/file"), "\n"); + case HOLDER_MD: + fd = open(at("img/md0"), O_RDONLY); + if (fd < 0) { + /* the child's node went with its tmpfs */ + mknod(at("md0"), S_IFBLK | 0600, makedev(MD_MAJOR, 0)); + fd = open(at("md0"), O_RDONLY); + } + if (fd < 0) + return -1; + ret = ioctl(fd, STOP_ARRAY); + close(fd); + return ret; + default: + return 0; + } +} + +static void zram_reset(void) +{ + holder_let_go_once(HOLDER_ZRAM); +} + +/* + * In its own mount namespace the child puts the image on a tmpfs next to + * P, mounts P from a loop device, and sets the holder up with its own + * filesystem or device mounted at C below P. + */ +static int holder_child(int to_parent, int from_parent, enum holder holder) +{ + bool fifo = holder == HOLDER_AUTOFS || holder == HOLDER_ZRAM; + /* room for a 4 MiB zone file or a floppy image on P */ + bool big = holder == HOLDER_ZLOOP || holder == HOLDER_GADGET; + struct report r = { { -1, -1 } }; + char dev[32], c; + int ifd, n, ret; + + if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL)) + return CHILD_NS; + if (mkdir(at("img"), 0755) || mount("tmpfs", at("img"), "tmpfs", 0, NULL)) + return child_fails(CHILD_NS); + + ifd = open(HOLDER_IMG, O_RDWR | O_CREAT | O_EXCL, 0600); + if (ifd < 0 || (fifo ? write_minix(ifd) : + big ? write_fat12_4k(ifd, 3072) : write_fat12(ifd))) + return child_fails(CHILD_IMAGE); + close(ifd); + n = loop_attach(HOLDER_IMG); + if (n < 0) + return child_fails(CHILD_LOOP); + snprintf(dev, sizeof(dev), "/dev/loop%d", n); + if (mkdir(HOLDER_MNT, 0755)) + return child_fails(CHILD_MOUNT); + if (fifo ? mount(dev, HOLDER_MNT, "minix", 0, NULL) : mount_fat(dev, HOLDER_MNT)) + return child_fails(errno == ENODEV ? CHILD_NOFS : CHILD_MOUNT); + if (mkdir(HOLDER_BELOW, 0755)) + return child_fails(CHILD_MOUNT); + + switch (holder) { + case HOLDER_AUTOFS: + ret = holder_autofs(); + break; + case HOLDER_ZRAM: + ret = holder_zram(); + break; + case HOLDER_ECRYPTFS: + ret = holder_ecryptfs(); + break; + case HOLDER_BINFMT_MISC: + ret = holder_binfmt_misc(); + break; + case HOLDER_FUSE: + ret = holder_fuse(); + break; + case HOLDER_ZLOOP: + ret = holder_zloop(); + break; + case HOLDER_GADGET: + ret = holder_gadget(); + break; + case HOLDER_MD: + ret = holder_md(); + break; + default: + ret = CHILD_HOLDER; + } + if (ret != CHILD_OK) + return child_fails(ret); + + /* P's device first, then the one the holder has bound for itself */ + r.n[0] = n; + if (nr_bound > 1) + r.n[1] = bound[1]; + if (write(to_parent, &r, sizeof(r)) != sizeof(r) || + read(from_parent, &c, 1) != 1) { + /* the parent went away and will not tell the devices to let go */ + umount2(HOLDER_BELOW, MNT_DETACH); + for (int i = 0; i < 50 && holder_let_go_once(holder); i++) + usleep(100000); + if (holder == HOLDER_GADGET) + gadget_remove(); + return child_fails(CHILD_PIPE); + } + return CHILD_OK; +} + +/* + * The release of C's filesystem may still be in flight when the child is + * gone, and a device that is still held refuses to let go, so try for a + * while. With the cycle unbroken it refuses for good. + */ +static void holder_let_go(enum holder holder) +{ + for (int i = 0; i < 50; i++) { + if (!holder_let_go_once(holder)) + return; + usleep(100000); + } +} + +/* + * rmdir of p from here, where it is a plain directory, detaches + * P in the child's namespace with C connected below it. Once the child is + * gone the holder's file on P is the only thing left that refers to P, + * and it is dropped only when C's filesystem dies, which waits for P. + * The loop device backing P tells whether that resolved. + */ +static void holder_cycle(struct __test_metadata *_metadata, + FIXTURE_DATA(loop_cycle) *self, enum holder holder) +{ + struct report r = { { -1, -1 } }; + int to_parent[2], from_parent[2]; + pid_t pid; + + ASSERT_EQ(pipe(to_parent), 0); + ASSERT_EQ(pipe(from_parent), 0); + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + close(to_parent[0]); + close(from_parent[1]); + _exit(holder_child(to_parent[1], from_parent[0], holder)); + } + close(to_parent[1]); + close(from_parent[0]); + + if (read(to_parent[0], &r, sizeof(r)) != sizeof(r)) + CHILD_GAVE_UP(pid); + self->gadget = holder == HOLDER_GADGET; + remember_loop(self, r.n[0]); + remember_loop(self, r.n[1]); + snprintf(self->dev, sizeof(self->dev), "/dev/loop%d", r.n[0]); + snprintf(self->sysfs, sizeof(self->sysfs), "/sys/block/loop%d/loop/backing_file", r.n[0]); + + ASSERT_EQ(rmdir(HOLDER_MNT), 0); + ASSERT_EQ(write(from_parent[1], "x", 1), 1); + CHILD_LEFT(pid); + close(to_parent[0]); + close(from_parent[1]); + + holder_let_go(holder); + assert_loop_released(_metadata, self); + /* the teardown takes the gadget and the holder's own loop device away */ +} + +TEST_F(loop_cycle, autofs_pipe_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_AUTOFS); +} + +TEST_F(loop_cycle, zram_writeback_node_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_ZRAM); +} + +TEST_F(loop_cycle, ecryptfs_lower_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_ECRYPTFS); +} + +TEST_F(loop_cycle, binfmt_misc_interpreter_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_BINFMT_MISC); +} + +TEST_F(loop_cycle, fuse_backing_file_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_FUSE); +} + +TEST_F(loop_cycle, zloop_zone_files_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_ZLOOP); +} + +/* up to 15 s for the disk to show up and 12 s for a device that is held */ +TEST_F_TIMEOUT(loop_cycle, mass_storage_lun_on_dead_mount_released, 120) +{ + holder_cycle(_metadata, self, HOLDER_GADGET); +} + +TEST_F(loop_cycle, md_bitmap_file_on_dead_mount_released) +{ + holder_cycle(_metadata, self, HOLDER_MD); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c b/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c new file mode 100644 index 000000000000..e3092454ae68 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/mount_cover_test.c @@ -0,0 +1,565 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An unmounted mount that would have stayed attached to its unmounted parent + * leaves a cover behind instead. A lookup on the parent at the mountpoint + * finds knullfs: an empty read-only directory that is shared by every cover + * and every kernel thread, so it can't be watched or locked and nothing can + * be mounted on it. The mount itself is a root from then on. Where a file + * was mounted, the stand-in is an empty regular file of the same instance. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdint.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/fanotify.h> +#include <sys/file.h> +#include <sys/inotify.h> +#include <sys/mount.h> +#include <sys/prctl.h> +#include <sys/stat.h> +#include <sys/statvfs.h> +#include <sys/syscall.h> +#include <sys/vfs.h> +#include <sys/wait.h> + +#include "../../kselftest_harness.h" +#include "../readdir_hold.h" + +#ifndef NULL_FS_MAGIC +#define NULL_FS_MAGIC 0x4E554C4C +#endif + +#ifndef TMPFS_MAGIC +#define TMPFS_MAGIC 0x01021994 +#endif + +#ifndef F_SETDELEG +#define F_SETDELEG (F_SETLEASE + 16) +#endif + +/* struct delegation of <linux/fcntl.h>, which doesn't mix with <fcntl.h> */ +struct delegation_req { + uint32_t d_flags; + uint16_t d_type; + uint16_t __pad; +}; + +#define DIR_LEN 64 +#define PATH_LEN 128 + +/* what the parent asks the child to do */ +#define CMD_RMDIR 'r' +#define CMD_CLOSE_T 't' +#define CMD_QUIT 'q' + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +static int touch(const char *path) +{ + int fd; + + fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644); + if (fd < 0) + return -1; + close(fd); + return 0; +} + +/* Become root in a new user namespace with a private mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + prctl(PR_SET_DUMPABLE, 1); + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + if (setgid(0) || setuid(0)) + return -1; + return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL); +} + +/* + * The child mounts P on @base/p, C on P/covered and the file P/src on P/file + * in a mount namespace of its own, binds P a second time at @base/q and hands + * out descriptors on P and on C. rmdir() of @base/p from here unmounts P + * together with C and the file bind. P and C are held by the descriptors and + * the unmounted children leave their covers behind. On request the child + * removes C's mountpoint through the bind. + * + * It also mounts T on @base/t with a child on T/covered, binds T at @base/u + * with a second child on the same dentry and hands out descriptors on T and + * U. rmdir() of both from here leaves two covers on one mountpoint. On request + * the child lets go of T. + */ +static int cover_child(const char *base, int to_parent, int from_parent) +{ + char p[PATH_LEN], c[PATH_LEN], q[PATH_LEN], qc[PATH_LEN]; + char f[PATH_LEN], src[PATH_LEN], t[PATH_LEN], u[PATH_LEN], tc[PATH_LEN]; + int fds[4], ret; + char cmd; + + if (unshare(CLONE_NEWNS) || mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL)) + return 1; + snprintf(p, sizeof(p), "%s/p", base); + if (mkdir(p, 0755) || mount("tmpfs", p, "tmpfs", 0, NULL)) + return 2; + snprintf(c, sizeof(c), "%s/p/covered", base); + if (mkdir(c, 0755) || mount("tmpfs", c, "tmpfs", 0, NULL)) + return 3; + snprintf(f, sizeof(f), "%s/p/file", base); + snprintf(src, sizeof(src), "%s/p/src", base); + if (touch(src) || touch(f) || mount(src, f, NULL, MS_BIND, NULL)) + return 4; + snprintf(q, sizeof(q), "%s/q", base); + if (mkdir(q, 0755) || mount(p, q, NULL, MS_BIND, NULL)) + return 5; + snprintf(t, sizeof(t), "%s/t", base); + if (mkdir(t, 0755) || mount("tmpfs", t, "tmpfs", 0, NULL)) + return 6; + snprintf(tc, sizeof(tc), "%s/t/covered", base); + if (mkdir(tc, 0755) || mount("tmpfs", tc, "tmpfs", 0, NULL)) + return 7; + snprintf(u, sizeof(u), "%s/u", base); + if (mkdir(u, 0755) || mount(t, u, NULL, MS_BIND, NULL)) + return 8; + snprintf(tc, sizeof(tc), "%s/u/covered", base); + if (mount("tmpfs", tc, "tmpfs", 0, NULL)) + return 9; + fds[0] = open(p, O_PATH | O_DIRECTORY | O_CLOEXEC); + fds[1] = open(c, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + fds[2] = open(t, O_PATH | O_DIRECTORY | O_CLOEXEC); + fds[3] = open(u, O_PATH | O_DIRECTORY | O_CLOEXEC); + if (fds[0] < 0 || fds[1] < 0 || fds[2] < 0 || fds[3] < 0) + return 10; + if (write(to_parent, fds, sizeof(fds)) != sizeof(fds)) + return 11; + + snprintf(qc, sizeof(qc), "%s/q/covered", base); + for (;;) { + if (read(from_parent, &cmd, 1) != 1) + return 12; + switch (cmd) { + case CMD_RMDIR: + ret = rmdir(qc) ? errno : 0; + if (write(to_parent, &ret, sizeof(ret)) != sizeof(ret)) + return 13; + break; + case CMD_CLOSE_T: + ret = close(fds[2]) ? errno : 0; + if (write(to_parent, &ret, sizeof(ret)) != sizeof(ret)) + return 13; + break; + case CMD_QUIT: + return 0; + default: + return 14; + } + } +} + +FIXTURE(mount_cover) { + char base[DIR_LEN]; + pid_t child; + int to_child; + int from_child; + int dfd; /* P, unmounted, held */ + int cfd; /* C, unmounted, held */ + int fd; /* what a lookup on P finds at C's mountpoint */ + int tfd; /* T, unmounted, held */ + int ufd; /* U, a bind of T, unmounted, held */ + int fan; /* fanotify group from before the user namespace, or -1 */ + struct readdir_hold hold; /* likewise from before, uffd -1 without */ +}; + +/* + * Everything the setup has made. The harness does not run the teardown + * when an assertion of the setup fails, so the setup calls this itself. + */ +static void cover_cleanup(FIXTURE_DATA(mount_cover) *self) +{ + char cmd = CMD_QUIT; + int status; + + if (self->fd >= 0) + close(self->fd); + if (self->cfd >= 0) + close(self->cfd); + if (self->dfd >= 0) + close(self->dfd); + if (self->tfd >= 0) + close(self->tfd); + if (self->ufd >= 0) + close(self->ufd); + if (self->fan >= 0) + close(self->fan); + readdir_hold_destroy(&self->hold); + if (self->child > 0) { + if (write(self->to_child, &cmd, 1) != 1) + kill(self->child, SIGKILL); + waitpid(self->child, &status, 0); + } + if (self->to_child >= 0) + close(self->to_child); + if (self->from_child >= 0) + close(self->from_child); + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +static int same_file(int fd1, int fd2) +{ + struct stat st1, st2; + + if (fstat(fd1, &st1) || fstat(fd2, &st2)) + return 0; + return st1.st_dev == st2.st_dev && st1.st_ino == st2.st_ino; +} + +FIXTURE_SETUP(mount_cover) +{ + int to_parent[2], to_child[2], fds[4], pidfd, status; + char dir[PATH_LEN]; + struct statfs sf; + + self->child = 0; + self->to_child = self->from_child = -1; + self->dfd = self->cfd = self->fd = self->tfd = self->ufd = -1; + + /* + * A group for plain events takes CAP_SYS_ADMIN in the initial user + * namespace, so get one before that is gone. An inode mark can be + * added to it from anywhere. + */ + self->fan = fanotify_init(FAN_CLASS_NOTIF | FAN_CLOEXEC, O_RDONLY); + /* same for the userfaultfd that holds a readdir in its fault */ + readdir_hold_init(&self->hold); + + snprintf(self->base, sizeof(self->base), "/tmp/mount_cover.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + if (enter_userns()) { + cover_cleanup(self); + SKIP(return, "test requires user namespaces"); + } + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0) + cover_cleanup(self); + + snprintf(dir, sizeof(dir), "%s/p", self->base); + ASSERT_EQ(pipe(to_parent), 0) + cover_cleanup(self); + ASSERT_EQ(pipe(to_child), 0) + cover_cleanup(self); + self->child = fork(); + ASSERT_GE(self->child, 0) + cover_cleanup(self); + if (self->child == 0) { + close(to_parent[0]); + close(to_child[1]); + _exit(cover_child(self->base, to_parent[1], to_child[0])); + } + close(to_parent[1]); + close(to_child[0]); + self->to_child = to_child[1]; + self->from_child = to_parent[0]; + if (read(self->from_child, fds, sizeof(fds)) != sizeof(fds)) { + pid_t pid = self->child; + + waitpid(pid, &status, 0); + self->child = 0; + cover_cleanup(self); + ASSERT_TRUE(false) + TH_LOG("child failed to set up: %s %d", + WIFEXITED(status) ? "step" : "signal", + WIFEXITED(status) ? WEXITSTATUS(status) : WTERMSIG(status)); + } + + pidfd = syscall(__NR_pidfd_open, self->child, 0); + ASSERT_GE(pidfd, 0) + cover_cleanup(self); + self->dfd = syscall(__NR_pidfd_getfd, pidfd, fds[0], 0); + self->cfd = syscall(__NR_pidfd_getfd, pidfd, fds[1], 0); + self->tfd = syscall(__NR_pidfd_getfd, pidfd, fds[2], 0); + self->ufd = syscall(__NR_pidfd_getfd, pidfd, fds[3], 0); + close(pidfd); + ASSERT_GE(self->dfd, 0) + cover_cleanup(self); + ASSERT_GE(self->cfd, 0) + cover_cleanup(self); + ASSERT_GE(self->tfd, 0) + cover_cleanup(self); + ASSERT_GE(self->ufd, 0) + cover_cleanup(self); + + /* unmounts P and C, C leaves its cover behind */ + ASSERT_EQ(rmdir(dir), 0) + cover_cleanup(self); + /* unmounts T and U with their children, two covers on one mountpoint */ + snprintf(dir, sizeof(dir), "%s/t", self->base); + ASSERT_EQ(rmdir(dir), 0) + cover_cleanup(self); + snprintf(dir, sizeof(dir), "%s/u", self->base); + ASSERT_EQ(rmdir(dir), 0) + cover_cleanup(self); + + self->fd = openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(self->fd, 0) + cover_cleanup(self); + ASSERT_EQ(fstatfs(self->fd, &sf), 0) + cover_cleanup(self); + ASSERT_EQ(sf.f_type, NULL_FS_MAGIC) + cover_cleanup(self); +} + +FIXTURE_TEARDOWN(mount_cover) +{ + cover_cleanup(self); +} + +TEST_F(mount_cover, not_watchable) +{ + char p[PATH_LEN]; + int ifd; + + snprintf(p, sizeof(p), "/proc/self/fd/%d", self->fd); + + ifd = inotify_init1(IN_CLOEXEC); + if (ifd < 0 && errno == ENOSYS) { + TH_LOG("no inotify in this kernel, skipping that part"); + } else { + ASSERT_GE(ifd, 0); + EXPECT_EQ(inotify_add_watch(ifd, p, IN_OPEN), -1); + EXPECT_EQ(errno, EINVAL); + close(ifd); + } + + /* dnotify ends up at the same place */ + EXPECT_EQ(fcntl(self->fd, F_NOTIFY, DN_ACCESS), -1); + EXPECT_EQ(errno, EINVAL); + + if (self->fan < 0) { + TH_LOG("no fanotify group without CAP_SYS_ADMIN in the initial user namespace, skipping the fanotify part"); + return; + } + EXPECT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD, FAN_OPEN, self->fd, NULL), -1); + EXPECT_EQ(errno, EINVAL); +} + +TEST_F(mount_cover, not_lockable) +{ + struct flock fl = { + .l_type = F_RDLCK, + .l_whence = SEEK_SET, + }; + + EXPECT_EQ(flock(self->fd, LOCK_EX | LOCK_NB), -1); + EXPECT_EQ(errno, ENOLCK); + EXPECT_EQ(fcntl(self->fd, F_SETLK, &fl), -1); + EXPECT_EQ(errno, ENOLCK); + EXPECT_EQ(fcntl(self->fd, F_GETLK, &fl), -1); + EXPECT_EQ(errno, ENOLCK); +} + +/* a lease is refused for the owner and for everybody else alike */ +TEST_F(mount_cover, not_leasable) +{ + struct delegation_req deleg = { + .d_type = F_RDLCK, + }; + + EXPECT_EQ(fcntl(self->fd, F_SETLEASE, F_RDLCK), -1); + EXPECT_TRUE(errno == EINVAL || errno == EACCES); + EXPECT_EQ(fcntl(self->fd, F_SETDELEG, &deleg), -1); + EXPECT_TRUE(errno == EINVAL || errno == EACCES); +} + +TEST_F(mount_cover, not_mountable) +{ + char p[PATH_LEN]; + + snprintf(p, sizeof(p), "/proc/self/fd/%d", self->fd); + EXPECT_EQ(mount("tmpfs", p, "tmpfs", 0, NULL), -1); + EXPECT_EQ(errno, ENOENT); +} + +TEST_F(mount_cover, read_only) +{ + struct statvfs sv; + + EXPECT_EQ(mkdirat(self->fd, "x", 0755), -1); + EXPECT_EQ(errno, ENOENT); + EXPECT_EQ(fchmod(self->fd, 0777), -1); + EXPECT_EQ(errno, EROFS); + /* the immutable inode is checked before the read-only mount */ + EXPECT_EQ(faccessat(self->fd, ".", W_OK, 0), -1); + EXPECT_EQ(errno, EPERM); + ASSERT_EQ(fstatvfs(self->fd, &sv), 0); + EXPECT_TRUE(sv.f_flag & ST_RDONLY); +} + +/* the stand-in is a root of its own, ".." stays put */ +TEST_F(mount_cover, island) +{ + int fd; + + fd = openat(self->fd, "..", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(fd, 0); + EXPECT_TRUE(same_file(fd, self->fd)); + close(fd); +} + +/* + * C is alive for as long as the child holds it, but it can't be reached + * through P anymore and it's a root of its own as well. + */ +TEST_F(mount_cover, held_child_detached) +{ + struct statfs sf; + int fd; + + ASSERT_EQ(fstatfs(self->cfd, &sf), 0); + EXPECT_EQ(sf.f_type, TMPFS_MAGIC); + + fd = openat(self->cfd, "..", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(fd, 0); + EXPECT_TRUE(same_file(fd, self->cfd)); + close(fd); + + fd = openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(fstatfs(fd, &sf), 0); + EXPECT_EQ(sf.f_type, NULL_FS_MAGIC); + close(fd); +} + +/* + * The cover goes with its mountpoint: once the child has removed C's + * mountpoint through the bind of P, the name is gone from P as well. + * What was opened through the cover before stays open. + */ +TEST_F(mount_cover, cover_goes_with_mountpoint) +{ + char cmd = CMD_RMDIR; + struct statfs sf; + int ret; + + ASSERT_EQ(write(self->to_child, &cmd, 1), 1); + ASSERT_EQ(read(self->from_child, &ret, sizeof(ret)), sizeof(ret)); + ASSERT_EQ(ret, 0); + + EXPECT_EQ(openat(self->dfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC), -1); + EXPECT_EQ(errno, ENOENT); + ASSERT_EQ(fstatfs(self->fd, &sf), 0); + EXPECT_EQ(sf.f_type, NULL_FS_MAGIC); +} + +/* + * A readdir of the stand-in holds nothing that others wait for: one holder + * sticks in the page fault of its buffer, and a create and a lookup of + * another come back meanwhile. + */ +TEST_F(mount_cover, readdir_blocks_nobody) +{ + bool stalled; + + if (self->hold.uffd < 0) + SKIP(return, "test requires userfaultfd"); + ASSERT_EQ(readdir_hold_check(&self->hold, self->fd, &stalled), 0); + EXPECT_FALSE(stalled); +} + +/* where a file was mounted, the stand-in is an empty regular file */ +TEST_F(mount_cover, file_stand_in) +{ + char src[PATH_LEN], p[PATH_LEN]; + struct statfs sf; + struct stat st; + char c; + int fd; + + fd = openat(self->dfd, "file", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(fstat(fd, &st), 0); + EXPECT_TRUE(S_ISREG(st.st_mode)); + ASSERT_EQ(fstatfs(fd, &sf), 0); + EXPECT_EQ(sf.f_type, NULL_FS_MAGIC); + /* the two stand-ins are two inodes */ + EXPECT_FALSE(same_file(fd, self->fd)); + EXPECT_EQ(read(fd, &c, 1), 0); + EXPECT_EQ(flock(fd, LOCK_EX | LOCK_NB), -1); + EXPECT_EQ(errno, ENOLCK); + + snprintf(src, sizeof(src), "%s/src", self->base); + ASSERT_EQ(touch(src), 0); + snprintf(p, sizeof(p), "/proc/self/fd/%d", fd); + EXPECT_EQ(mount(src, p, NULL, MS_BIND, NULL), -1); + EXPECT_EQ(errno, ENOENT); + close(fd); + + EXPECT_EQ(openat(self->dfd, "file", O_WRONLY | O_CLOEXEC), -1); + EXPECT_EQ(errno, EPERM); + EXPECT_EQ(openat(self->dfd, "file", O_RDONLY | O_DIRECTORY | O_CLOEXEC), -1); + EXPECT_EQ(errno, ENOTDIR); +} + +/* + * Two unmounted parents left covers on the same dentry. A lookup on + * either finds a stand-in, and U's cover stays when T goes. + */ +TEST_F(mount_cover, shared_mountpoint) +{ + char cmd = CMD_CLOSE_T; + struct statfs sf; + int fd, ret; + + fd = openat(self->tfd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(fstatfs(fd, &sf), 0); + EXPECT_EQ(sf.f_type, NULL_FS_MAGIC); + close(fd); + + fd = openat(self->ufd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(fstatfs(fd, &sf), 0); + EXPECT_EQ(sf.f_type, NULL_FS_MAGIC); + close(fd); + + /* the last references to T go, with T its cover */ + ASSERT_EQ(write(self->to_child, &cmd, 1), 1); + ASSERT_EQ(read(self->from_child, &ret, sizeof(ret)), sizeof(ret)); + ASSERT_EQ(ret, 0); + close(self->tfd); + self->tfd = -1; + + fd = openat(self->ufd, "covered", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(fstatfs(fd, &sf), 0); + EXPECT_EQ(sf.f_type, NULL_FS_MAGIC); + close(fd); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c b/tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c new file mode 100644 index 000000000000..0928a584eddc --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/nsfs_rbind_loop_test.c @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A recursive bind mount of a namespace file that lives in another mount + * namespace copies whatever is stacked on top of it there. If that includes + * the file of the caller's own mount namespace, or of an older one, the copy + * would pin the namespace it is put in forever. The bind mount has to be + * refused, a plain bind mount of the file itself still works. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/wait.h> + +#include "../../kselftest_harness.h" + +#define DIR_LEN 64 +#define PATH_LEN 128 + +/* Child exit codes. */ +enum { + CHILD_OK, + CHILD_UNSHARE, /* could not create the newer mount namespace */ + CHILD_PIPE, /* the parent went away */ + CHILD_TMPFS, /* could not mount the tmpfs in the new namespace */ + CHILD_REC_ALLOWED, /* the recursive bind mount was not refused */ + CHILD_REC_ERRNO, /* it was refused with the wrong error */ + CHILD_PLAIN_REFUSED, /* the plain bind mount of the file was refused */ +}; + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +static int create_file(const char *path) +{ + int fd = open(path, O_WRONLY | O_CREAT | O_CLOEXEC, 0644); + + if (fd < 0) + return -1; + close(fd); + return 0; +} + +/* Become root in a new user namespace with a private mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + if (setgid(0) || setuid(0)) + return -1; + return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL); +} + +static int send_msg(int fd, char msg) +{ + return write(fd, &msg, 1) == 1 ? 0 : -1; +} + +static char recv_msg(int fd) +{ + char msg; + + if (read(fd, &msg, 1) != 1) + return 0; + return msg; +} + +FIXTURE(nsfs_rbind_loop) { + char dir[DIR_LEN]; + char x[PATH_LEN]; +}; + +FIXTURE_SETUP(nsfs_rbind_loop) +{ + snprintf(self->dir, sizeof(self->dir), "/tmp/nsfs_rbind_loop.XXXXXX"); + ASSERT_NE(mkdtemp(self->dir), NULL); + if (enter_userns()) { + rmdir(self->dir); + SKIP(return, "test requires user namespaces"); + } + ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0); + snprintf(self->x, sizeof(self->x), "%s/x", self->dir); + ASSERT_EQ(create_file(self->x), 0); +} + +FIXTURE_TEARDOWN(nsfs_rbind_loop) +{ + umount2(self->dir, MNT_DETACH); + rmdir(self->dir); +} + +/* + * The child in the newer mount namespace binds the network namespace file + * mount of the parent through @fd. Recursively that would copy the mount of + * its own mount namespace file that the parent stacked on top. + */ +static int newer_ns_child(const char *dir, int fd, int to_parent, int from_parent) +{ + char src[32], y[PATH_LEN]; + + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + if (send_msg(to_parent, 'r') || recv_msg(from_parent) != 'g') + return CHILD_PIPE; + + snprintf(y, sizeof(y), "%s/y", dir); + if (mount("tmpfs", y, "tmpfs", 0, NULL)) + return CHILD_TMPFS; + snprintf(src, sizeof(src), "/proc/self/fd/%d", fd); + snprintf(y, sizeof(y), "%s/y/f", dir); + if (create_file(y)) + return CHILD_TMPFS; + + if (!mount(src, y, NULL, MS_BIND | MS_REC, NULL)) + return CHILD_REC_ALLOWED; + if (errno != EINVAL) + return CHILD_REC_ERRNO; + if (mount(src, y, NULL, MS_BIND, NULL)) + return CHILD_PLAIN_REFUSED; + umount2(y, MNT_DETACH); + return CHILD_OK; +} + +TEST_F(nsfs_rbind_loop, own_ns_file_below_foreign_source) +{ + int to_child[2], to_parent[2], fd, status; + char p[PATH_LEN]; + pid_t pid; + + snprintf(p, sizeof(p), "%s/y", self->dir); + ASSERT_EQ(mkdir(p, 0755), 0); + + /* M, a mount of our network namespace file, held by a descriptor */ + ASSERT_EQ(mount("/proc/self/ns/net", self->x, NULL, MS_BIND, NULL), 0); + fd = open(self->x, O_PATH | O_CLOEXEC); + ASSERT_GE(fd, 0); + + ASSERT_EQ(pipe(to_child), 0); + ASSERT_EQ(pipe(to_parent), 0); + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + close(to_child[1]); + close(to_parent[0]); + _exit(newer_ns_child(self->dir, fd, to_parent[1], to_child[0])); + } + close(to_child[0]); + close(to_parent[1]); + ASSERT_EQ(recv_msg(to_parent[0]), 'r'); + + /* the child's mount namespace file on top of M */ + snprintf(p, sizeof(p), "/proc/%d/ns/mnt", pid); + ASSERT_EQ(mount(p, self->x, NULL, MS_BIND, NULL), 0); + + ASSERT_EQ(send_msg(to_child[1], 'g'), 0); + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFEXITED(status)); + ASSERT_EQ(WEXITSTATUS(status), CHILD_OK); + + close(fd); + ASSERT_EQ(umount2(self->x, MNT_DETACH), 0); + ASSERT_EQ(umount2(self->x, MNT_DETACH), 0); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c b/tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c new file mode 100644 index 000000000000..ea1b33fa0bf8 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/overmount_ns_file_test.c @@ -0,0 +1,190 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A mount namespace file bind-mounted on top of the mount that is moved + * onto a shared mount isn't copied to the peers and slaves. The mount that + * already sits at the destination in a slave has to end up on top of the + * propagated copy, not below the root of the mount namespace file where no + * path walk ever finds it. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <signal.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> + +#include "../wrappers.h" +#include "../../kselftest_harness.h" + +#define DIR_LEN 64 +#define PATH_LEN 128 + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +static int create_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0644); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +/* the first bytes of the file at @path, "" if it can't be read */ +static const char *read_file(const char *path, char *buf, size_t len) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd >= 0) { + n = read(fd, buf, len - 1); + close(fd); + } + buf[n > 0 ? n : 0] = '\0'; + return buf; +} + +/* Become root in a new user namespace with a private mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + if (setgid(0) || setuid(0)) + return -1; + return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL); +} + +FIXTURE(overmount_ns_file) { + char dir[DIR_LEN]; + pid_t child; +}; + +FIXTURE_SETUP(overmount_ns_file) +{ + self->child = -1; + snprintf(self->dir, sizeof(self->dir), "/tmp/overmount_ns_file.XXXXXX"); + ASSERT_NE(mkdtemp(self->dir), NULL); + if (enter_userns()) { + rmdir(self->dir); + SKIP(return, "test requires user namespaces"); + } + ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0); +} + +FIXTURE_TEARDOWN(overmount_ns_file) +{ + if (self->child > 0) { + kill(self->child, SIGKILL); + waitpid(self->child, NULL, 0); + } + umount2(self->dir, MNT_DETACH); + rmdir(self->dir); +} + +/* + * A is a shared tmpfs and B its slave with Q, a bind mount of a file, on + * B/file. S is a bind mount of a file with N, a bind mount of a newer mount + * namespace's file, on top of it. S is moved onto A/file. Its copy S' lands + * on B/file below Q, without N. B/file keeps reading Q and once Q is + * unmounted it reads S'. + */ +TEST_F(overmount_ns_file, existing_mount_stays_on_top) +{ + char a[PATH_LEN], b[PATH_LEN], s[PATH_LEN], p[PATH_LEN], buf[16]; + int fd, pfd[2]; + char c; + + snprintf(a, sizeof(a), "%s/A", self->dir); + snprintf(b, sizeof(b), "%s/B", self->dir); + snprintf(s, sizeof(s), "%s/s", self->dir); + ASSERT_EQ(mkdir(a, 0755), 0); + ASSERT_EQ(mkdir(b, 0755), 0); + ASSERT_EQ(mkdir(s, 0755), 0); + + /* A shared, B its slave, Q on B/file */ + ASSERT_EQ(mount("tmpfs", a, "tmpfs", 0, NULL), 0); + ASSERT_EQ(mount(NULL, a, NULL, MS_SHARED, NULL), 0); + snprintf(p, sizeof(p), "%s/A/file", self->dir); + ASSERT_EQ(create_file(p, "A"), 0); + ASSERT_EQ(mount(a, b, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, b, NULL, MS_SLAVE, NULL), 0); + snprintf(p, sizeof(p), "%s/Q", self->dir); + ASSERT_EQ(create_file(p, "Q"), 0); + snprintf(b, sizeof(b), "%s/B/file", self->dir); + ASSERT_EQ(mount(p, b, NULL, MS_BIND, NULL), 0); + + /* S on s/f, pinned by a file descriptor before N goes on top */ + ASSERT_EQ(mount("tmpfs", s, "tmpfs", 0, NULL), 0); + snprintf(p, sizeof(p), "%s/s/f", self->dir); + ASSERT_EQ(create_file(p, "f"), 0); + snprintf(s, sizeof(s), "%s/s/S", self->dir); + ASSERT_EQ(create_file(s, "S"), 0); + ASSERT_EQ(mount(s, p, NULL, MS_BIND, NULL), 0); + fd = open(p, O_PATH | O_CLOEXEC); + ASSERT_GE(fd, 0); + + /* a newer mount namespace whose file can be bound */ + ASSERT_EQ(pipe(pfd), 0); + self->child = fork(); + ASSERT_GE(self->child, 0); + if (self->child == 0) { + close(pfd[0]); + if (unshare(CLONE_NEWNS) || write(pfd[1], "r", 1) != 1) + _exit(1); + pause(); + _exit(0); + } + close(pfd[1]); + ASSERT_EQ(read(pfd[0], &c, 1), 1); + close(pfd[0]); + snprintf(s, sizeof(s), "/proc/%d/ns/mnt", self->child); + ASSERT_EQ(mount(s, p, NULL, MS_BIND, NULL), 0); + + ASSERT_STREQ(read_file(b, buf, sizeof(buf)), "Q"); + + snprintf(a, sizeof(a), "%s/A/file", self->dir); + ASSERT_EQ(sys_move_mount(fd, "", AT_FDCWD, a, MOVE_MOUNT_F_EMPTY_PATH), 0); + close(fd); + + /* Q is still on top of the copy in B and can be unmounted */ + EXPECT_STREQ(read_file(b, buf, sizeof(buf)), "Q"); + EXPECT_EQ(umount2(b, 0), 0); + EXPECT_STREQ(read_file(b, buf, sizeof(buf)), "S"); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c b/tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c new file mode 100644 index 000000000000..46a56b722f3e --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/overmount_reparent_test.c @@ -0,0 +1,181 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * propagate_umount() slides a surviving overmount off a mount that is being + * unmounted and unmounts that mount right after. If a file descriptor keeps + * the unmounted mount alive its ->overmount is left pointing at the moved + * mount, which is freed on its own schedule. move_mount(MOVE_MOUNT_BENEATH) + * with such a file descriptor as the target walks ->overmount in + * topmost_overmount() before it checks that the target is still mounted and + * must not step into the freed mount. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/syscall.h> + +#include "../wrappers.h" +#include "../../kselftest_harness.h" + +#ifndef MOVE_MOUNT_BENEATH +#define MOVE_MOUNT_BENEATH 0x00000200 +#endif + +#ifndef FSCONFIG_CMD_CREATE +#define FSCONFIG_CMD_CREATE 6 +#endif + +#define DIR_LEN 64 +#define PATH_LEN 128 + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +/* Become root in a new user namespace with a private mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + if (setgid(0) || setuid(0)) + return -1; + return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL); +} + +/* A detached tmpfs mount to serve as the source of a move. */ +static int detached_tmpfs(void) +{ + int sfd, mfd; + + sfd = sys_fsopen("tmpfs", 0); + if (sfd < 0) + return -1; + if (sys_fsconfig(sfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0)) { + close(sfd); + return -1; + } + mfd = sys_fsmount(sfd, 0, 0); + close(sfd); + return mfd; +} + +FIXTURE(overmount_reparent) { + char dir[DIR_LEN]; +}; + +FIXTURE_SETUP(overmount_reparent) +{ + snprintf(self->dir, sizeof(self->dir), "/tmp/overmount_reparent.XXXXXX"); + ASSERT_NE(mkdtemp(self->dir), NULL); + if (enter_userns()) { + rmdir(self->dir); + SKIP(return, "test requires user namespaces"); + } +} + +FIXTURE_TEARDOWN(overmount_reparent) +{ + char p[PATH_LEN]; + + snprintf(p, sizeof(p), "%s/b", self->dir); + umount2(p, MNT_DETACH); + snprintf(p, sizeof(p), "%s/a", self->dir); + umount2(p, MNT_DETACH); +} + +/* + * A shared mount base_a is bind-mounted to base_b as its peer. A mount X on + * base_a/mp propagates a copy X_b onto base_b/mp. A file descriptor pins X_b, + * then X_b is made private and an overmount is stacked on its root. + * + * A lazy unmount of X propagates to X_b: X_b is committed to the unmount while + * its overmount survives, so propagate_umount() reparents the overmount onto + * base_b and unmounts X_b, which the file descriptor keeps alive. Unmounting + * the reparented overmount frees it. X_b->overmount now dangles unless + * mnt_change_mountpoint() reset it. + * + * move_mount(MOVE_MOUNT_BENEATH) through the file descriptor makes + * do_lock_mount() follow X_b->overmount in topmost_overmount() before it + * notices that X_b is no longer mounted. With the pointer reset the move fails + * cleanly with ENOENT; otherwise it reads the freed overmount. + */ +TEST_F(overmount_reparent, dead_overmount_holder_not_followed) +{ + char a[PATH_LEN], b[PATH_LEN], amp[PATH_LEN], bmp[PATH_LEN]; + int fd, mfd, ret; + + snprintf(a, sizeof(a), "%s/a", self->dir); + snprintf(b, sizeof(b), "%s/b", self->dir); + ASSERT_EQ(mkdir(a, 0755), 0); + ASSERT_EQ(mkdir(b, 0755), 0); + + ASSERT_EQ(mount("tmpfs", a, "tmpfs", 0, NULL), 0); + ASSERT_EQ(mount(NULL, a, NULL, MS_SHARED, NULL), 0); + ASSERT_EQ(mount(a, b, NULL, MS_BIND, NULL), 0); + + snprintf(amp, sizeof(amp), "%s/a/mp", self->dir); + snprintf(bmp, sizeof(bmp), "%s/b/mp", self->dir); + /* base_a and base_b share one superblock, so this dir is in both. */ + ASSERT_EQ(mkdir(amp, 0755), 0); + + /* X on base_a/mp propagates a copy X_b onto base_b/mp. */ + ASSERT_EQ(mount("tmpfs", amp, "tmpfs", 0, NULL), 0); + + /* Pin X_b before anything is stacked on top of it. */ + fd = open(bmp, O_PATH | O_CLOEXEC); + ASSERT_GE(fd, 0); + + /* Keep the overmount local to X_b. */ + ASSERT_EQ(mount(NULL, bmp, NULL, MS_PRIVATE, NULL), 0); + + /* The overmount on X_b's root: X_b->overmount points at it. */ + ASSERT_EQ(mount("tmpfs", bmp, "tmpfs", 0, NULL), 0); + + /* Reparents the overmount onto base_b and unmounts X_b. */ + ASSERT_EQ(umount2(amp, MNT_DETACH), 0); + + /* Free the reparented overmount. */ + ASSERT_EQ(umount2(bmp, MNT_DETACH), 0); + usleep(100000); + + mfd = detached_tmpfs(); + ASSERT_GE(mfd, 0); + + ret = sys_move_mount(mfd, "", fd, "", + MOVE_MOUNT_F_EMPTY_PATH | MOVE_MOUNT_T_EMPTY_PATH | + MOVE_MOUNT_BENEATH); + EXPECT_EQ(ret, -1); + EXPECT_EQ(errno, ENOENT); + + close(mfd); + close(fd); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mount_cycle/settings b/tools/testing/selftests/filesystems/mount_cycle/settings new file mode 100644 index 000000000000..694d70710ff0 --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/settings @@ -0,0 +1 @@ +timeout=300 diff --git a/tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c b/tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c new file mode 100644 index 000000000000..233f8ea92efc --- /dev/null +++ b/tools/testing/selftests/filesystems/mount_cycle/unmounted_tree_test.c @@ -0,0 +1,484 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An unmounted mount tree that a file descriptor keeps alive is put and + * vacated behind the root's back, so nothing may walk it under + * namespace_sem: it can't be copied recursively and its propagation can't + * be changed. The mount at its root alone can still be copied and that copy + * is an ordinary mount. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> +#include <unistd.h> +#include <linux/mount.h> + +#include "../wrappers.h" +#include "../../kselftest_harness.h" + +#ifndef __NR_mount_setattr +#define __NR_mount_setattr 442 +#endif + +#ifndef __NR_pidfd_open +#define __NR_pidfd_open 434 +#endif + +#define PATH_LEN 64 + +static inline int sys_mount_setattr(int dfd, const char *path, unsigned int flags, + struct mount_attr *attr, size_t size) +{ + return syscall(__NR_mount_setattr, dfd, path, flags, attr, size); +} + +static inline int sys_pidfd_open(pid_t pid, unsigned int flags) +{ + return syscall(__NR_pidfd_open, pid, flags); +} + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +static int touch(const char *path) +{ + int fd; + + fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644); + if (fd < 0) + return -1; + close(fd); + return 0; +} + +/* Become root in a new user namespace with a private mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + if (setgid(0) || setuid(0)) + return -1; + return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL); +} + +/* A private mount namespace of our own, all mounts private. */ +static int own_mntns(void) +{ + if (unshare(CLONE_NEWNS)) + return -1; + return mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL); +} + +FIXTURE(unmounted_tree) { + char dir[PATH_LEN]; +}; + +FIXTURE_SETUP(unmounted_tree) +{ + snprintf(self->dir, sizeof(self->dir), "/tmp/unmounted_tree.XXXXXX"); + ASSERT_NE(mkdtemp(self->dir), NULL); + if (enter_userns()) { + rmdir(self->dir); + SKIP(return, "test requires user namespaces"); + } + ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0); +} + +FIXTURE_TEARDOWN(unmounted_tree) +{ + umount2(self->dir, MNT_DETACH); + rmdir(self->dir); +} + +/* A recursive copy of the mount @fd refers to must fail with EINVAL. */ +static void assert_not_walked(struct __test_metadata *_metadata, int fd, + const char *target) +{ + char link[PATH_LEN]; + int tfd; + + snprintf(link, sizeof(link), "/proc/self/fd/%d", fd); + EXPECT_EQ(mount(link, target, NULL, MS_BIND | MS_REC, NULL), -1); + EXPECT_EQ(errno, EINVAL); + + tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | AT_RECURSIVE | OPEN_TREE_CLONE | + OPEN_TREE_CLOEXEC); + EXPECT_LT(tfd, 0); + EXPECT_EQ(errno, EINVAL); + if (tfd >= 0) + close(tfd); +} + +/* The mount @fd refers to can be copied on its own and mounted on @target. */ +static void assert_root_copied(struct __test_metadata *_metadata, int fd, + const char *target) +{ + int tfd; + + tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + ASSERT_GE(tfd, 0); + ASSERT_EQ(sys_move_mount(tfd, "", AT_FDCWD, target, MOVE_MOUNT_F_EMPTY_PATH), 0); + close(tfd); + ASSERT_EQ(umount2(target, MNT_DETACH), 0); +} + +/* + * A lazily unmounted bind mount of an nsfs file, kept alive by an fd, can't + * be copied recursively anymore. Copying the mount itself still works and + * so does copying a mounted bind mount of the same file. + */ +TEST_F(unmounted_tree, detached_nsfs_bind_not_walked) +{ + char x[PATH_LEN], y[PATH_LEN], z[PATH_LEN], link[PATH_LEN]; + int fd, tfd; + + snprintf(x, sizeof(x), "%s/x", self->dir); + snprintf(y, sizeof(y), "%s/y", self->dir); + snprintf(z, sizeof(z), "%s/z", self->dir); + ASSERT_EQ(touch(x), 0); + ASSERT_EQ(touch(y), 0); + ASSERT_EQ(touch(z), 0); + + ASSERT_EQ(mount("/proc/self/ns/net", x, NULL, MS_BIND, NULL), 0); + fd = open(x, O_PATH | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(umount2(x, MNT_DETACH), 0); + + assert_not_walked(_metadata, fd, y); + + snprintf(link, sizeof(link), "/proc/self/fd/%d", fd); + ASSERT_EQ(mount(link, y, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(umount2(y, MNT_DETACH), 0); + assert_root_copied(_metadata, fd, y); + /* opening it is fine too */ + tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | OPEN_TREE_CLOEXEC); + ASSERT_GE(tfd, 0); + close(tfd); + close(fd); + + ASSERT_EQ(mount("/proc/self/ns/net", y, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(y, z, NULL, MS_BIND | MS_REC, NULL), 0); + tfd = sys_open_tree(AT_FDCWD, y, AT_RECURSIVE | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + ASSERT_GE(tfd, 0); + close(tfd); +} + +/* Bind-mount the pidfd @pidfd on @target the way pidfd_bind_mount does. */ +static int bind_pidfd(int pidfd, const char *target) +{ + int tfd, ret; + + tfd = sys_open_tree(pidfd, "", AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + if (tfd < 0) + return -1; + ret = sys_move_mount(tfd, "", AT_FDCWD, target, MOVE_MOUNT_F_EMPTY_PATH); + close(tfd); + return ret; +} + +/* The same for a bind mount of a pidfd. */ +TEST_F(unmounted_tree, detached_pidfs_bind_not_walked) +{ + char x[PATH_LEN], y[PATH_LEN]; + int pidfd, fd, tfd; + + snprintf(x, sizeof(x), "%s/x", self->dir); + snprintf(y, sizeof(y), "%s/y", self->dir); + ASSERT_EQ(touch(x), 0); + ASSERT_EQ(touch(y), 0); + + pidfd = sys_pidfd_open(getpid(), 0); + ASSERT_GE(pidfd, 0); + ASSERT_EQ(bind_pidfd(pidfd, x), 0); + fd = open(x, O_PATH | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_EQ(umount2(x, MNT_DETACH), 0); + + assert_not_walked(_metadata, fd, y); + assert_root_copied(_metadata, fd, y); + close(fd); + + ASSERT_EQ(bind_pidfd(pidfd, y), 0); + tfd = sys_open_tree(AT_FDCWD, y, AT_RECURSIVE | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + ASSERT_GE(tfd, 0); + close(tfd); + close(pidfd); +} + +/* What a child in its own mount namespace reports back. */ +enum { + CHILD_OK, + CHILD_NS, /* could not set up the mount namespace */ + CHILD_MOUNT, /* could not set up the mounts */ + CHILD_PIPE, /* the parent went away */ + CHILD_WALKED, /* the detached tree was copied recursively */ + CHILD_ROOT_REFUSED, /* the mount at its root wasn't copied */ + CHILD_STILL_MOUNTED, /* the copy survived the unlink of its mountpoint */ + CHILD_ERRNO, /* refused, but not with EINVAL */ +}; + +static const char *child_reason(int code) +{ + static const char *const reasons[] = { + [CHILD_OK] = "ok", + [CHILD_NS] = "could not set up the mount namespace", + [CHILD_MOUNT] = "could not set up the mounts", + [CHILD_PIPE] = "the parent went away", + [CHILD_WALKED] = "the detached tree was copied recursively", + [CHILD_ROOT_REFUSED] = "the mount at its root wasn't copied", + [CHILD_STILL_MOUNTED] = "the copy survived the unlink of its mountpoint", + [CHILD_ERRNO] = "refused, but not with EINVAL", + }; + + if (code < 0 || code >= (int)(sizeof(reasons) / sizeof(reasons[0]))) + return "child died"; + return reasons[code]; +} + +struct child_args { + const char *x; /* the nsfs bind mount goes here */ + const char *s; /* a file to bind on top of it */ + const char *y; /* a target to copy to */ +}; + +/* + * Run @fn in a child in a mount namespace of its own. Once the child + * reports that its mounts are in place, unlink @victim here, where it is a + * plain file, and let the child carry on. Returns what the child reported. + */ +static int run_child(struct __test_metadata *_metadata, + int (*fn)(const struct child_args *, int, int), + const struct child_args *a, const char *victim) +{ + int to_parent[2], to_child[2], status; + pid_t pid; + char c; + + if (pipe(to_parent) || pipe(to_child)) + return -1; + pid = fork(); + if (pid < 0) + return -1; + if (pid == 0) { + close(to_parent[0]); + close(to_child[1]); + _exit(fn(a, to_parent[1], to_child[0])); + } + close(to_parent[1]); + close(to_child[0]); + if (read(to_parent[0], &c, 1) == 1) { + /* not a mountpoint in this namespace, so the file can go */ + EXPECT_EQ(unlink(victim), 0); + EXPECT_EQ(write(to_child[1], "", 1), 1); + } + close(to_parent[0]); + close(to_child[1]); + if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status)) + return -1; + return WEXITSTATUS(status); +} + +/* + * An nsfs bind mount on @x, an fd on it and a bind mount of @s on top. Once + * the parent has unlinked @x underneath, the first mount is detached and the + * second one, which stayed attached to it, has been vacated. The tree may + * not be walked, the mount at its root may still be copied. + */ +static int vacant_child(const struct child_args *a, int to_parent, int from_parent) +{ + char link[PATH_LEN]; + int fd, tfd; + char c; + + if (own_mntns()) + return CHILD_NS; + if (mount("/proc/self/ns/net", a->x, NULL, MS_BIND, NULL)) + return CHILD_MOUNT; + fd = open(a->x, O_PATH | O_CLOEXEC); + if (fd < 0 || mount(a->s, a->x, NULL, MS_BIND, NULL)) + return CHILD_MOUNT; + + if (write(to_parent, "", 1) != 1 || read(from_parent, &c, 1) != 1) + return CHILD_PIPE; + + tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | AT_RECURSIVE | OPEN_TREE_CLONE | + OPEN_TREE_CLOEXEC); + if (tfd >= 0) { + close(tfd); + return CHILD_WALKED; + } + if (errno != EINVAL) + return CHILD_ERRNO; + snprintf(link, sizeof(link), "/proc/self/fd/%d", fd); + if (!mount(link, a->y, NULL, MS_BIND | MS_REC, NULL)) + return CHILD_WALKED; + if (errno != EINVAL) + return CHILD_ERRNO; + + tfd = sys_open_tree(fd, "", AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + if (tfd < 0) + return CHILD_ROOT_REFUSED; + close(tfd); + /* the last reference on the detached mount collects the vacant one */ + close(fd); + return CHILD_OK; +} + +/* + * unlink() of the child's mountpoint from here, where it is a plain file, + * detaches the child's nsfs bind mount with the mount on top left attached + * to it. The mount on top loses its last reference right there and is + * vacated. The child must not be able to walk that tree. + */ +TEST_F(unmounted_tree, vacant_child_not_copied) +{ + char x[PATH_LEN], s[PATH_LEN], y[PATH_LEN]; + struct child_args a = { .x = x, .s = s, .y = y }; + int ret; + + snprintf(x, sizeof(x), "%s/x", self->dir); + snprintf(s, sizeof(s), "%s/s", self->dir); + snprintf(y, sizeof(y), "%s/y", self->dir); + ASSERT_EQ(touch(x), 0); + ASSERT_EQ(touch(s), 0); + ASSERT_EQ(touch(y), 0); + + ret = run_child(_metadata, vacant_child, &a, x); + ASSERT_EQ(ret, CHILD_OK) + TH_LOG("child: %s", child_reason(ret)); +} + +/* + * An nsfs bind mount on @x, lazily unmounted but kept alive by an fd, and a + * copy of it on @y. That copy is a mount like any other: once the parent has + * unlinked @y underneath, it is unmounted and its attributes can't be changed + * anymore. A copy that took the flag along is only unhooked, stays in the + * namespace and still takes the change. + */ +static int ordinary_copy(const struct child_args *a, int to_parent, int from_parent) +{ + struct mount_attr attr = { + .attr_set = MOUNT_ATTR_NOSUID, + }; + char link[PATH_LEN]; + int fd, fdy; + char c; + + if (own_mntns()) + return CHILD_NS; + if (mount("/proc/self/ns/net", a->x, NULL, MS_BIND, NULL)) + return CHILD_MOUNT; + fd = open(a->x, O_PATH | O_CLOEXEC); + if (fd < 0 || umount2(a->x, MNT_DETACH)) + return CHILD_MOUNT; + snprintf(link, sizeof(link), "/proc/self/fd/%d", fd); + if (mount(link, a->y, NULL, MS_BIND, NULL)) + return CHILD_MOUNT; + fdy = open(a->y, O_PATH | O_CLOEXEC); + if (fdy < 0) + return CHILD_MOUNT; + + if (write(to_parent, "", 1) != 1 || read(from_parent, &c, 1) != 1) + return CHILD_PIPE; + + if (!sys_mount_setattr(fdy, "", AT_EMPTY_PATH, &attr, sizeof(attr))) + return CHILD_STILL_MOUNTED; + if (errno != EINVAL) + return CHILD_ERRNO; + close(fdy); + close(fd); + return CHILD_OK; +} + +/* + * A copy of a detached nsfs bind mount must not take after its source and + * read as unmounted. unlink() of its mountpoint from here unmounts it like + * any other mount, so the child can't change its attributes anymore. + */ +TEST_F(unmounted_tree, copy_of_detached_bind_is_ordinary) +{ + char x[PATH_LEN], y[PATH_LEN]; + struct child_args a = { .x = x, .y = y }; + int ret; + + snprintf(x, sizeof(x), "%s/x", self->dir); + snprintf(y, sizeof(y), "%s/y", self->dir); + ASSERT_EQ(touch(x), 0); + ASSERT_EQ(touch(y), 0); + + ret = run_child(_metadata, ordinary_copy, &a, y); + ASSERT_EQ(ret, CHILD_OK) + TH_LOG("child: %s", child_reason(ret)); +} + +/* + * mount_setattr() may change the propagation of a detached tree while that + * tree is still the root of an anonymous mount namespace, but not once the + * tree has been dissolved and only a second fd keeps its root alive. + */ +TEST_F(unmounted_tree, detached_tree_setattr_refused) +{ + struct mount_attr attr = { + .propagation = MS_SHARED, + }; + char a[PATH_LEN], b[PATH_LEN]; + int tfd, fd; + + snprintf(a, sizeof(a), "%s/a", self->dir); + snprintf(b, sizeof(b), "%s/a/b", self->dir); + ASSERT_EQ(mkdir(a, 0755), 0); + ASSERT_EQ(mount("tmpfs", a, "tmpfs", 0, NULL), 0); + ASSERT_EQ(mkdir(b, 0755), 0); + ASSERT_EQ(mount("tmpfs", b, "tmpfs", 0, NULL), 0); + + tfd = sys_open_tree(AT_FDCWD, self->dir, AT_RECURSIVE | OPEN_TREE_CLONE | + OPEN_TREE_CLOEXEC); + ASSERT_GE(tfd, 0); + ASSERT_EQ(sys_mount_setattr(tfd, "", AT_EMPTY_PATH | AT_RECURSIVE, &attr, sizeof(attr)), 0); + attr.propagation = MS_PRIVATE; + ASSERT_EQ(sys_mount_setattr(tfd, "", AT_EMPTY_PATH | AT_RECURSIVE, &attr, sizeof(attr)), 0); + + /* a second reference on the root, then dissolve the tree */ + fd = sys_open_tree(tfd, "", AT_EMPTY_PATH | OPEN_TREE_CLOEXEC); + ASSERT_GE(fd, 0); + close(tfd); + + attr.propagation = MS_SHARED; + ASSERT_EQ(sys_mount_setattr(fd, "", AT_EMPTY_PATH | AT_RECURSIVE, &attr, sizeof(attr)), -1); + ASSERT_EQ(errno, EINVAL); + attr.propagation = MS_PRIVATE; + ASSERT_EQ(sys_mount_setattr(fd, "", AT_EMPTY_PATH, &attr, sizeof(attr)), -1); + ASSERT_EQ(errno, EINVAL); + close(fd); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/open_tree_ns/.gitignore b/tools/testing/selftests/filesystems/open_tree_ns/.gitignore index fb12b93fbcaa..76f95c0ae5ef 100644 --- a/tools/testing/selftests/filesystems/open_tree_ns/.gitignore +++ b/tools/testing/selftests/filesystems/open_tree_ns/.gitignore @@ -1 +1,2 @@ open_tree_ns_test +open_tree_ns_covered_test diff --git a/tools/testing/selftests/filesystems/open_tree_ns/Makefile b/tools/testing/selftests/filesystems/open_tree_ns/Makefile index 4976ed1d7d4a..fb2aa77b6edb 100644 --- a/tools/testing/selftests/filesystems/open_tree_ns/Makefile +++ b/tools/testing/selftests/filesystems/open_tree_ns/Makefile @@ -1,5 +1,5 @@ # SPDX-License-Identifier: GPL-2.0 -TEST_GEN_PROGS := open_tree_ns_test +TEST_GEN_PROGS := open_tree_ns_test open_tree_ns_covered_test CFLAGS += -Wall -O0 -g $(KHDR_INCLUDES) $(TOOLS_INCLUDES) LDLIBS := -lcap diff --git a/tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c b/tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c new file mode 100644 index 000000000000..27f5cddfb9ef --- /dev/null +++ b/tools/testing/selftests/filesystems/open_tree_ns/open_tree_ns_covered_test.c @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * open_tree(OPEN_TREE_NAMESPACE) by a caller that isn't privileged over the + * mount namespace it copies from must not reveal what the mounts below the + * copied mount cover. Without AT_RECURSIVE the copy is refused when there's + * anything mounted below the requested directory. With AT_RECURSIVE + * unbindable mounts are copied as well. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/wait.h> + +#include "../wrappers.h" +#include "../../kselftest_harness.h" + +#ifndef OPEN_TREE_NAMESPACE +#define OPEN_TREE_NAMESPACE (1 << 1) +#endif + +#define DIR_LEN 64 +#define PATH_LEN 128 + +/* Child exit codes. */ +enum { + CHILD_OK, + CHILD_USERNS, /* could not create the child user namespace */ + CHILD_NONREC_ALLOWED, /* the non-recursive copy was not refused */ + CHILD_NONREC_ERRNO, /* it was refused with the wrong error */ + CHILD_REC_REFUSED, /* the recursive copy failed */ + CHILD_SETNS, /* could not enter the new mount namespace */ + CHILD_NO_COVER, /* the covering mount is missing in the copy */ + CHILD_REVEALED, /* the covered file is visible in the copy */ +}; + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +static int create_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0644); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +/* Become root in a new user namespace, the uid @uid is mapped to 0. */ +static int enter_userns(uid_t uid, gid_t gid) +{ + char map[32]; + + if (unshare(CLONE_NEWUSER)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + return setgid(0) || setuid(0) ? -1 : 0; +} + +FIXTURE(open_tree_ns_covered) { + char dir[DIR_LEN]; + char cover[PATH_LEN]; +}; + +FIXTURE_VARIANT(open_tree_ns_covered) { + int propagation; +}; + +FIXTURE_VARIANT_ADD(open_tree_ns_covered, private_cover) { + .propagation = MS_PRIVATE, +}; + +FIXTURE_VARIANT_ADD(open_tree_ns_covered, unbindable_cover) { + .propagation = MS_UNBINDABLE, +}; + +/* + * Root in a user namespace owns a private mount namespace with a tmpfs + * on @dir and a second tmpfs covering @dir/covered/under.txt. + */ +FIXTURE_SETUP(open_tree_ns_covered) +{ + char p[PATH_LEN]; + int fd; + + snprintf(self->dir, sizeof(self->dir), "/tmp/open_tree_ns_covered.XXXXXX"); + ASSERT_NE(mkdtemp(self->dir), NULL); + if (enter_userns(getuid(), getgid()) || unshare(CLONE_NEWNS)) { + rmdir(self->dir); + SKIP(return, "test requires user namespaces"); + } + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + ASSERT_EQ(mount("tmpfs", self->dir, "tmpfs", 0, NULL), 0); + + /* the flag has to exist before its rules can be checked */ + fd = sys_open_tree(AT_FDCWD, self->dir, + OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC); + if (fd < 0 && (errno == EINVAL || errno == ENOSYS)) { + umount2(self->dir, MNT_DETACH); + rmdir(self->dir); + SKIP(return, "OPEN_TREE_NAMESPACE not supported"); + } + ASSERT_GE(fd, 0); + close(fd); + + snprintf(self->cover, sizeof(self->cover), "%s/covered", self->dir); + ASSERT_EQ(mkdir(self->cover, 0755), 0); + snprintf(p, sizeof(p), "%s/covered/under.txt", self->dir); + ASSERT_EQ(create_file(p, "hidden"), 0); + ASSERT_EQ(mount("tmpfs", self->cover, "tmpfs", 0, NULL), 0); + ASSERT_EQ(mount(NULL, self->cover, NULL, variant->propagation, NULL), 0); +} + +FIXTURE_TEARDOWN(open_tree_ns_covered) +{ + umount2(self->dir, MNT_DETACH); + rmdir(self->dir); +} + +/* A caller in a new user namespace that doesn't own the mount namespace. */ +static int foreign_child(const char *dir) +{ + struct stat st; + int fd; + + if (enter_userns(0, 0)) + return CHILD_USERNS; + + fd = sys_open_tree(AT_FDCWD, dir, OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC); + if (fd >= 0) + return CHILD_NONREC_ALLOWED; + if (errno != EINVAL) + return CHILD_NONREC_ERRNO; + + fd = sys_open_tree(AT_FDCWD, dir, + OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC | AT_RECURSIVE); + if (fd < 0) + return CHILD_REC_REFUSED; + if (setns(fd, CLONE_NEWNS)) + return CHILD_SETNS; + if (stat("/covered", &st)) + return CHILD_NO_COVER; + if (!access("/covered/under.txt", F_OK) || errno != ENOENT) + return CHILD_REVEALED; + return CHILD_OK; +} + +TEST_F(open_tree_ns_covered, foreign_user_namespace) +{ + int status, fd; + pid_t pid; + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(foreign_child(self->dir)); + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFEXITED(status)); + ASSERT_EQ(WEXITSTATUS(status), CHILD_OK); + + /* the owner of the mount namespace keeps bind mount semantics */ + fd = sys_open_tree(AT_FDCWD, self->dir, + OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC); + ASSERT_GE(fd, 0); + close(fd); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/overlayfs/.gitignore b/tools/testing/selftests/filesystems/overlayfs/.gitignore index 077f7a128168..b343cc430051 100644 --- a/tools/testing/selftests/filesystems/overlayfs/.gitignore +++ b/tools/testing/selftests/filesystems/overlayfs/.gitignore @@ -2,3 +2,4 @@ dev_in_maps set_layers_via_fds idmapped_mounts +automount_in_layer diff --git a/tools/testing/selftests/filesystems/overlayfs/Makefile b/tools/testing/selftests/filesystems/overlayfs/Makefile index b3185f684add..382a59bda5e7 100644 --- a/tools/testing/selftests/filesystems/overlayfs/Makefile +++ b/tools/testing/selftests/filesystems/overlayfs/Makefile @@ -9,6 +9,7 @@ LOCAL_HDRS += ../wrappers.h log.h TEST_GEN_PROGS := dev_in_maps TEST_GEN_PROGS += set_layers_via_fds TEST_GEN_PROGS += idmapped_mounts +TEST_GEN_PROGS += automount_in_layer include ../../lib.mk diff --git a/tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c b/tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c new file mode 100644 index 000000000000..0476c4582bdf --- /dev/null +++ b/tools/testing/selftests/filesystems/overlayfs/automount_in_layer.c @@ -0,0 +1,174 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A layer of an overlay is a private clone of the mount it was given and + * belongs to no mount namespace. fanotify hands out descriptors on it. An + * automount triggered through one has no namespace to go into: the open has + * to fail, not oops with namespace_sem held. + */ +#define _GNU_SOURCE +#include <dirent.h> +#include <errno.h> +#include <fcntl.h> +#include <poll.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <linux/magic.h> +#include <sys/fanotify.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/vfs.h> + +#include "../../kselftest_harness.h" + +#define DIR_LEN 64 +#define PATH_LEN 192 + +static bool have_fs(const char *name) +{ + char line[128]; + bool found = false; + FILE *f; + + f = fopen("/proc/filesystems", "re"); + if (!f) + return false; + while (fgets(line, sizeof(line), f)) { + char *nl = strchr(line, '\n'); + char *tab = strchr(line, '\t'); + + if (nl) + *nl = 0; + if (tab && !strcmp(tab + 1, name)) + found = true; + } + fclose(f); + return found; +} + +static int mnt_id_of(int fd) +{ + char path[64], buf[4096], *p; + ssize_t n; + int info; + + snprintf(path, sizeof(path), "/proc/self/fdinfo/%d", fd); + info = open(path, O_RDONLY | O_CLOEXEC); + if (info < 0) + return -1; + n = read(info, buf, sizeof(buf) - 1); + close(info); + if (n <= 0) + return -1; + buf[n] = 0; + p = strstr(buf, "mnt_id:"); + return p ? atoi(p + strlen("mnt_id:")) : -1; +} + +static bool on_debugfs(int fd) +{ + struct statfs sf; + + return !fstatfs(fd, &sf) && sf.f_type == DEBUGFS_MAGIC; +} + +FIXTURE(layer) { + char base[DIR_LEN]; + int fan; + int evfd; /* on the layer clone of the lower debugfs */ +}; + +FIXTURE_SETUP(layer) +{ + char lower[PATH_LEN], other[PATH_LEN], ovl[PATH_LEN], opts[2 * PATH_LEN + 16]; + struct fanotify_event_metadata *ev; + struct pollfd pfd; + char buf[4096]; + int lfd, lower_id; + ssize_t n; + DIR *d; + + self->fan = -1; + self->evfd = -1; + if (geteuid()) + SKIP(return, "test requires root"); + if (!have_fs("debugfs") || !have_fs("tracefs")) + SKIP(return, "test requires debugfs with the tracefs automount"); + if (!have_fs("overlay")) + SKIP(return, "test requires overlayfs"); + + snprintf(self->base, sizeof(self->base), "/tmp/layer.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, "mode=0755"), 0); + snprintf(lower, sizeof(lower), "%s/lower", self->base); + snprintf(other, sizeof(other), "%s/other", self->base); + snprintf(ovl, sizeof(ovl), "%s/ovl", self->base); + ASSERT_EQ(mkdir(lower, 0755), 0); + ASSERT_EQ(mkdir(other, 0755), 0); + ASSERT_EQ(mkdir(ovl, 0755), 0); + ASSERT_EQ(mount("debugfs", lower, "debugfs", 0, NULL), 0); + snprintf(opts, sizeof(opts), "lowerdir=%s:%s", lower, other); + ASSERT_EQ(mount("overlay", ovl, "overlay", MS_RDONLY, opts), 0); + + self->fan = fanotify_init(FAN_CLASS_NOTIF | FAN_NONBLOCK | FAN_CLOEXEC, + O_RDONLY | O_CLOEXEC); + ASSERT_GE(self->fan, 0); + ASSERT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD | FAN_MARK_FILESYSTEM, + FAN_OPEN | FAN_ONDIR, AT_FDCWD, lower), 0); + lfd = open(lower, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(lfd, 0); + lower_id = mnt_id_of(lfd); + close(lfd); + + /* the overlay opens its lower directory through the layer clone */ + d = opendir(ovl); + ASSERT_NE(d, NULL); + closedir(d); + pfd.fd = self->fan; + pfd.events = POLLIN; + ASSERT_EQ(poll(&pfd, 1, 5000), 1); + n = read(self->fan, buf, sizeof(buf)); + ASSERT_GT(n, 0); + for (ev = (void *)buf; FAN_EVENT_OK(ev, n); ev = FAN_EVENT_NEXT(ev, n)) { + if (ev->fd < 0) + continue; + if (self->evfd < 0 && on_debugfs(ev->fd) && + mnt_id_of(ev->fd) != lower_id) + self->evfd = ev->fd; + else + close(ev->fd); + } + ASSERT_GE(self->evfd, 0) + TH_LOG("no event on the layer clone"); +} + +FIXTURE_TEARDOWN(layer) +{ + if (self->evfd >= 0) + close(self->evfd); + if (self->fan >= 0) + close(self->fan); + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +TEST_F(layer, automount_below_the_clone_is_refused) +{ + struct stat st; + int fd; + + /* the automount point is there */ + ASSERT_EQ(fstatat(self->evfd, "tracing", &st, AT_NO_AUTOMOUNT), 0); + /* the clone is in no namespace, so the automount has nowhere to go */ + fd = openat(self->evfd, "tracing", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + EXPECT_LT(fd, 0); + EXPECT_EQ(errno, EINVAL); + if (fd >= 0) + close(fd); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/readdir_hold.h b/tools/testing/selftests/filesystems/readdir_hold.h new file mode 100644 index 000000000000..57eee1bef470 --- /dev/null +++ b/tools/testing/selftests/filesystems/readdir_hold.h @@ -0,0 +1,224 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * Hold a readdir of a directory in the page fault of its buffer and see + * whether a create and a lookup in that directory wait for it. For a + * directory that is permanently empty they must not. + */ +#ifndef __SELFTESTS_READDIR_HOLD_H +#define __SELFTESTS_READDIR_HOLD_H + +#include <errno.h> +#include <fcntl.h> +#include <poll.h> +#include <pthread.h> +#include <stdbool.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <linux/userfaultfd.h> +#include <sys/ioctl.h> +#include <sys/mman.h> +#include <sys/stat.h> +#include <sys/syscall.h> + +#define HOLD_FAULT_MS 5000 /* for the reader to reach the fault */ +#define HOLD_QUEUE_MS 2000 /* for the create to queue up behind it */ +#define HOLD_LOOKUP_MS 5000 /* for the lookup to come back */ + +struct readdir_hold { + int uffd; + int taskdir; /* /proc/self/task, from before any namespace change */ + long page_size; + char *page; /* faults until released */ + int dfd; + pid_t creator_tid; + int created; + int done[2]; /* the finder writes a byte when it is back */ +}; + +/* + * The fault happens in the kernel, so the userfaultfd needs CAP_SYS_PTRACE + * in the initial user namespace or vm.unprivileged_userfaultfd. Call this + * before entering a user namespace. + */ +static inline int readdir_hold_init(struct readdir_hold *h) +{ + struct uffdio_api api = { .api = UFFD_API }; + + h->page = MAP_FAILED; + h->page_size = sysconf(_SC_PAGESIZE); + h->taskdir = open("/proc/self/task", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + h->uffd = syscall(__NR_userfaultfd, O_CLOEXEC | O_NONBLOCK); + if (h->uffd < 0 || h->taskdir < 0 || ioctl(h->uffd, UFFDIO_API, &api)) { + if (h->uffd >= 0) + close(h->uffd); + if (h->taskdir >= 0) + close(h->taskdir); + h->uffd = h->taskdir = -1; + return -1; + } + return 0; +} + +static inline void readdir_hold_destroy(struct readdir_hold *h) +{ + if (h->uffd >= 0) + close(h->uffd); + if (h->taskdir >= 0) + close(h->taskdir); + h->uffd = h->taskdir = -1; +} + +static inline void *readdir_hold_reader(void *arg) +{ + struct readdir_hold *h = arg; + + /* the first byte written to the buffer faults until released */ + syscall(__NR_getdents64, h->dfd, h->page, h->page_size); + return NULL; +} + +static inline void *readdir_hold_creator(void *arg) +{ + struct readdir_hold *h = arg; + + h->creator_tid = syscall(__NR_gettid); + /* takes the directory lock exclusive before it fails */ + mkdirat(h->dfd, "x", 0755); + __atomic_store_n(&h->created, 1, __ATOMIC_RELEASE); + return NULL; +} + +static inline void *readdir_hold_finder(void *arg) +{ + struct readdir_hold *h = arg; + int fd; + + /* a lookup that misses the dcache takes the lock shared */ + fd = openat(h->dfd, "no_such_name", O_RDONLY | O_CLOEXEC); + if (fd >= 0) + close(fd); + if (write(h->done[1], "x", 1) != 1) + perror("readdir_hold: finder"); + return NULL; +} + +/* the creator is back, or waits in the kernel for the lock */ +static inline bool readdir_hold_creator_settled(struct readdir_hold *h) +{ + char path[32], buf[256], *p; + ssize_t n; + int fd; + + if (__atomic_load_n(&h->created, __ATOMIC_ACQUIRE)) + return true; + if (!h->creator_tid) + return false; + snprintf(path, sizeof(path), "%d/stat", h->creator_tid); + fd = openat(h->taskdir, path, O_RDONLY | O_CLOEXEC); + if (fd < 0) + return false; + n = read(fd, buf, sizeof(buf) - 1); + close(fd); + if (n <= 0) + return false; + buf[n] = 0; + /* "pid (comm) state ..." */ + p = strrchr(buf, ')'); + return p && p[1] == ' ' && p[2] == 'D'; +} + +static inline bool readdir_hold_faulted(struct readdir_hold *h) +{ + struct pollfd pfd = { .fd = h->uffd, .events = POLLIN }; + struct uffd_msg msg; + + if (poll(&pfd, 1, HOLD_FAULT_MS) != 1) + return false; + if (read(h->uffd, &msg, sizeof(msg)) != sizeof(msg)) + return false; + return msg.event == UFFD_EVENT_PAGEFAULT; +} + +/* let the reader go on */ +static inline void readdir_hold_release(struct readdir_hold *h) +{ + struct uffdio_copy cp = { + .dst = (unsigned long)h->page, + .len = h->page_size, + }; + void *zero; + + zero = calloc(1, h->page_size); + if (!zero) + return; + cp.src = (unsigned long)zero; + if (ioctl(h->uffd, UFFDIO_COPY, &cp) && errno != EEXIST) + perror("readdir_hold: UFFDIO_COPY"); + free(zero); +} + +/* + * A readdir of @dfd that sticks in the fault of its buffer, then a create + * and a lookup in @dfd. Whether the lookup came back while the readdir was + * still stuck goes to @stalled. Returns -1 when that could not be found + * out. + */ +static inline int readdir_hold_check(struct readdir_hold *h, int dfd, + bool *stalled) +{ + pthread_t reader, creator, finder; + struct uffdio_register reg = {}; + struct pollfd pfd; + int ret = -1, i; + + h->dfd = dfd; + h->created = 0; + h->creator_tid = 0; + h->page = mmap(NULL, h->page_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (h->page == MAP_FAILED) + return -1; + reg.range.start = (unsigned long)h->page; + reg.range.len = h->page_size; + reg.mode = UFFDIO_REGISTER_MODE_MISSING; + if (ioctl(h->uffd, UFFDIO_REGISTER, ®) || pipe2(h->done, O_CLOEXEC)) + goto out_page; + + if (pthread_create(&reader, NULL, readdir_hold_reader, h)) + goto out_pipe; + if (!readdir_hold_faulted(h)) + goto out_reader; + if (pthread_create(&creator, NULL, readdir_hold_creator, h)) + goto out_reader; + for (i = 0; i < HOLD_QUEUE_MS / 10 && !readdir_hold_creator_settled(h); i++) + usleep(10000); + if (pthread_create(&finder, NULL, readdir_hold_finder, h)) + goto out_creator; + + pfd.fd = h->done[0]; + pfd.events = POLLIN; + *stalled = poll(&pfd, 1, HOLD_LOOKUP_MS) != 1; + ret = 0; + + readdir_hold_release(h); + pthread_join(finder, NULL); +out_creator: + if (ret) + readdir_hold_release(h); + pthread_join(creator, NULL); +out_reader: + if (ret) + readdir_hold_release(h); + pthread_join(reader, NULL); +out_pipe: + close(h->done[0]); + close(h->done[1]); +out_page: + munmap(h->page, h->page_size); + h->page = MAP_FAILED; + return ret; +} + +#endif /* __SELFTESTS_READDIR_HOLD_H */ diff --git a/tools/testing/selftests/filesystems/rw_hint_test.c b/tools/testing/selftests/filesystems/rw_hint_test.c new file mode 100644 index 000000000000..d1930f82f63b --- /dev/null +++ b/tools/testing/selftests/filesystems/rw_hint_test.c @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * F_SET_RW_HINT is refused on an immutable inode. Nothing is ever written + * to it and it may be shared with everybody, like a namespace file or the + * root of an empty mount namespace. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdint.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/ioctl.h> +#include <sys/stat.h> +#include <sys/wait.h> + +#include "../kselftest_harness.h" + +/* <linux/fs.h> and <linux/fcntl.h> don't mix with the libc headers */ +#ifndef FS_IOC_GETFLAGS +#define FS_IOC_GETFLAGS _IOR('f', 1, long) +#define FS_IOC_SETFLAGS _IOW('f', 2, long) +#endif +#ifndef FS_IMMUTABLE_FL +#define FS_IMMUTABLE_FL 0x00000010 +#endif +#ifndef F_LINUX_SPECIFIC_BASE +#define F_LINUX_SPECIFIC_BASE 1024 +#endif +#ifndef F_GET_RW_HINT +#define F_GET_RW_HINT (F_LINUX_SPECIFIC_BASE + 11) +#define F_SET_RW_HINT (F_LINUX_SPECIFIC_BASE + 12) +#endif +#ifndef RWH_WRITE_LIFE_SHORT +#define RWH_WRITE_LIFE_SHORT 2 +#endif +#ifndef UNSHARE_EMPTY_MNTNS +#define UNSHARE_EMPTY_MNTNS 0x00100000 +#endif + +static int set_hint(int fd, uint64_t hint) +{ + return fcntl(fd, F_SET_RW_HINT, &hint); +} + +static long get_hint(int fd) +{ + uint64_t hint; + + if (fcntl(fd, F_GET_RW_HINT, &hint)) + return -1; + return hint; +} + +TEST(immutable_file) +{ + char path[] = "/tmp/rw_hint.XXXXXX"; + int fd, flags; + + if (geteuid()) + SKIP(return, "test requires root"); + + fd = mkstemp(path); + ASSERT_GE(fd, 0); + unlink(path); + ASSERT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), 0); + EXPECT_EQ(get_hint(fd), RWH_WRITE_LIFE_SHORT); + + if (ioctl(fd, FS_IOC_GETFLAGS, &flags)) { + close(fd); + SKIP(return, "no file attributes on this filesystem"); + } + flags |= FS_IMMUTABLE_FL; + ASSERT_EQ(ioctl(fd, FS_IOC_SETFLAGS, &flags), 0); + EXPECT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), -1); + EXPECT_EQ(errno, EPERM); + flags &= ~FS_IMMUTABLE_FL; + ASSERT_EQ(ioctl(fd, FS_IOC_SETFLAGS, &flags), 0); + EXPECT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), 0); + close(fd); +} + +TEST(namespace_file) +{ + int fd; + + if (geteuid()) + SKIP(return, "test requires root"); + + fd = open("/proc/self/ns/mnt", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd, 0); + EXPECT_EQ(set_hint(fd, RWH_WRITE_LIFE_SHORT), -1); + EXPECT_EQ(errno, EPERM); + close(fd); +} + +TEST(empty_mntns_root) +{ + int status; + pid_t pid; + + if (geteuid()) + SKIP(return, "test requires root"); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + int fd; + + if (unshare(UNSHARE_EMPTY_MNTNS)) + _exit(errno == EINVAL ? 100 : 1); + fd = open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + if (fd < 0) + _exit(2); + if (set_hint(fd, RWH_WRITE_LIFE_SHORT) == 0) + _exit(3); + _exit(errno == EPERM ? 0 : 4); + } + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFEXITED(status)); + if (WEXITSTATUS(status) == 100) + SKIP(return, "UNSHARE_EMPTY_MNTNS not supported"); + EXPECT_EQ(WEXITSTATUS(status), 0); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/umount_propagation/Makefile b/tools/testing/selftests/filesystems/umount_propagation/Makefile index fc0a0783018b..31f783a92b23 100644 --- a/tools/testing/selftests/filesystems/umount_propagation/Makefile +++ b/tools/testing/selftests/filesystems/umount_propagation/Makefile @@ -1,5 +1,5 @@ # SPDX-License-Identifier: GPL-2.0 -TEST_GEN_PROGS := umount_propagation_test +TEST_GEN_PROGS := umount_propagation_test shrink_submounts_test locked_mount_test CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) diff --git a/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c b/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c new file mode 100644 index 000000000000..0799b2ba545d --- /dev/null +++ b/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c @@ -0,0 +1,432 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * MNT_LOCKED keeps the owner of a user namespace from revealing what a + * mount covers. The lock has to be set on the copy in that namespace and + * only there, it has to survive the expiry of a mount placed beneath the + * locked one, and it has to survive the propagated umount of such a copy. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mount.h> +#include <sys/prctl.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> + +#include "../../kselftest_harness.h" + +#ifndef MOVE_MOUNT_F_EMPTY_PATH +#define MOVE_MOUNT_F_EMPTY_PATH 0x00000004 +#endif +#ifndef MOVE_MOUNT_BENEATH +#define MOVE_MOUNT_BENEATH 0x00000200 +#endif + +#define DIR_LEN 64 +#define PATH_LEN 192 +#define FLAGS (MS_NOSUID | MS_NODEV | MS_NOEXEC) + +/* exit codes of the children, each test says what they mean */ +enum { + CHILD_OK, + CHILD_SETUP, + CHILD_STEP1, + CHILD_STEP2, + CHILD_STEP3, + CHILD_STEP4, +}; + +static int write_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +static int create_file(const char *path, const char *s) +{ + ssize_t n = -1; + int fd; + + fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644); + if (fd >= 0) { + n = write(fd, s, strlen(s)); + close(fd); + } + return n == (ssize_t)strlen(s) ? 0 : -1; +} + +/* Root in a new user namespace with a copy of the mount namespace. */ +static int enter_userns(void) +{ + uid_t uid = getuid(); + gid_t gid = getgid(); + char map[32]; + + prctl(PR_SET_DUMPABLE, 1); + if (unshare(CLONE_NEWUSER | CLONE_NEWNS)) + return -1; + if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT) + return -1; + snprintf(map, sizeof(map), "0 %d 1", uid); + if (write_file("/proc/self/uid_map", map)) + return -1; + snprintf(map, sizeof(map), "0 %d 1", gid); + if (write_file("/proc/self/gid_map", map)) + return -1; + return setgid(0) || setuid(0) ? -1 : 0; +} + +static int wait_child(pid_t pid) +{ + int status; + + if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status)) + return -1; + return WEXITSTATUS(status); +} + +static int wait_byte(int fd) +{ + char c; + + return read(fd, &c, 1) == 1 ? 0 : -1; +} + +static int send_byte(int fd) +{ + return write(fd, "x", 1) == 1 ? 0 : -1; +} + +/* A read of @path fails: the file is covered. */ +static bool covered(const char *path) +{ + int fd = open(path, O_RDONLY | O_CLOEXEC); + + if (fd >= 0) + close(fd); + return fd < 0; +} + +FIXTURE(locked_mount) { + char base[DIR_LEN]; + bool tracing; /* debugfs with the tracefs automount is there */ +}; + +FIXTURE_SETUP(locked_mount) +{ + char p[PATH_LEN]; + struct stat st; + + if (geteuid()) + SKIP(return, "test requires root"); + + snprintf(self->base, sizeof(self->base), "/tmp/locked_mount.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, "mode=0755"), 0); + + /* an automount that a user can name: the tracefs below debugfs */ + snprintf(p, sizeof(p), "%s/dbg", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + self->tracing = !mount("debugfs", p, "debugfs", FLAGS, NULL); + if (self->tracing) { + /* the cases trigger it themselves */ + snprintf(p, sizeof(p), "%s/dbg/tracing", self->base); + self->tracing = !fstatat(AT_FDCWD, p, &st, AT_NO_AUTOMOUNT) && + S_ISDIR(st.st_mode); + } +} + +FIXTURE_TEARDOWN(locked_mount) +{ + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +/* + * The child keeps a directory descriptor on the host's debugfs mount, moves + * to a user namespace of its own and triggers the automount through the + * descriptor. The mount goes below the host's mount and propagates into the + * child's copy. The child's copy has to be locked, the host's not. + */ +static int automount_child(const char *base, int dfd, int to_host, int from_host) +{ + char p[PATH_LEN]; + struct stat st; + + if (enter_userns()) + return CHILD_SETUP; + if (fstatat(dfd, "tracing/.", &st, 0)) + return CHILD_STEP1; + if (send_byte(to_host) || wait_byte(from_host)) + return CHILD_SETUP; + /* the flags of the copy are locked: EPERM */ + snprintf(p, sizeof(p), "%s/dbg/tracing", base); + if (!mount(NULL, p, NULL, MS_REMOUNT | MS_BIND, NULL) || errno != EPERM) + return CHILD_STEP2; + return CHILD_OK; +} + +TEST_F(locked_mount, automount_locked_in_the_triggering_namespace) +{ + int to_host[2], from_host[2], dfd, ret; + char p[PATH_LEN]; + pid_t pid; + + if (!self->tracing) + SKIP(return, "test requires debugfs with the tracefs automount"); + + snprintf(p, sizeof(p), "%s/dbg", self->base); + ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0); + dfd = open(p, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(dfd, 0); + ASSERT_EQ(pipe(to_host), 0); + ASSERT_EQ(pipe(from_host), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(automount_child(self->base, dfd, to_host[1], from_host[0])); + close(dfd); + ASSERT_EQ(wait_byte(to_host[0]), 0); + + /* the host's own mount isn't locked: the flags can go */ + snprintf(p, sizeof(p), "%s/dbg/tracing", self->base); + EXPECT_EQ(mount(NULL, p, NULL, MS_REMOUNT | MS_BIND, NULL), 0); + + ASSERT_EQ(send_byte(from_host[1]), 0); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + EXPECT_EQ(ret, CHILD_OK); +} + +/* + * A copy of a tree with a locked cover. The child puts a shrinkable mount + * beneath the cover, which hands the lock down, unmounts the cover and then + * asks for the umount of an unlocked ancestor. That expires the shrinkable + * mount on a kernel that doesn't look at the lock, and the covered + * directory is bare. + */ +static int expiry_child(const char *base) +{ + char srv[PATH_LEN], shr[PATH_LEN], x[PATH_LEN], c[PATH_LEN]; + char hidden[PATH_LEN], secret[PATH_LEN]; + + snprintf(srv, sizeof(srv), "%s/srv", base); + snprintf(shr, sizeof(shr), "%s/shr", base); + snprintf(x, sizeof(x), "%s/x", base); + snprintf(c, sizeof(c), "%s/c", base); + snprintf(hidden, sizeof(hidden), "%s/x/hidden", base); + snprintf(secret, sizeof(secret), "%s/x/hidden/secret", base); + + if (enter_userns()) + return CHILD_SETUP; + if (mount(srv, x, NULL, MS_BIND | MS_REC, NULL) || + mount(shr, c, NULL, MS_BIND, NULL)) + return CHILD_SETUP; + /* the cover is locked */ + if (!umount2(hidden, 0) || errno != EINVAL) + return CHILD_STEP1; + if (syscall(__NR_move_mount, AT_FDCWD, c, AT_FDCWD, hidden, MOVE_MOUNT_BENEATH)) + return CHILD_STEP2; + /* the lock moved down, the cover may go */ + if (umount2(hidden, 0)) + return CHILD_STEP3; + /* the holder of the lock may not, in any way */ + if (!umount2(hidden, 0) || errno != EINVAL) + return CHILD_STEP4; + if (chdir(x)) + return CHILD_SETUP; + umount2(x, 0); + return covered(secret) ? CHILD_OK : CHILD_STEP4; +} + +TEST_F(locked_mount, expiry_leaves_a_locked_mount_alone) +{ + char p[PATH_LEN], q[PATH_LEN]; + pid_t pid; + int ret; + + if (!self->tracing) + SKIP(return, "test requires debugfs with the tracefs automount"); + + snprintf(p, sizeof(p), "%s/dbg/tracing", self->base); + snprintf(q, sizeof(q), "%s/shr", self->base); + ASSERT_EQ(mkdir(q, 0755), 0); + /* a bind of an automount is shrinkable as well */ + ASSERT_EQ(mount(p, q, NULL, MS_BIND, NULL), 0); + snprintf(p, sizeof(p), "%s/srv", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0); + snprintf(p, sizeof(p), "%s/srv/hidden", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + snprintf(p, sizeof(p), "%s/srv/hidden/secret", self->base); + ASSERT_EQ(create_file(p, "covered-by-root\n"), 0); + snprintf(p, sizeof(p), "%s/srv/hidden", self->base); + ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0); + snprintf(p, sizeof(p), "%s/x", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + snprintf(p, sizeof(p), "%s/c", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(expiry_child(self->base)); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + EXPECT_EQ(ret, CHILD_OK); +} + +/* + * The copy of the mount tree that a user namespace gets at its creation has + * the automounts in it locked unless they are on an expiry list. A bind of + * such a tree inside the namespace copies the lock to the automount but not + * to the root of the bind, so the child may ask for the umount of that root. + * That must not expire the locked automount below it: the root is busy, the + * umount fails and the automount has to be there afterwards. + */ +static int copied_tree_child(const char *base) +{ + char dbg[PATH_LEN], x[PATH_LEN], tracing[PATH_LEN]; + struct stat root, st; + + snprintf(dbg, sizeof(dbg), "%s/dbg", base); + snprintf(x, sizeof(x), "%s/x", base); + snprintf(tracing, sizeof(tracing), "%s/x/tracing", base); + + if (enter_userns()) + return CHILD_SETUP; + if (mount(dbg, x, NULL, MS_BIND | MS_REC, NULL)) + return CHILD_SETUP; + /* the copy of the automount carries the lock */ + if (!umount2(tracing, 0) || errno != EINVAL) + return CHILD_STEP1; + /* the root of the bind doesn't; keep it busy */ + if (chdir(x)) + return CHILD_SETUP; + if (!umount2(x, 0) || errno != EBUSY) + return CHILD_STEP2; + /* the automount below it must not have gone */ + if (stat(x, &root) || fstatat(AT_FDCWD, tracing, &st, AT_NO_AUTOMOUNT)) + return CHILD_SETUP; + return st.st_dev != root.st_dev ? CHILD_OK : CHILD_STEP3; +} + +TEST_F(locked_mount, umount_of_a_bind_leaves_a_locked_automount_alone) +{ + char p[PATH_LEN]; + struct stat st; + pid_t pid; + int ret; + + if (!self->tracing) + SKIP(return, "test requires debugfs with the tracefs automount"); + + /* the copy the child gets has to contain the automount */ + snprintf(p, sizeof(p), "%s/dbg/tracing/.", self->base); + ASSERT_EQ(stat(p, &st), 0); + snprintf(p, sizeof(p), "%s/x", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + _exit(copied_tree_child(self->base)); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + EXPECT_EQ(ret, CHILD_OK); +} + +/* + * Z is a user namespace with a copy of a tree in which P/d is covered by a + * locked mount. The host mounts X on P/d, which propagates beneath Z's + * cover, and unmounts it again. Z's cover has to be locked afterwards as + * it was before. + */ +static int propagation_child(const char *base, int to_host, int from_host) +{ + char d[PATH_LEN], secret[PATH_LEN]; + + snprintf(d, sizeof(d), "%s/P/d", base); + snprintf(secret, sizeof(secret), "%s/P/d/secret", base); + + if (enter_userns()) + return CHILD_SETUP; + /* the cover is locked */ + if (!umount2(d, 0) || errno != EINVAL) + return CHILD_STEP1; + if (send_byte(to_host) || wait_byte(from_host)) + return CHILD_SETUP; + /* X came and went beneath it: still locked */ + if (!umount2(d, 0) || errno != EINVAL) + return CHILD_STEP2; + return covered(secret) ? CHILD_OK : CHILD_STEP3; +} + +TEST_F(locked_mount, propagated_copy_keeps_the_cover_locked) +{ + int to_host[2], from_host[2], ret; + char p[PATH_LEN]; + pid_t pid; + + snprintf(p, sizeof(p), "%s/P", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0); + ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0); + snprintf(p, sizeof(p), "%s/P/d", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + snprintf(p, sizeof(p), "%s/P/d/secret", self->base); + ASSERT_EQ(create_file(p, "covered-by-root\n"), 0); + ASSERT_EQ(pipe(to_host), 0); + ASSERT_EQ(pipe(from_host), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + /* M: a manager's namespace that covers P/d for Z */ + pid_t z; + + if (unshare(CLONE_NEWNS)) + _exit(CHILD_SETUP); + snprintf(p, sizeof(p), "%s/P", self->base); + if (mount(NULL, p, NULL, MS_SLAVE, NULL)) + _exit(CHILD_SETUP); + snprintf(p, sizeof(p), "%s/P/d", self->base); + if (mount("tmpfs", p, "tmpfs", 0, "mode=0755")) + _exit(CHILD_SETUP); + z = fork(); + if (z < 0) + _exit(CHILD_SETUP); + if (z == 0) + _exit(propagation_child(self->base, to_host[1], from_host[0])); + _exit(wait_child(z)); + } + ASSERT_EQ(wait_byte(to_host[0]), 0); + + /* the host mounts on P/d and unmounts again; both propagate */ + snprintf(p, sizeof(p), "%s/P/d", self->base); + ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0); + ASSERT_EQ(umount2(p, 0), 0); + + ASSERT_EQ(send_byte(from_host[1]), 0); + ret = wait_child(pid); + TH_LOG("child exit code %d", ret); + EXPECT_EQ(ret, CHILD_OK); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c b/tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c new file mode 100644 index 000000000000..b024ae3417ce --- /dev/null +++ b/tools/testing/selftests/filesystems/umount_propagation/shrink_submounts_test.c @@ -0,0 +1,211 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A synchronous umount first unmounts the shrinkable submounts of the + * victim that aren't busy. Every one of them has to be checked right + * before it is unmounted: unmounting one can slide a busy mount to where + * the propagated copy of the next one is looked up. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/fanotify.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/statfs.h> +#include <sys/vfs.h> +#include <linux/magic.h> + +#include "../../kselftest_harness.h" + +#ifndef FAN_REPORT_MNT +#define FAN_REPORT_MNT 0x00004000 +#endif +#ifndef FAN_MARK_MNTNS +#define FAN_MARK_MNTNS 0x00000110 +#endif +#ifndef FAN_MNT_ATTACH +#define FAN_MNT_ATTACH 0x01000000 +#endif +#ifndef FAN_MNT_DETACH +#define FAN_MNT_DETACH 0x02000000 +#endif + +#define DIR_LEN 64 +#define PATH_LEN 128 + +FIXTURE(shrink_submounts) { + char base[DIR_LEN]; + char automount[PATH_LEN]; + bool mounted; + int fan; +}; + +/* + * Shrinkable mounts come from an automount. The tracefs mount below debugfs + * is one and bind mounts inherit the flag. + */ +FIXTURE_SETUP(shrink_submounts) +{ + struct stat st; + char p[PATH_LEN]; + + self->mounted = false; + self->fan = -1; + + if (geteuid() != 0) + SKIP(return, "test requires CAP_SYS_ADMIN"); + + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + + snprintf(self->base, sizeof(self->base), "/tmp/shrink_submounts.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0); + self->mounted = true; + ASSERT_EQ(mount(NULL, self->base, NULL, MS_PRIVATE, NULL), 0); + + snprintf(p, sizeof(p), "%s/dbg", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + if (mount("debugfs", p, "debugfs", 0, NULL)) { + umount2(self->base, MNT_DETACH); + rmdir(self->base); + SKIP(return, "test requires debugfs"); + } + snprintf(self->automount, sizeof(self->automount), "%s/dbg/tracing", + self->base); + snprintf(p, sizeof(p), "%s/dbg/tracing/.", self->base); + if (stat(p, &st)) { + umount2(self->base, MNT_DETACH); + rmdir(self->base); + SKIP(return, "test requires the tracefs automount"); + } +} + +FIXTURE_TEARDOWN(shrink_submounts) +{ + if (self->fan >= 0) + close(self->fan); + chdir("/"); + if (self->mounted) + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +static bool mounted_tmpfs(const char *path) +{ + struct statfs st; + + return !statfs(path, &st) && st.f_type == TMPFS_MAGIC; +} + +/* + * P is a shared bind mount of the automount, P1 a slave of P that is shared + * in turn and P2 its peer. B, another bind mount of the automount, goes on + * P/options and propagates copies Bc1 and Bc2 onto P1/options and + * P2/options. R, a tmpfs and our working directory, sits on top of Bc2 which + * is made private first. P, with B on it, is moved to P1/instances. + * + * A synchronous umount of P1 unmounts the shrinkable submounts Bc1 and B + * first. Unmounting Bc1 takes Bc2 along and slides R to P2/options where + * the propagated copy of B is looked up next. R is busy, so B has to stay + * and the umount fails with EBUSY. + */ +TEST_F(shrink_submounts, busy_mount_moved_into_reach) +{ + char p[PATH_LEN], p1[PATH_LEN], p2[PATH_LEN], r[PATH_LEN], cwd[PATH_LEN]; + int nsfd; + + snprintf(p, sizeof(p), "%s/p", self->base); + snprintf(p1, sizeof(p1), "%s/p1", self->base); + snprintf(p2, sizeof(p2), "%s/p2", self->base); + ASSERT_EQ(mkdir(p, 0755), 0); + ASSERT_EQ(mkdir(p1, 0755), 0); + ASSERT_EQ(mkdir(p2, 0755), 0); + + /* watch the mount namespace so that the detached mounts get queued */ + self->fan = fanotify_init(FAN_REPORT_MNT, O_RDONLY); + if (self->fan >= 0) { + nsfd = open("/proc/self/ns/mnt", O_RDONLY | O_CLOEXEC); + ASSERT_GE(nsfd, 0); + EXPECT_EQ(fanotify_mark(self->fan, FAN_MARK_ADD | FAN_MARK_MNTNS, + FAN_MNT_ATTACH | FAN_MNT_DETACH, nsfd, NULL), 0); + close(nsfd); + } + + ASSERT_EQ(mount(self->automount, p, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0); + ASSERT_EQ(mount(p, p1, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, p1, NULL, MS_SLAVE, NULL), 0); + ASSERT_EQ(mount(NULL, p1, NULL, MS_SHARED, NULL), 0); + ASSERT_EQ(mount(p1, p2, NULL, MS_BIND, NULL), 0); + + /* B on P/options, copies on P1/options and P2/options */ + snprintf(r, sizeof(r), "%s/p/options", self->base); + ASSERT_EQ(mount(self->automount, r, NULL, MS_BIND, NULL), 0); + + /* R on top of Bc2 */ + snprintf(r, sizeof(r), "%s/p2/options", self->base); + ASSERT_EQ(mount(NULL, r, NULL, MS_PRIVATE, NULL), 0); + ASSERT_EQ(mount("R", r, "tmpfs", 0, NULL), 0); + ASSERT_EQ(chdir(r), 0); + + /* P, with B on it, below the victim */ + snprintf(cwd, sizeof(cwd), "%s/p1/instances", self->base); + ASSERT_EQ(mount(p, cwd, NULL, MS_MOVE, NULL), 0); + + ASSERT_TRUE(mounted_tmpfs(r)); + ASSERT_EQ(umount2(p1, 0), -1); + EXPECT_EQ(errno, EBUSY); + + /* R is still mounted and still our working directory */ + EXPECT_TRUE(mounted_tmpfs(r)); + ASSERT_NE(getcwd(cwd, sizeof(cwd)), NULL); + EXPECT_STREQ(cwd, r); +} + +/* + * T is shared and Q, a slave of T, has T moved into it, so Q receives + * propagation from its own child. M, another bind mount of the automount, is + * on T/options and that is the dentry T sits on in Q. Unmounting M makes T + * the propagated victim at that dentry in Q and takes T along. The shrink + * walk of V has just unmounted M and continues in the children of T. + */ +TEST_F(shrink_submounts, parent_goes_with_child) +{ + char v[PATH_LEN], t[PATH_LEN], q[PATH_LEN], p[PATH_LEN]; + struct stat before, after; + + snprintf(v, sizeof(v), "%s/v", self->base); + snprintf(t, sizeof(t), "%s/v/t", self->base); + snprintf(q, sizeof(q), "%s/v/q", self->base); + ASSERT_EQ(mkdir(v, 0755), 0); + ASSERT_EQ(mount("V", v, "tmpfs", 0, NULL), 0); + ASSERT_EQ(mount(NULL, v, NULL, MS_PRIVATE, NULL), 0); + ASSERT_EQ(stat(v, &before), 0); + ASSERT_EQ(mkdir(t, 0755), 0); + ASSERT_EQ(mkdir(q, 0755), 0); + + /* T shared, M on T/options before anything receives from T */ + ASSERT_EQ(mount(self->automount, t, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, t, NULL, MS_SHARED, NULL), 0); + snprintf(p, sizeof(p), "%s/v/t/options", self->base); + ASSERT_EQ(mount(self->automount, p, NULL, MS_BIND, NULL), 0); + + /* Q, a slave of T, and T moved into Q at the dentry M sits on */ + ASSERT_EQ(mount(t, q, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, q, NULL, MS_SLAVE, NULL), 0); + snprintf(p, sizeof(p), "%s/v/q/options", self->base); + ASSERT_EQ(mount(t, p, NULL, MS_MOVE, NULL), 0); + + /* M, T and then Q go, V is empty and can be unmounted */ + ASSERT_EQ(umount2(v, 0), 0); + ASSERT_EQ(stat(v, &after), 0); + EXPECT_NE(before.st_dev, after.st_dev); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/kselftest/runner.sh b/tools/testing/selftests/kselftest/runner.sh index 311811dc55a0..ee2c1f0403d9 100644 --- a/tools/testing/selftests/kselftest/runner.sh +++ b/tools/testing/selftests/kselftest/runner.sh @@ -38,8 +38,12 @@ tap_prefix() tap_timeout() { + # nommu doesn't support timeout command (missing fork(2)) + if [ "$NOMMU" = "1" ] ; then + echo "timeout isn't supported for NOMMU" + $1 # Make sure tests will time out if utility is available. - if [ -x /usr/bin/timeout ] ; then + elif [ -x /usr/bin/timeout ] ; then /usr/bin/timeout --foreground "$kselftest_timeout" \ /usr/bin/timeout "$kselftest_timeout" $1 else diff --git a/tools/testing/selftests/kvm/arm64/vgic_init.c b/tools/testing/selftests/kvm/arm64/vgic_init.c index 47e34b43afb2..5a30f3cb039b 100644 --- a/tools/testing/selftests/kvm/arm64/vgic_init.c +++ b/tools/testing/selftests/kvm/arm64/vgic_init.c @@ -5,6 +5,7 @@ * Copyright (C) 2020, Red Hat, Inc. */ #include <linux/kernel.h> +#include <linux/sizes.h> #include <sys/syscall.h> #include <asm/kvm.h> #include <asm/kvm_para.h> @@ -13,12 +14,21 @@ #include "test_util.h" #include "kvm_util.h" +#include "gic.h" #include "processor.h" #include "vgic.h" #include "gic_v3.h" #define NR_VCPUS 4 +#define REDIST_RETRY_REGION0_BASE GICR_BASE_GPA +#define REDIST_RETRY_REGION1_BASE \ + (REDIST_RETRY_REGION0_BASE + 2 * KVM_VGIC_V3_REDIST_SIZE) +#define REDIST_RETRY_DIST_BASE \ + (REDIST_RETRY_REGION1_BASE + KVM_VGIC_V3_REDIST_SIZE) +#define REDIST_RETRY_REGION2_BASE \ + (REDIST_RETRY_DIST_BASE + KVM_VGIC_V3_DIST_SIZE) + #define REG_OFFSET(vcpu, offset) (((u64)vcpu << 32) | offset) #define VGIC_DEV_IS_V2(_d) ((_d) == KVM_DEV_TYPE_ARM_VGIC_V2) @@ -65,6 +75,23 @@ static void guest_code(void) GUEST_DONE(); } +static void guest_check_redist_retry(void) +{ + unsigned int i; + + /* The first three redistributors span adjacent regions 0 and 1. */ + for (i = 0; i < NR_VCPUS; i++) { + u64 base = i < 3 ? REDIST_RETRY_REGION0_BASE + + i * KVM_VGIC_V3_REDIST_SIZE : + REDIST_RETRY_REGION2_BASE; + u64 typer = readq((void *)(unsigned long)(base + GICR_TYPER)); + + GUEST_ASSERT_EQ(GICR_TYPER_CPU_NUMBER(typer), i); + } + + GUEST_DONE(); +} + /* we don't want to assert on run execution, hence that helper */ static int run_vcpu(struct kvm_vcpu *vcpu) { @@ -73,6 +100,7 @@ static int run_vcpu(struct kvm_vcpu *vcpu) static struct vm_gic vm_gic_create_with_vcpus(u32 gic_dev_type, u32 nr_vcpus, + void *guest_code, struct kvm_vcpu *vcpus[]) { struct vm_gic v; @@ -338,7 +366,7 @@ static void test_vgic_then_vcpus(u32 gic_dev_type) struct vm_gic v; int ret, i; - v = vm_gic_create_with_vcpus(gic_dev_type, 1, vcpus); + v = vm_gic_create_with_vcpus(gic_dev_type, 1, guest_code, vcpus); subtest_dist_rdist(&v); @@ -359,7 +387,8 @@ static void test_vcpus_then_vgic(u32 gic_dev_type) struct vm_gic v; int ret; - v = vm_gic_create_with_vcpus(gic_dev_type, NR_VCPUS, vcpus); + v = vm_gic_create_with_vcpus(gic_dev_type, NR_VCPUS, guest_code, + vcpus); subtest_dist_rdist(&v); @@ -411,7 +440,8 @@ static void test_v3_new_redist_regions(void) u64 addr; int ret; - v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus); + v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, + guest_code, vcpus); subtest_v3_redist_regions(&v); kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_CTRL, KVM_DEV_ARM_VGIC_CTRL_INIT, NULL); @@ -422,7 +452,8 @@ static void test_v3_new_redist_regions(void) /* step2 */ - v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus); + v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, + guest_code, vcpus); subtest_v3_redist_regions(&v); addr = REDIST_REGION_ATTR_ADDR(1, 0x280000, 0, 2); @@ -436,7 +467,8 @@ static void test_v3_new_redist_regions(void) /* step 3 */ - v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus); + v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, + guest_code, vcpus); subtest_v3_redist_regions(&v); ret = __kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR, @@ -457,6 +489,70 @@ static void test_v3_new_redist_regions(void) vm_gic_destroy(&v); } +static void test_v3_redist_region_retry(void) +{ + struct kvm_vcpu *vcpus[NR_VCPUS]; + struct vm_gic v; + struct ucall uc; + u64 addr; + int ret; + + v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, + guest_check_redist_retry, vcpus); + + addr = REDIST_REGION_ATTR_ADDR(2, REDIST_RETRY_REGION0_BASE, 0, 0); + kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR, + KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, &addr); + + addr = REDIST_REGION_ATTR_ADDR(1, REDIST_RETRY_REGION1_BASE, 0, 1); + kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR, + KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, &addr); + + addr = REDIST_RETRY_DIST_BASE; + kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR, + KVM_VGIC_V3_ADDR_TYPE_DIST, &addr); + + addr = REDIST_REGION_ATTR_ADDR(1, REDIST_RETRY_DIST_BASE, 0, 2); + ret = __kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR, + KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, + &addr); + TEST_ASSERT(ret && errno == EINVAL, + "register redist region colliding with dist"); + + addr = REDIST_REGION_ATTR_ADDR(1, REDIST_RETRY_REGION2_BASE, 0, 2); + kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_ADDR, + KVM_VGIC_V3_ADDR_TYPE_REDIST_REGION, &addr); + + virt_map(v.vm, REDIST_RETRY_REGION0_BASE, REDIST_RETRY_REGION0_BASE, + vm_calc_num_guest_pages(v.vm->mode, + 3 * KVM_VGIC_V3_REDIST_SIZE)); + virt_map(v.vm, REDIST_RETRY_REGION2_BASE, REDIST_RETRY_REGION2_BASE, + vm_calc_num_guest_pages(v.vm->mode, + KVM_VGIC_V3_REDIST_SIZE)); + + kvm_device_attr_set(v.gic_fd, KVM_DEV_ARM_VGIC_GRP_CTRL, + KVM_DEV_ARM_VGIC_CTRL_INIT, NULL); + + vcpu_run(vcpus[0]); + switch (get_ucall(vcpus[0], &uc)) { + case UCALL_DONE: + break; + case UCALL_ABORT: + REPORT_GUEST_ASSERT(uc); + break; + case UCALL_NONE: + if (vcpus[0]->run->exit_reason == KVM_EXIT_MMIO) + TEST_FAIL("Unexpected MMIO exit at 0x%llx", + vcpus[0]->run->mmio.phys_addr); + fallthrough; + default: + TEST_FAIL("Unexpected ucall %lu, exit_reason %u", + uc.cmd, vcpus[0]->run->exit_reason); + } + + vm_gic_destroy(&v); +} + static void test_v3_typer_accesses(void) { struct vm_gic v; @@ -608,7 +704,8 @@ static void test_v3_redist_ipa_range_check_at_vcpu_run(void) int ret, i; u64 addr; - v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, 1, vcpus); + v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, 1, guest_code, + vcpus); /* Set space for 3 redists, we have 1 vcpu, so this succeeds. */ addr = max_phys_size - (3 * 2 * 0x10000); @@ -641,7 +738,8 @@ static void test_v3_its_region(void) u64 addr; int its_fd, ret; - v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus); + v = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, + guest_code, vcpus); its_fd = kvm_create_device(v.vm, KVM_DEV_TYPE_ARM_VGIC_ITS); addr = 0x401000; @@ -684,7 +782,8 @@ static void test_v3_nassgicap(void) u32 typer2; int ret; - vm = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, vcpus); + vm = vm_gic_create_with_vcpus(KVM_DEV_TYPE_ARM_VGIC_V3, NR_VCPUS, + guest_code, vcpus); kvm_device_attr_get(vm.gic_fd, KVM_DEV_ARM_VGIC_GRP_DIST_REGS, GICD_TYPER2, &typer2); has_nassgicap = typer2 & GICD_TYPER2_nASSGIcap; @@ -978,6 +1077,7 @@ void run_tests(u32 gic_dev_type) if (VGIC_DEV_IS_V3(gic_dev_type)) { test_v3_new_redist_regions(); + test_v3_redist_region_retry(); test_v3_typer_accesses(); test_v3_last_bit_redist_regions(); test_v3_last_bit_single_rdist(); diff --git a/tools/testing/selftests/mm/uffd-unit-tests.c b/tools/testing/selftests/mm/uffd-unit-tests.c index 580178630ede..97484c82f24e 100644 --- a/tools/testing/selftests/mm/uffd-unit-tests.c +++ b/tools/testing/selftests/mm/uffd-unit-tests.c @@ -2049,28 +2049,29 @@ static void uffd_move_swap_test_common(uffd_global_test_opts_t *gopts, bool rwp) { unsigned long page_size = gopts->page_size; - struct uffdio_move move = { }; - int pagemap_fd; + struct uffdio_move move = { + .dst = (unsigned long)gopts->area_dst, + .src = (unsigned long)gopts->area_src, + .len = page_size, + }; + int pagemap_fd = pagemap_open(); if (rwp) { if (uffd_register_rwp(gopts->uffd, gopts->area_src, page_size)) err("register src failure"); - } else if (uffd_register(gopts->uffd, gopts->area_src, page_size, - false, true, false)) { - err("register src failure"); - } - if (uffd_register(gopts->uffd, gopts->area_dst, page_size, - true, false, false)) - err("register dst failure"); - - if (rwp) rwprotect_range(gopts->uffd, (unsigned long)gopts->area_src, page_size, true); - else + } else { + if (uffd_register(gopts->uffd, gopts->area_src, page_size, + false, true, false)) + err("register src failure"); wp_range(gopts->uffd, (unsigned long)gopts->area_src, page_size, true); + } + if (uffd_register(gopts->uffd, gopts->area_dst, page_size, + true, false, false)) + err("register dst failure"); - pagemap_fd = pagemap_open(); if (madvise(gopts->area_src, page_size, MADV_PAGEOUT)) err("MADV_PAGEOUT"); if (!pagemap_is_swapped(pagemap_fd, gopts->area_src)) { @@ -2078,11 +2079,10 @@ static void uffd_move_swap_test_common(uffd_global_test_opts_t *gopts, goto out; } - move.dst = (unsigned long)gopts->area_dst; - move.src = (unsigned long)gopts->area_src; - move.len = page_size; - if (ioctl(gopts->uffd, UFFDIO_MOVE, &move)) - err("UFFDIO_MOVE"); + if (ioctl(gopts->uffd, UFFDIO_MOVE, &move)) { + uffd_test_fail("UFFDIO_MOVE failed: %s", strerror(errno)); + goto out; + } if (pagemap_get_entry(pagemap_fd, gopts->area_dst) & PM_UFFD_WP) uffd_test_fail("uffd bit moved into an area registered for missing faults only"); diff --git a/tools/testing/selftests/net/.gitignore b/tools/testing/selftests/net/.gitignore index dacd36ed8455..21412510ac26 100644 --- a/tools/testing/selftests/net/.gitignore +++ b/tools/testing/selftests/net/.gitignore @@ -41,6 +41,7 @@ skf_net_off socket so_incoming_cpu so_netns_cookie +so_reserve_mem so_rcv_listener stress_reuseport_listen tap diff --git a/tools/testing/selftests/net/Makefile b/tools/testing/selftests/net/Makefile index cab3f2c90039..54beea2e348c 100644 --- a/tools/testing/selftests/net/Makefile +++ b/tools/testing/selftests/net/Makefile @@ -197,6 +197,7 @@ TEST_GEN_PROGS := \ sk_connect_zero_addr \ sk_so_peek_off \ so_incoming_cpu \ + so_reserve_mem \ tap \ tcp_port_share \ tls \ diff --git a/tools/testing/selftests/net/config b/tools/testing/selftests/net/config index 737e7e6327b3..d355cf980597 100644 --- a/tools/testing/selftests/net/config +++ b/tools/testing/selftests/net/config @@ -7,6 +7,7 @@ CONFIG_BRIDGE_VLAN_FILTERING=y CONFIG_CAN=m CONFIG_CAN_DEV=m CONFIG_CAN_VXCAN=m +CONFIG_CGROUPS=y CONFIG_CRYPTO_ARIA=y CONFIG_CRYPTO_CHACHA20POLY1305=m CONFIG_CRYPTO_SHA1=y @@ -61,6 +62,7 @@ CONFIG_L2TP_V3=y CONFIG_MACSEC=m CONFIG_MACVLAN=y CONFIG_MACVTAP=y +CONFIG_MEMCG=y CONFIG_MPLS=y CONFIG_MPLS_IPTUNNEL=m CONFIG_MPLS_ROUTING=m diff --git a/tools/testing/selftests/net/cork_fragsize.py b/tools/testing/selftests/net/cork_fragsize.py index 7afd643d07ec..ae6be804b7a2 100755 --- a/tools/testing/selftests/net/cork_fragsize.py +++ b/tools/testing/selftests/net/cork_fragsize.py @@ -54,7 +54,7 @@ def check_kernel_config(option: str) -> bool | None: return False except OSError: continue - return None + return None def assert_debug_kernel() -> None: @@ -71,7 +71,8 @@ def assert_debug_kernel() -> None: def check_dmesg_clean(func: str) -> bool: ''' - Check if the given function produced a WARN in dmesg. + Check if the given function produced a WARN in dmesg. Returns True if dmesg + is clean, i.e. doesn't contain traces of a WARNING in the given function. ''' with subprocess.Popen(['dmesg'], stdout=subprocess.PIPE) as dmesg: diff --git a/tools/testing/selftests/net/so_reserve_mem.c b/tools/testing/selftests/net/so_reserve_mem.c new file mode 100644 index 000000000000..aca5960ddff2 --- /dev/null +++ b/tools/testing/selftests/net/so_reserve_mem.c @@ -0,0 +1,448 @@ +// SPDX-License-Identifier: GPL-2.0 + +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <limits.h> +#include <linux/sock_diag.h> +#include <net/if.h> +#include <netinet/in.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/ioctl.h> +#include <sys/mount.h> +#include <sys/socket.h> +#include <sys/stat.h> +#include <unistd.h> + +#include "kselftest_harness.h" + +#ifndef IPPROTO_MPTCP +#define IPPROTO_MPTCP 262 +#endif + +#define SO_RESERVE_MEM_MAX (1 << 30) + +/* cgroup2 is mounted here, on a tmpfs, in a private mount namespace. */ +#define CG_TMP "/tmp" +#define CG_MNT CG_TMP "/cgroup2" + +static int cg_write(const char *dir, const char *file, const char *buf, + int flags) +{ + ssize_t len = strlen(buf); + char path[PATH_MAX]; + int fd, ret; + + snprintf(path, sizeof(path), "%s/%s", dir, file); + fd = open(path, O_WRONLY | flags); + if (fd < 0) + return -1; + ret = write(fd, buf, len) == len ? 0 : -1; + close(fd); + return ret; +} + +static int cg_read(const char *dir, const char *file, char *buf, size_t size) +{ + char path[PATH_MAX]; + ssize_t n; + int fd; + + snprintf(path, sizeof(path), "%s/%s", dir, file); + fd = open(path, O_RDONLY); + if (fd < 0) + return -1; + n = read(fd, buf, size - 1); + close(fd); + if (n < 0) + return -1; + buf[n] = '\0'; + return 0; +} + +static int cg_enter(const char *dir) +{ + char buf[32]; + + snprintf(buf, sizeof(buf), "%d\n", getpid()); + return cg_write(dir, "cgroup.procs", buf, 0); +} + +/* Path of the current cgroup, below CG_MNT. */ +static int cg_get_current(char *buf, size_t size) +{ + char line[PATH_MAX]; + int ret = -1; + FILE *f; + + f = fopen("/proc/self/cgroup", "r"); + if (!f) + return -1; + while (fgets(line, sizeof(line), f)) { + if (strncmp(line, "0::", 3)) + continue; + line[strcspn(line, "\n")] = '\0'; + if (snprintf(buf, size, "%s%s", CG_MNT, line + 3) < (int)size) + ret = 0; + break; + } + fclose(f); + return ret; +} + +static int get_reserve_mem(struct __test_metadata *_metadata, int fd) +{ + int val = -1; + socklen_t len = sizeof(val); + + EXPECT_EQ(getsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, &len), 0); + return val; +} + +static __u32 get_fwd_alloc(struct __test_metadata *_metadata, int fd) +{ + __u32 meminfo[SK_MEMINFO_VARS] = {}; + socklen_t len = sizeof(meminfo); + + EXPECT_EQ(getsockopt(fd, SOL_SOCKET, SO_MEMINFO, meminfo, &len), 0); + return meminfo[SK_MEMINFO_FWD_ALLOC]; +} + +/* Wait until a single SO_MEMINFO snapshot shows an empty write queue + * and at least @min_fwd_alloc bytes of forward alloc: ACK processing + * (possibly running on another CPU) first decrements sk_wmem_queued, + * then uncharges sk_forward_alloc. + */ +static void wait_wmem_drained(struct __test_metadata *_metadata, int fd, + __u32 min_fwd_alloc) +{ + __u32 meminfo[SK_MEMINFO_VARS] = {}; + socklen_t len; + int i; + + for (i = 0; i < 5000; i++) { + len = sizeof(meminfo); + ASSERT_EQ(getsockopt(fd, SOL_SOCKET, SO_MEMINFO, meminfo, &len), 0); + if (meminfo[SK_MEMINFO_WMEM_QUEUED] == 0 && + meminfo[SK_MEMINFO_FWD_ALLOC] >= min_fwd_alloc) + return; + usleep(1000); + } + EXPECT_EQ(meminfo[SK_MEMINFO_WMEM_QUEUED], 0U); + EXPECT_GE(meminfo[SK_MEMINFO_FWD_ALLOC], min_fwd_alloc); +} + +FIXTURE(so_reserve_mem) +{ + char cg_orig[PATH_MAX]; /* cgroup the test started in */ + char cg_test[PATH_MAX]; /* cgroup the test sockets are charged to */ + long page_size; + bool cg_created; +}; + +/* The cgroup2 mount lives in a private mount namespace which goes away + * with the test process, no need to unmount it. + */ +FIXTURE_TEARDOWN(so_reserve_mem) +{ + if (!self->cg_created) + return; + + /* Leave cg_test so that it can be removed. */ + EXPECT_EQ(cg_enter(self->cg_orig), 0); + EXPECT_EQ(rmdir(self->cg_test), 0); + self->cg_created = false; +} + +FIXTURE_SETUP(so_reserve_mem) +{ + struct ifreq ifr = { + .ifr_name = "lo", + .ifr_flags = IFF_UP, + }; + int fd, err, ret, val = 0; + char buf[256]; + + self->page_size = sysconf(_SC_PAGESIZE); + ASSERT_GT(self->page_size, 0); + + if (unshare(CLONE_NEWNS | CLONE_NEWNET)) + SKIP(return, "Failed to unshare namespaces (need root)"); + + /* Make sure the mounts below do not propagate to the host. */ + ASSERT_EQ(mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + + if (mount("none", CG_TMP, "tmpfs", 0, NULL) || + mkdir(CG_MNT, 0755) || + mount("none", CG_MNT, "cgroup2", 0, NULL)) + SKIP(return, "Failed to mount cgroup2"); + + ASSERT_EQ(cg_get_current(self->cg_orig, sizeof(self->cg_orig)), 0); + + /* cg_test needs the memory controller to be enabled in the root of + * the hierarchy (which is not necessarily the global root cgroup + * when running in a cgroup namespace). Like selftests/cgroup, leave + * it enabled on exit: disabling it could break other users of the + * hierarchy. + */ + ASSERT_EQ(cg_read(CG_MNT, "cgroup.subtree_control", buf, sizeof(buf)), 0); + if (!strstr(buf, "memory") && + cg_write(CG_MNT, "cgroup.subtree_control", "+memory", 0)) + SKIP(return, "cgroup2 memory controller not available"); + + /* cg_test is a leaf cgroup, it can always host the test process. */ + snprintf(self->cg_test, sizeof(self->cg_test), + "%s/ksft_so_reserve_mem_%d", CG_MNT, getpid()); + if (mkdir(self->cg_test, 0755)) + SKIP(return, "Failed to create test cgroup"); + self->cg_created = true; + + ret = cg_enter(self->cg_test); + if (ret) + so_reserve_mem_teardown(_metadata, self, variant); + ASSERT_EQ(ret, 0); + + /* Bring up loopback and verify memcg socket accounting is enabled */ + fd = socket(AF_INET, SOCK_STREAM, 0); + ret = fd < 0 ? -1 : ioctl(fd, SIOCSIFFLAGS, &ifr); + if (ret) { + if (fd >= 0) + close(fd); + so_reserve_mem_teardown(_metadata, self, variant); + } + ASSERT_EQ(ret, 0); + + ret = setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)); + err = errno; + close(fd); + if (ret) { + so_reserve_mem_teardown(_metadata, self, variant); + if (err == EOPNOTSUPP) + SKIP(return, "memcg socket accounting not enabled"); + } + ASSERT_EQ(ret, 0); +} + +static void check_non_tcp_rejected(struct __test_metadata *_metadata, + int domain, int type, int protocol, + int val) +{ + int fd = socket(domain, type, protocol); + int zero = 0; + + if (fd < 0) { + /* Protocol not available, or no CAP_NET_RAW */ + EXPECT_TRUE(errno == EAFNOSUPPORT || + errno == EPROTONOSUPPORT || + errno == ENOPROTOOPT || + errno == EPERM || + errno == EACCES); + TH_LOG("socket(%d, %d, %d): %s, skipped", + domain, type, protocol, strerror(errno)); + return; + } + EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &zero, sizeof(zero)), -1); + EXPECT_EQ(errno, EOPNOTSUPP); + EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1); + EXPECT_EQ(errno, EOPNOTSUPP); + close(fd); +} + +TEST_F(so_reserve_mem, non_tcp_rejected) +{ + int val = self->page_size * 4; + + check_non_tcp_rejected(_metadata, AF_INET, SOCK_DGRAM, 0, val); + check_non_tcp_rejected(_metadata, AF_UNIX, SOCK_STREAM, 0, val); + check_non_tcp_rejected(_metadata, AF_INET, SOCK_RAW, IPPROTO_ICMP, val); + check_non_tcp_rejected(_metadata, AF_INET, SOCK_STREAM, IPPROTO_MPTCP, val); +} + +TEST_F(so_reserve_mem, grow_shrink_and_rounding) +{ + int ps = self->page_size; + int fd, val; + + fd = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(fd, 0); + + EXPECT_EQ(get_reserve_mem(_metadata, fd), 0); + EXPECT_EQ(get_fwd_alloc(_metadata, fd), 0U); + + /* Negative or > SO_RESERVE_MEM_MAX value -> EINVAL */ + val = -1; + EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1); + EXPECT_EQ(errno, EINVAL); + + val = SO_RESERVE_MEM_MAX + 1; + EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1); + EXPECT_EQ(errno, EINVAL); + + val = INT_MAX; + EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1); + EXPECT_EQ(errno, EINVAL); + + /* 1 byte rounds up to 1 page */ + val = 1; + ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, fd), ps); + EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)ps); + + /* Grow to 16 pages */ + val = 16 * ps; + ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, fd), 16 * ps); + EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)(16 * ps)); + + /* Shrink by 1 byte (rounds delta down to 0 -> stays 16 pages) */ + val = 16 * ps - 1; + ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, fd), 16 * ps); + EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)(16 * ps)); + + /* Shrink to 4 pages */ + val = 4 * ps; + ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, fd), 4 * ps); + EXPECT_EQ(get_fwd_alloc(_metadata, fd), (__u32)(4 * ps)); + + /* Release all */ + val = 0; + ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, fd), 0); + EXPECT_EQ(get_fwd_alloc(_metadata, fd), 0U); + + close(fd); +} + +TEST_F(so_reserve_mem, cgroup_memory_max) +{ + int ps = self->page_size; + char buf[32]; + int fd, val; + + /* The socket is charged to cg_test */ + fd = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(fd, 0); + + /* Move the test process back to its original cgroup before lowering + * cg_test's memory.max, so that the limit only governs the socket's + * memcg charges and cannot trigger OOM on the test process itself. + */ + ASSERT_EQ(cg_enter(self->cg_orig), 0); + + /* Limit cg_test memory to 8 pages and try to reserve 128 pages. + * O_NONBLOCK: do not try to reclaim cg_test usage above the new limit. + */ + snprintf(buf, sizeof(buf), "%d\n", 8 * ps); + ASSERT_EQ(cg_write(self->cg_test, "memory.max", buf, O_NONBLOCK), 0); + + val = 128 * ps; + EXPECT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), -1); + EXPECT_EQ(errno, ENOMEM); + EXPECT_EQ(get_reserve_mem(_metadata, fd), 0); + + /* Restore unlimited memory.max */ + ASSERT_EQ(cg_write(self->cg_test, "memory.max", "max\n", 0), 0); + + val = 4 * ps; + ASSERT_EQ(setsockopt(fd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, fd), 4 * ps); + + close(fd); +} + +TEST_F(so_reserve_mem, accept_child_zero_reserve) +{ + struct sockaddr_in addr = { + .sin_family = AF_INET, + .sin_addr.s_addr = htonl(INADDR_LOOPBACK), + }; + socklen_t alen = sizeof(addr); + int ps = self->page_size; + int lfd, cfd, sfd, val; + + lfd = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(lfd, 0); + + /* Set SO_RESERVE_MEM on listener before listen() */ + val = 4 * ps; + ASSERT_EQ(setsockopt(lfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + ASSERT_EQ(bind(lfd, (struct sockaddr *)&addr, sizeof(addr)), 0); + ASSERT_EQ(listen(lfd, 2), 0); + ASSERT_EQ(getsockname(lfd, (struct sockaddr *)&addr, &alen), 0); + + /* Grow SO_RESERVE_MEM on listener after listen() */ + val = 8 * ps; + ASSERT_EQ(setsockopt(lfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, lfd), 8 * ps); + EXPECT_EQ(get_fwd_alloc(_metadata, lfd), (__u32)(8 * ps)); + + cfd = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(cfd, 0); + ASSERT_EQ(connect(cfd, (struct sockaddr *)&addr, sizeof(addr)), 0); + + sfd = accept(lfd, NULL, NULL); + ASSERT_GE(sfd, 0); + + /* Child after accept() must have 0 reserve while listener keeps 8 pages */ + EXPECT_EQ(get_reserve_mem(_metadata, sfd), 0); + EXPECT_EQ(get_fwd_alloc(_metadata, sfd), 0U); + EXPECT_EQ(get_reserve_mem(_metadata, lfd), 8 * ps); + EXPECT_EQ(get_fwd_alloc(_metadata, lfd), (__u32)(8 * ps)); + + /* Child can still independently set its own SO_RESERVE_MEM */ + val = 6 * ps; + ASSERT_EQ(setsockopt(sfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + EXPECT_EQ(get_reserve_mem(_metadata, sfd), 6 * ps); + EXPECT_EQ(get_fwd_alloc(_metadata, sfd), (__u32)(6 * ps)); + + close(sfd); + close(cfd); + close(lfd); +} + +TEST_F(so_reserve_mem, preserved_after_traffic) +{ + struct sockaddr_in addr = { + .sin_family = AF_INET, + .sin_addr.s_addr = htonl(INADDR_LOOPBACK), + }; + socklen_t alen = sizeof(addr); + int ps = self->page_size; + int lfd, cfd, sfd, val; + char buf[8192] = {}; + + lfd = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(lfd, 0); + ASSERT_EQ(bind(lfd, (struct sockaddr *)&addr, sizeof(addr)), 0); + ASSERT_EQ(listen(lfd, 1), 0); + ASSERT_EQ(getsockname(lfd, (struct sockaddr *)&addr, &alen), 0); + + cfd = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(cfd, 0); + val = 16 * ps; + ASSERT_EQ(setsockopt(cfd, SOL_SOCKET, SO_RESERVE_MEM, &val, sizeof(val)), 0); + ASSERT_EQ(connect(cfd, (struct sockaddr *)&addr, sizeof(addr)), 0); + + sfd = accept(lfd, NULL, NULL); + ASSERT_GE(sfd, 0); + + /* Send & drain traffic; cfd must retain its 16-page forward alloc, + * while sfd (0 reserve) reclaims its forward alloc back to 0. + */ + ASSERT_EQ(send(cfd, buf, sizeof(buf), 0), (ssize_t)sizeof(buf)); + ASSERT_EQ(recv(sfd, buf, sizeof(buf), MSG_WAITALL), (ssize_t)sizeof(buf)); + wait_wmem_drained(_metadata, cfd, 16 * ps); + + EXPECT_EQ(get_fwd_alloc(_metadata, sfd), 0U); + + close(sfd); + close(cfd); + close(lfd); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/nfsd/.gitignore b/tools/testing/selftests/nfsd/.gitignore new file mode 100644 index 000000000000..19e6dec04d8e --- /dev/null +++ b/tools/testing/selftests/nfsd/.gitignore @@ -0,0 +1 @@ +nfsd_netlink_listener diff --git a/tools/testing/selftests/nfsd/Makefile b/tools/testing/selftests/nfsd/Makefile new file mode 100644 index 000000000000..15ac65549d25 --- /dev/null +++ b/tools/testing/selftests/nfsd/Makefile @@ -0,0 +1,6 @@ +# SPDX-License-Identifier: GPL-2.0 +CFLAGS += $(KHDR_INCLUDES) -Wall + +TEST_GEN_PROGS := nfsd_netlink_listener + +include ../lib.mk diff --git a/tools/testing/selftests/nfsd/config b/tools/testing/selftests/nfsd/config new file mode 100644 index 000000000000..0eef03af3503 --- /dev/null +++ b/tools/testing/selftests/nfsd/config @@ -0,0 +1,14 @@ +CONFIG_NAMESPACES=y +CONFIG_NET_NS=y +CONFIG_SHMEM=y +CONFIG_TMPFS=y +CONFIG_UNIX=y +CONFIG_INET=y +CONFIG_IPV6=y +CONFIG_MULTIUSER=y +CONFIG_PROC_FS=y +CONFIG_FILE_LOCKING=y +CONFIG_INOTIFY_USER=y +CONFIG_SUNRPC=y +CONFIG_NFSD=y +CONFIG_NFSD_V4=y diff --git a/tools/testing/selftests/nfsd/nfsd_netlink_listener.c b/tools/testing/selftests/nfsd/nfsd_netlink_listener.c new file mode 100644 index 000000000000..106360f87b99 --- /dev/null +++ b/tools/testing/selftests/nfsd/nfsd_netlink_listener.c @@ -0,0 +1,1323 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Regression tests for the NFSD generic-netlink listener interface + * (NFSD_CMD_LISTENER_SET / NFSD_CMD_LISTENER_GET). + * + * Three groups: + * validation - malformed/abusive LISTENER_SET requests are rejected by + * nfsd_nl_validate_listeners(), before nfsd_mutex is taken. + * functional - create/add/remove listeners and verify LISTENER_GET + * reflects the set (round-trip of transport + addr:port). + * semantics - once threads are running (THREADS_SET) a listener change + * is refused with -EBUSY. + * + * Each test runs in its own private net + mount namespace (unshare in + * FIXTURE_SETUP). /run is masked there: a pathname AF_LOCAL connect is not + * scoped by the network namespace, since unix_find_bsd() resolves by inode + * and takes no struct net, so the kernel's rpcbind client would otherwise be + * able to reach the rpcbind running on the host. Anything that creates a + * serv is served by the per-netns rpcbind stub below instead. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <poll.h> +#include <sched.h> +#include <signal.h> +#include <stddef.h> +#include <stdint.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <unistd.h> +#include <sys/mman.h> +#include <sys/mount.h> +#include <sys/prctl.h> +#include <sys/socket.h> +#include <sys/ioctl.h> +#include <sys/stat.h> +#include <sys/time.h> +#include <sys/un.h> +#include <sys/wait.h> +#include <net/if.h> +#include <netinet/in.h> +#include <linux/netlink.h> +#include <linux/genetlink.h> +#include <linux/nfsd_netlink.h> + +#include "../kselftest_harness.h" + +#define NLA_ALIGN4(len) (((len) + 3) & ~3) +#define TEST_PORT 20049 +#define MAX_LISTENERS 8 +#define RECV_TIMEO_SEC 30 + +static int nfsd_family = -1; /* set per-test in FIXTURE_SETUP */ + +/* Extack message from the last genl_request(); empty if there was none. */ +static char last_extack[128]; + +static void die(const char *msg) +{ + perror(msg); + exit(1); +} + +/* ------------------- minimal generic-netlink plumbing ------------------- */ + +static int genl_open(void) +{ + struct sockaddr_nl sa = { .nl_family = AF_NETLINK }; + struct timeval tv = { .tv_sec = RECV_TIMEO_SEC }; + int fd = socket(AF_NETLINK, SOCK_RAW, NETLINK_GENERIC); + int on = 1; + + if (fd < 0) + die("socket(NETLINK_GENERIC)"); + if (bind(fd, (void *)&sa, sizeof(sa)) < 0) + die("bind(netlink)"); + setsockopt(fd, SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)); + /* + * Ask for extack, and cap the ack so the request is not echoed back: + * the TLVs then always follow the fixed part of the error message. + */ + setsockopt(fd, SOL_NETLINK, NETLINK_EXT_ACK, &on, sizeof(on)); + setsockopt(fd, SOL_NETLINK, NETLINK_CAP_ACK, &on, sizeof(on)); + return fd; +} + +/* Stash the extack message of an ack, if it carries one. */ +static void parse_extack(const char *rbuf) +{ + const struct nlmsghdr *nlh = (const void *)rbuf; + const struct nlattr *na; + int off, left; + + last_extack[0] = '\0'; + if (nlh->nlmsg_type != NLMSG_ERROR || + !(nlh->nlmsg_flags & NLM_F_ACK_TLVS)) + return; + + off = NLMSG_HDRLEN + NLMSG_ALIGN(sizeof(struct nlmsgerr)); + left = nlh->nlmsg_len - off; + na = (const void *)(rbuf + off); + + while (left >= (int)NLA_HDRLEN) { + if ((na->nla_type & NLA_TYPE_MASK) == NLMSGERR_ATTR_MSG) { + strncpy(last_extack, (const char *)na + NLA_HDRLEN, + sizeof(last_extack) - 1); + last_extack[sizeof(last_extack) - 1] = '\0'; + return; + } + left -= NLA_ALIGN4(na->nla_len); + na = (const void *)((const char *)na + NLA_ALIGN4(na->nla_len)); + } +} + +/* Append an attribute at @off; return the new (aligned) offset. */ +static int put_attr(char *buf, int off, uint16_t type, + const void *data, int len) +{ + struct nlattr *na = (void *)(buf + off); + + na->nla_type = type; + na->nla_len = NLA_HDRLEN + len; + if (len) + memcpy(buf + off + NLA_HDRLEN, data, len); + return off + NLA_ALIGN4(NLA_HDRLEN + len); +} + +/* Build a genl message header into @buf; return the offset past it. */ +static int genl_hdr(char *buf, uint16_t type, uint16_t flags, uint8_t cmd) +{ + struct nlmsghdr *nlh = (void *)buf; + struct genlmsghdr *gnl = (void *)(buf + NLMSG_HDRLEN); + + memset(buf, 0, NLMSG_HDRLEN + GENL_HDRLEN); + nlh->nlmsg_type = type; + nlh->nlmsg_flags = flags; + nlh->nlmsg_seq = 1; + gnl->cmd = cmd; + gnl->version = 1; + return NLMSG_HDRLEN + GENL_HDRLEN; +} + +/* Send an nfsd command with an ACK; return the ACK errno (<= 0). */ +static int genl_request(uint8_t cmd, const char *attrs, int attrs_len) +{ + char buf[1 << 20], rbuf[4096]; + struct nlmsghdr *nlh = (void *)buf; + int fd = genl_open(); + int off, n, ret; + + off = genl_hdr(buf, nfsd_family, NLM_F_REQUEST | NLM_F_ACK, cmd); + if (attrs_len) { + memcpy(buf + off, attrs, attrs_len); + off += attrs_len; + } + nlh->nlmsg_len = off; + + if (send(fd, buf, off, 0) < 0) + die("send(genl)"); + + last_extack[0] = '\0'; + n = recv(fd, rbuf, sizeof(rbuf), 0); + if (n < 0) { + ret = (errno == EAGAIN || errno == EWOULDBLOCK) ? -ETIMEDOUT : -errno; + } else if (((struct nlmsghdr *)rbuf)->nlmsg_type == NLMSG_ERROR) { + ret = ((struct nlmsgerr *)NLMSG_DATA(rbuf))->error; + parse_extack(rbuf); + } else { + ret = 0; + } + close(fd); + return ret; +} + +/* Send a command and return the full reply message; -errno on failure. */ +static int genl_request_reply(uint8_t cmd, char *rbuf, size_t rlen) +{ + char buf[256]; + struct nlmsghdr *nlh = (void *)buf; + int fd = genl_open(); + int off, n, ret; + + off = genl_hdr(buf, nfsd_family, NLM_F_REQUEST, cmd); + nlh->nlmsg_len = off; + + if (send(fd, buf, off, 0) < 0) + die("send(genl reply)"); + + n = recv(fd, rbuf, rlen, 0); + if (n < 0) + ret = (errno == EAGAIN || errno == EWOULDBLOCK) ? -ETIMEDOUT : -errno; + else if (((struct nlmsghdr *)rbuf)->nlmsg_type == NLMSG_ERROR) + ret = ((struct nlmsgerr *)NLMSG_DATA(rbuf))->error; + else + ret = n; + close(fd); + return ret; +} + +/* Resolve the "nfsd" genl family id; -1 if not registered. */ +static int genl_resolve_nfsd(void) +{ + char buf[1024], rbuf[4096]; + struct nlmsghdr *nlh = (void *)buf; + struct nlmsghdr *rh = (void *)rbuf; + struct nlattr *na; + int fd, off, left, id = -1; + + fd = genl_open(); + off = genl_hdr(buf, GENL_ID_CTRL, NLM_F_REQUEST, CTRL_CMD_GETFAMILY); + off = put_attr(buf, off, CTRL_ATTR_FAMILY_NAME, + NFSD_FAMILY_NAME, sizeof(NFSD_FAMILY_NAME)); + nlh->nlmsg_len = off; + + if (send(fd, buf, off, 0) < 0) + die("send(GETFAMILY)"); + if (recv(fd, rbuf, sizeof(rbuf), 0) < 0) + die("recv(GETFAMILY)"); + close(fd); + + if (rh->nlmsg_type == NLMSG_ERROR) + return -1; + + na = (void *)((char *)NLMSG_DATA(rh) + GENL_HDRLEN); + left = rh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN; + while (left >= (int)NLA_HDRLEN) { + if (na->nla_type == CTRL_ATTR_FAMILY_ID) { + id = *(uint16_t *)((char *)na + NLA_HDRLEN); + break; + } + left -= NLA_ALIGN4(na->nla_len); + na = (void *)((char *)na + NLA_ALIGN4(na->nla_len)); + } + return id; +} + +/* ------------------- listener request builders ------------------- */ + +/* Fine-grained control for negative tests: any field can be omitted/malformed. */ +struct raw_listener { + const char *xprt; /* NULL -> omit NFSD_A_SOCK_TRANSPORT_NAME */ + int emit_addr; /* 0 -> omit NFSD_A_SOCK_ADDR */ + const void *addr; + int addr_len; /* bytes to emit for NFSD_A_SOCK_ADDR */ +}; + +static int put_raw_listener(char *buf, int off, const struct raw_listener *r) +{ + struct nlattr *nest = (void *)(buf + off); + int inner = off + NLA_HDRLEN; + + if (r->emit_addr) + inner = put_attr(buf, inner, NFSD_A_SOCK_ADDR, r->addr, r->addr_len); + if (r->xprt) + inner = put_attr(buf, inner, NFSD_A_SOCK_TRANSPORT_NAME, + r->xprt, strlen(r->xprt) + 1); + nest->nla_type = NFSD_A_SERVER_SOCK_ADDR | NLA_F_NESTED; + nest->nla_len = inner - off; + return off + NLA_ALIGN4(nest->nla_len); +} + +/* Well-formed loopback listener for @family (AF_INET or AF_INET6). */ +static int put_listener_af(char *buf, int off, const char *xprt, int family, + uint16_t port) +{ + struct sockaddr_storage ss = {0}; + struct raw_listener r = { .xprt = xprt, .emit_addr = 1, .addr = &ss }; + + if (family == AF_INET6) { + struct sockaddr_in6 *s6 = (void *)&ss; + + s6->sin6_family = AF_INET6; + s6->sin6_port = htons(port); + s6->sin6_addr = in6addr_loopback; + r.addr_len = sizeof(*s6); + } else { + struct sockaddr_in *s4 = (void *)&ss; + + s4->sin_family = AF_INET; + s4->sin_port = htons(port); + s4->sin_addr.s_addr = htonl(INADDR_LOOPBACK); + r.addr_len = sizeof(*s4); + } + return put_raw_listener(buf, off, &r); +} + +static int put_listener(char *buf, int off, const char *xprt, uint16_t port) +{ + return put_listener_af(buf, off, xprt, AF_INET, port); +} + +/* ------------------- LISTENER_GET parsing ------------------- */ + +struct listener_ent { + char xprt[16]; + int family; + uint16_t port; + struct in_addr a4; + struct in6_addr a6; +}; + +static int parse_listener_get(const char *rbuf, int len, + struct listener_ent *out, int max) +{ + const struct nlmsghdr *nlh = (const void *)rbuf; + const struct nlattr *na; + int left, count = 0; + + (void)len; + na = (const void *)(rbuf + NLMSG_HDRLEN + GENL_HDRLEN); + left = nlh->nlmsg_len - NLMSG_HDRLEN - GENL_HDRLEN; + + while (left >= (int)NLA_HDRLEN) { + int alen = na->nla_len; + + if ((na->nla_type & NLA_TYPE_MASK) == NFSD_A_SERVER_SOCK_ADDR && + count < max) { + const struct nlattr *in = (const void *)((char *)na + NLA_HDRLEN); + int ileft = alen - NLA_HDRLEN; + struct listener_ent *e = &out[count]; + + memset(e, 0, sizeof(*e)); + while (ileft >= (int)NLA_HDRLEN) { + const void *d = (const char *)in + NLA_HDRLEN; + int t = in->nla_type & NLA_TYPE_MASK; + + if (t == NFSD_A_SOCK_TRANSPORT_NAME) { + strncpy(e->xprt, d, sizeof(e->xprt) - 1); + } else if (t == NFSD_A_SOCK_ADDR) { + const struct sockaddr_storage *ss = d; + + e->family = ss->ss_family; + if (ss->ss_family == AF_INET) { + const struct sockaddr_in *s = d; + + e->a4 = s->sin_addr; + e->port = ntohs(s->sin_port); + } else if (ss->ss_family == AF_INET6) { + const struct sockaddr_in6 *s = d; + + e->a6 = s->sin6_addr; + e->port = ntohs(s->sin6_port); + } + } + ileft -= NLA_ALIGN4(in->nla_len); + in = (const void *)((char *)in + NLA_ALIGN4(in->nla_len)); + } + count++; + } + left -= NLA_ALIGN4(alen); + na = (const void *)((char *)na + NLA_ALIGN4(alen)); + } + return count; +} + +/* ------------------- convenience wrappers ------------------- */ + +static int listener_set(const char *attrs, int len) +{ + return genl_request(NFSD_CMD_LISTENER_SET, attrs, len); +} + +/* + * Enable exactly one NFS version in this netns. NFSD_CMD_VERSION_SET clears + * every version first, so one nest is enough to leave the server v4-only. + * It refuses once a serv exists, so call it before any listener. + */ +static int version_set_only(uint32_t major, uint32_t minor) +{ + char attrs[64]; + struct nlattr *nest = (void *)attrs; + int inner = NLA_HDRLEN; + + inner = put_attr(attrs, inner, NFSD_A_VERSION_MAJOR, + &major, sizeof(major)); + inner = put_attr(attrs, inner, NFSD_A_VERSION_MINOR, + &minor, sizeof(minor)); + inner = put_attr(attrs, inner, NFSD_A_VERSION_ENABLED, NULL, 0); + nest->nla_type = NFSD_A_SERVER_PROTO_VERSION | NLA_F_NESTED; + nest->nla_len = inner; + + return genl_request(NFSD_CMD_VERSION_SET, attrs, NLA_ALIGN4(inner)); +} + +/* Fetch the current listeners; returns count (>=0) or -errno. */ +static int listener_get(struct listener_ent *out, int max) +{ + char rbuf[8192]; + int n = genl_request_reply(NFSD_CMD_LISTENER_GET, rbuf, sizeof(rbuf)); + + if (n < 0) + return n; + return parse_listener_get(rbuf, n, out, max); +} + +/* + * Every listener these tests create comes from put_listener_af(), so the + * address is always loopback. Match on it too: without that, a reply that + * gave the right transport and port on the wrong address (0.0.0.0, say) + * would pass. + */ +static struct listener_ent *find_listener(struct listener_ent *e, int n, + const char *xprt, int family, + uint16_t port) +{ + int i; + + for (i = 0; i < n; i++) { + if (e[i].family != family || e[i].port != port || + strcmp(e[i].xprt, xprt)) + continue; + if (family == AF_INET6) { + if (memcmp(&e[i].a6, &in6addr_loopback, sizeof(e[i].a6))) + continue; + } else if (e[i].a4.s_addr != htonl(INADDR_LOOPBACK)) { + continue; + } + return &e[i]; + } + return NULL; +} + +/* Start (@n > 0) or stop (@n == 0) nfsd threads in this netns. */ +static int threads_set(int n) +{ + char attrs[64]; + uint32_t v = n; + int off = put_attr(attrs, 0, NFSD_A_SERVER_THREADS, &v, sizeof(v)); + + return genl_request(NFSD_CMD_THREADS_SET, attrs, off); +} + +/* ------------------- per-netns local rpcbind stub ------------------- */ + +/* + * Creating a listener registers with rpcbind: nfsd_nl_listener_set_doit() + * passes no SVC_SOCK_ANONYMOUS for the first entry of a request, so + * pmap_register is true in svc_setup_socket(). The fixture's server has v3 + * enabled, and nfsd_version3 does not set vs_rpcb_optnl, so a failure there + * comes back out of svc_register() and takes the listener down with it. + * With nothing listening, every attempt first waits out the local rpcbind + * timeout. The abstract AF_LOCAL name the kernel tries first is per-netns + * (unix_find_abstract() takes a struct net), so answer it here and stay out + * of the host's rpcbind. + * + * Arguments are never decoded. The NULL procedure gets an empty success and + * SET/UNSET get TRUE, for both RPCBVERS_2 and RPCBVERS_4. v4 has to be + * answered because __svc_rpcb_register6() turns a v4 refusal into + * -EAFNOSUPPORT, which would leave every IPv6 listener unregistered. + * + * In RPCB_STUB_REFUSE mode SET is answered FALSE instead, which + * rpcb_register_call() reports as -EACCES. UNSET is left alone: only + * svc_unregister() issues it, and it discards the result. + * + * In RPCB_STUB_SILENT mode a SET or an UNSET is read and nothing is written + * back, so the kernel waits out its own timeout. That is the only mode that + * makes rpcb_register_call() report a call that got no answer, which is what + * the per-net failure count records. The NULL procedure is still answered: + * rpcb_create_af_local() builds its client without RPC_CLNT_CREATE_NOPING, so + * rpc_create() pings, and a ping that goes unanswered drops the kernel onto + * the loopback rpcb_create_local_net() client, which never reaches this stub. + * + * The stub also keeps counters and the mode in a page shared with the test, so + * a test can assert that the kernel never talked to rpcbind at all, or that it + * dropped the local rpcbind client and had to reconnect. + * + * The mode lives there rather than in the child so that a test can change it + * with a serv already up. Killing and restarting the stub would close the + * connection the kernel holds, and rpcb_register_call() issues UNSET over + * AF_LOCAL with RPC_TASK_NOCONNECT, so the next call would fail at once with + * -ENOTCONN instead of waiting out a timeout. + */ +#define RPCB_PROGRAM 100000 +#define RPCB_PROC_NULL 0 +#define RPCB_PROC_SET 1 +#define RPCB_PROC_UNSET 2 +#define RPCB_ABSTRACT_NAME "/run/rpcbind.sock" +#define RPCB_STUB_MAXCONN 4 + +enum { RPCB_STUB_ACCEPT, RPCB_STUB_REFUSE, RPCB_STUB_SILENT }; + +struct rpcb_stub_stats { + unsigned int conns; /* connections accepted */ + unsigned int calls; /* calls received */ + unsigned int mode; /* RPCB_STUB_*, read on every call */ +}; + +static volatile struct rpcb_stub_stats *rpcb_stats; /* MAP_SHARED */ + +static int rpcb_stats_alloc(void) +{ + void *p = mmap(NULL, sizeof(*rpcb_stats), PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_ANONYMOUS, -1, 0); + + if (p == MAP_FAILED) + return -1; + rpcb_stats = p; + return 0; +} + +/* + * The stub bumps these before it replies and the kernel waits for that reply, + * so whatever a netlink request provoked is visible once it returns. + */ +static int rpcb_calls(void) +{ + return rpcb_stats ? (int)rpcb_stats->calls : 0; +} + +static int rpcb_conns(void) +{ + return rpcb_stats ? (int)rpcb_stats->conns : 0; +} + +/* Takes effect on the stub's next call; the caller has not sent one yet. */ +static void rpcb_stub_set_mode(int mode) +{ + rpcb_stats->mode = mode; +} + +static int rpcb_stub_listen(void) +{ + struct sockaddr_un sun = { .sun_family = AF_UNIX }; + size_t nlen = strlen(RPCB_ABSTRACT_NAME); + socklen_t alen; + int fd; + + /* Abstract names are length-delimited, so the length must match. */ + memcpy(sun.sun_path + 1, RPCB_ABSTRACT_NAME, nlen); + alen = offsetof(struct sockaddr_un, sun_path) + 1 + nlen; + + fd = socket(AF_UNIX, SOCK_STREAM, 0); + if (fd < 0) + return -1; + if (bind(fd, (struct sockaddr *)&sun, alen) < 0 || + listen(fd, RPCB_STUB_MAXCONN) < 0) { + close(fd); + return -1; + } + return fd; +} + +static int rpcb_stub_read(int fd, void *buf, size_t len) +{ + size_t done = 0; + + while (done < len) { + ssize_t n = read(fd, (char *)buf + done, len - done); + + if (n <= 0) + return -1; + done += n; + } + return 0; +} + +/* Handle one record-marked RPC call. Returns -1 when the peer is done. */ +static int rpcb_stub_call(int fd) +{ + unsigned int len, nrep = 6, mode = rpcb_stats->mode; + uint32_t mark, call[6], rep[7]; + size_t replen; + + if (rpcb_stub_read(fd, &mark, sizeof(mark))) + return -1; + len = ntohl(mark) & 0x7fffffff; + if (len < sizeof(call) || len > 4096) + return -1; + if (rpcb_stub_read(fd, call, sizeof(call))) + return -1; + + /* xid, msg_type, rpcvers, prog, vers, proc; the rest is discarded */ + for (len -= sizeof(call); len; ) { + char sink[256]; + unsigned int n = len > sizeof(sink) ? sizeof(sink) : len; + + if (rpcb_stub_read(fd, sink, n)) + return -1; + len -= n; + } + + if (rpcb_stats) + rpcb_stats->calls++; + + rep[0] = call[0]; /* xid */ + rep[1] = htonl(1); /* REPLY */ + rep[2] = htonl(0); /* MSG_ACCEPTED */ + rep[3] = htonl(0); /* verifier flavor AUTH_NULL */ + rep[4] = htonl(0); /* verifier length */ + rep[5] = htonl(0); /* SUCCESS */ + + if (ntohl(call[3]) != RPCB_PROGRAM) { + rep[5] = htonl(1); /* PROG_UNAVAIL */ + } else { + unsigned int proc = ntohl(call[5]); + + switch (proc) { + case RPCB_PROC_NULL: + break; + case RPCB_PROC_SET: + rep[6] = htonl(mode == RPCB_STUB_REFUSE ? 0 : 1); + nrep = 7; + break; + case RPCB_PROC_UNSET: + rep[6] = htonl(1); /* TRUE */ + nrep = 7; + break; + default: + rep[5] = htonl(3); /* PROC_UNAVAIL */ + } + + /* + * Answer nothing, so the caller waits out its timeout. The + * NULL procedure is answered even here: the kernel pings at + * client creation, and a ping with no answer takes it off + * this socket entirely. + */ + if (mode == RPCB_STUB_SILENT && proc != RPCB_PROC_NULL) + return 0; + } + + replen = nrep * sizeof(rep[0]); + mark = htonl(0x80000000 | replen); + if (write(fd, &mark, sizeof(mark)) != (ssize_t)sizeof(mark) || + write(fd, rep, replen) != (ssize_t)replen) + return -1; + return 0; +} + +static void rpcb_stub_serve(int lfd) +{ + struct pollfd pfd[1 + RPCB_STUB_MAXCONN]; + nfds_t n = 1, i; + + pfd[0].fd = lfd; + + for (;;) { + /* stop polling the listener when full, or poll() spins */ + pfd[0].events = n < 1 + RPCB_STUB_MAXCONN ? POLLIN : 0; + + if (poll(pfd, n, -1) < 0) + return; + + if (pfd[0].revents & POLLIN) { + int c = accept(lfd, NULL, NULL); + + if (c >= 0) { + pfd[n].fd = c; + pfd[n].events = POLLIN; + /* + * poll() ran with the old n, so it did not + * write this revents. The loop below reads it. + */ + pfd[n].revents = 0; + n++; + if (rpcb_stats) + rpcb_stats->conns++; + } + } + + for (i = 1; i < n; i++) { + if (!(pfd[i].revents & (POLLIN | POLLHUP | POLLERR))) + continue; + if (rpcb_stub_call(pfd[i].fd)) { + close(pfd[i].fd); + pfd[i] = pfd[--n]; + } + } + } +} + +/* Returns the stub's pid, or -1. The socket is listening before we fork. */ +static pid_t rpcb_stub_start(int mode) +{ + int lfd = rpcb_stub_listen(); + pid_t pid; + + if (lfd < 0) + return -1; + + rpcb_stats->mode = mode; + + pid = fork(); + if (pid < 0) { + close(lfd); + return -1; + } + if (pid == 0) { + signal(SIGPIPE, SIG_IGN); + prctl(PR_SET_PDEATHSIG, SIGKILL); + if (getppid() == 1) /* raced with parent exit */ + _exit(0); + rpcb_stub_serve(lfd); + _exit(0); + } + + close(lfd); + return pid; +} + +/* --------------------------- fixture --------------------------- */ + +FIXTURE(nfsd_listener) { + pid_t rpcbd; +}; + +FIXTURE_SETUP(nfsd_listener) +{ + struct ifreq ifr = {0}; + struct stat st; + int s; + + if (geteuid() != 0) + SKIP(return, "must be run as root"); + if (unshare(CLONE_NEWNET | CLONE_NEWNS) < 0) + SKIP(return, "unshare(NEWNET|NEWNS): %s", strerror(errno)); + if (mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL) < 0) + SKIP(return, "mount(/ private): %s", strerror(errno)); + + /* + * Keep the kernel's rpcbind client inside this namespace. The + * abstract socket it tries first is per-netns, but the + * "/var/run/rpcbind.sock" fallback is not, so hide the path. + */ + if (mount("tmpfs", "/run", "tmpfs", 0, NULL) < 0) + SKIP(return, "mount(tmpfs on /run): %s", strerror(errno)); + if (lstat("/var/run", &st) == 0 && S_ISDIR(st.st_mode) && + mount("tmpfs", "/var/run", "tmpfs", 0, NULL) < 0) + SKIP(return, "mount(tmpfs on /var/run): %s", strerror(errno)); + + /* + * Bring loopback up so listener binds (127.0.0.1 / ::1) work. Root + * without CAP_NET_ADMIN in this netns gets -EPERM here, so skip. + */ + s = socket(AF_INET, SOCK_DGRAM, 0); + ASSERT_GE(s, 0); + strcpy(ifr.ifr_name, "lo"); + if (ioctl(s, SIOCGIFFLAGS, &ifr) < 0) { + close(s); + SKIP(return, "SIOCGIFFLAGS(lo): %s", strerror(errno)); + } + ifr.ifr_flags |= IFF_UP | IFF_RUNNING; + if (ioctl(s, SIOCSIFFLAGS, &ifr) < 0) { + close(s); + SKIP(return, "SIOCSIFFLAGS(lo): %s", strerror(errno)); + } + close(s); + + nfsd_family = genl_resolve_nfsd(); + if (nfsd_family < 0) + SKIP(return, "nfsd genl family not found (modprobe nfsd?)"); + + if (rpcb_stats_alloc() < 0) + SKIP(return, "mmap(rpcbind stub counters): %s", strerror(errno)); + + self->rpcbd = rpcb_stub_start(RPCB_STUB_ACCEPT); + if (self->rpcbd < 0) + SKIP(return, "cannot start the rpcbind stub: %s", + strerror(errno)); +} + +FIXTURE_TEARDOWN(nfsd_listener) +{ + /* + * A listener holds a reference to this netns, which outlives the test + * process, so anything still up leaks it. Threads pin the listeners in + * turn; dropping them destroys the serv and everything under it. + */ + if (nfsd_family >= 0 && listener_set(NULL, 0) == -EBUSY) + threads_set(0); + + if (self->rpcbd > 0) { + kill(self->rpcbd, SIGKILL); + waitpid(self->rpcbd, NULL, 0); + } + if (rpcb_stats) { + munmap((void *)rpcb_stats, sizeof(*rpcb_stats)); + rpcb_stats = NULL; + } +} + +/* ===================== validation / negative ===================== */ + +TEST_F(nfsd_listener, val_empty_list_ok) +{ + EXPECT_EQ(0, listener_set(NULL, 0)); +} + +TEST_F(nfsd_listener, val_too_many) +{ + static char attrs[1 << 20]; + int i, off = 0; + + for (i = 0; i < 1025; i++) /* > NFSD_NL_LISTENER_MAX (1024) */ + off = put_listener(attrs, off, "udp", TEST_PORT); + EXPECT_EQ(-E2BIG, listener_set(attrs, off)); +} + +TEST_F(nfsd_listener, val_missing_addr) +{ + char attrs[64]; + struct raw_listener r = { .xprt = "tcp", .emit_addr = 0 }; + int off = put_raw_listener(attrs, 0, &r); + + EXPECT_EQ(-EINVAL, listener_set(attrs, off)); +} + +TEST_F(nfsd_listener, val_missing_transport) +{ + struct sockaddr_in s4 = { .sin_family = AF_INET, .sin_port = htons(TEST_PORT) }; + struct raw_listener r = { .xprt = NULL, .emit_addr = 1, + .addr = &s4, .addr_len = sizeof(s4) }; + char attrs[64]; + int off = put_raw_listener(attrs, 0, &r); + + EXPECT_EQ(-EINVAL, listener_set(attrs, off)); +} + +/* + * A name matching no transport class must be refused before nfsd_mutex is + * taken, so it never reaches svc_xprt_create_from_sa() and its + * request_module("svc%s", name) upcall. + * + * The errno cannot show that -- svc_xprt_create_from_sa() returns + * -EPROTONOSUPPORT for an unknown name too. The rpcbind traffic can: + * getting that far means nfsd_create_serv() ran, and svc_bind() pings + * rpcbind at client creation and then sweeps stale entries with + * svc_unregister(). A silent stub is the proof nothing was created. + */ +TEST_F(nfsd_listener, val_bad_transport) +{ + char attrs[64]; + int off = put_listener(attrs, 0, "bogus_xprt", TEST_PORT); + + ASSERT_EQ(0, rpcb_calls()); + EXPECT_EQ(-EPROTONOSUPPORT, listener_set(attrs, off)); + EXPECT_EQ(0, rpcb_calls()); +} + +TEST_F(nfsd_listener, val_addr_too_short) +{ + unsigned char tiny = 0; + struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, + .addr = &tiny, .addr_len = 1 }; + char attrs[64]; + int off = put_raw_listener(attrs, 0, &r); + + EXPECT_EQ(-EINVAL, listener_set(attrs, off)); +} + +TEST_F(nfsd_listener, val_inet_short) +{ + struct sockaddr_in s4 = { .sin_family = AF_INET, .sin_port = htons(TEST_PORT) }; + struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, .addr = &s4, + .addr_len = sizeof(sa_family_t) + 2 }; + char attrs[64]; + int off = put_raw_listener(attrs, 0, &r); + + EXPECT_EQ(-EINVAL, listener_set(attrs, off)); +} + +TEST_F(nfsd_listener, val_inet6_short) +{ + struct sockaddr_in6 s6 = { .sin6_family = AF_INET6, .sin6_port = htons(TEST_PORT) }; + struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, .addr = &s6, + .addr_len = sizeof(struct sockaddr_in) }; + char attrs[64]; + int off = put_raw_listener(attrs, 0, &r); + + EXPECT_EQ(-EINVAL, listener_set(attrs, off)); +} + +TEST_F(nfsd_listener, val_bad_family) +{ + struct sockaddr_storage ss = { .ss_family = AF_UNIX }; + struct raw_listener r = { .xprt = "tcp", .emit_addr = 1, .addr = &ss, + .addr_len = sizeof(struct sockaddr_in) }; + char attrs[64]; + int off = put_raw_listener(attrs, 0, &r); + + EXPECT_EQ(-EAFNOSUPPORT, listener_set(attrs, off)); +} + +TEST_F(nfsd_listener, val_second_entry_bad) +{ + struct sockaddr_storage ss = { .ss_family = AF_UNIX }; + struct raw_listener bad = { .xprt = "tcp", .emit_addr = 1, .addr = &ss, + .addr_len = sizeof(struct sockaddr_in) }; + struct listener_ent got[MAX_LISTENERS]; + char attrs[128]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + + off = put_raw_listener(attrs, off, &bad); + /* The whole request is rejected during validation; nothing applied. */ + EXPECT_EQ(-EAFNOSUPPORT, listener_set(attrs, off)); + /* + * Again the errno alone does not say so: svc_xprt_create_from_sa() + * also returns -EAFNOSUPPORT, and the doit keeps the listeners it did + * manage to create, so the well-formed tcp entry ahead of the bad one + * would still be up. + */ + EXPECT_EQ(0, listener_get(got, MAX_LISTENERS)); +} + +/* + * A rejected request must leave the listeners that are already up alone. + * The errno alone does not show that: svc_xprt_create_from_sa() returns + * -EPROTONOSUPPORT for an unknown name too. What differs is how far the + * request gets -- without the check in nfsd_nl_validate_listeners(), + * nfsd_nl_listener_set_doit() has already moved the unmatched tcp listener + * off sv_permsocks and run svc_xprt_destroy_all() on it by the time the + * name fails. + */ +TEST_F(nfsd_listener, val_reject_keeps_listeners) +{ + struct listener_ent got[MAX_LISTENERS]; + char good[64], bad[64]; + int og = put_listener(good, 0, "tcp", TEST_PORT); + int ob = put_listener(bad, 0, "bogus_xprt", TEST_PORT); + + ASSERT_EQ(0, listener_set(good, og)); + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + + EXPECT_EQ(-EPROTONOSUPPORT, listener_set(bad, ob)); + + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT)); +} + +/* ===================== functional / round-trip ===================== */ + +/* LISTENER_GET with no serv in this netns returns an empty list. */ +TEST_F(nfsd_listener, func_get_empty) +{ + struct listener_ent got[MAX_LISTENERS]; + + EXPECT_EQ(0, listener_get(got, MAX_LISTENERS)); +} + +TEST_F(nfsd_listener, func_create_tcp) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + + ASSERT_EQ(0, listener_set(attrs, off)); + EXPECT_STREQ("", last_extack); /* nothing to warn about */ + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT)); +} + +TEST_F(nfsd_listener, func_create_udp) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off = put_listener(attrs, 0, "udp", TEST_PORT); + + ASSERT_EQ(0, listener_set(attrs, off)); + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "udp", AF_INET, TEST_PORT)); +} + +TEST_F(nfsd_listener, func_create_multi) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[128]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + + off = put_listener(attrs, off, "udp", TEST_PORT); + ASSERT_EQ(0, listener_set(attrs, off)); + ASSERT_EQ(2, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 2, "tcp", AF_INET, TEST_PORT)); + EXPECT_NE(NULL, find_listener(got, 2, "udp", AF_INET, TEST_PORT)); +} + +TEST_F(nfsd_listener, func_idempotent) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + + ASSERT_EQ(0, listener_set(attrs, off)); + EXPECT_EQ(0, listener_set(attrs, off)); /* re-set same list */ + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT)); +} + +TEST_F(nfsd_listener, func_add) +{ + struct listener_ent got[MAX_LISTENERS]; + char one[64], two[128]; + int o1 = put_listener(one, 0, "tcp", TEST_PORT); + int o2 = put_listener(two, 0, "tcp", TEST_PORT); + + o2 = put_listener(two, o2, "udp", TEST_PORT); + ASSERT_EQ(0, listener_set(one, o1)); + ASSERT_EQ(0, listener_set(two, o2)); /* add udp, keep tcp */ + ASSERT_EQ(2, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 2, "tcp", AF_INET, TEST_PORT)); + EXPECT_NE(NULL, find_listener(got, 2, "udp", AF_INET, TEST_PORT)); +} + +TEST_F(nfsd_listener, func_remove_subset) +{ + struct listener_ent got[MAX_LISTENERS]; + char both[128], one[64]; + int ob = put_listener(both, 0, "tcp", TEST_PORT); + int oo = put_listener(one, 0, "tcp", TEST_PORT); + + ob = put_listener(both, ob, "udp", TEST_PORT); + ASSERT_EQ(0, listener_set(both, ob)); + ASSERT_EQ(0, listener_set(one, oo)); /* drop udp */ + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT)); +} + +/* + * LISTENER_GET cannot tell a destroyed serv from a live one with no + * permsocks: nfsd_nl_listener_get_doit() replies empty either way. The + * rpcbind client can. nfsd_destroy_serv() is the only path that reaches + * svc_xprt_destroy_all(..., unregister=true) -> svc_rpcb_cleanup() -> + * rpcb_put_local(), which drops the last user and shuts the local client + * down; the next serv then has to connect again. Leaving the serv in place + * would keep the first connection and the stub would see just the one. + */ +TEST_F(nfsd_listener, func_empty_destroys) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + int conns; + + ASSERT_EQ(0, listener_set(attrs, off)); + conns = rpcb_conns(); + ASSERT_GT(conns, 0); + + EXPECT_EQ(0, listener_set(NULL, 0)); /* empty -> destroy serv */ + EXPECT_EQ(0, listener_get(got, MAX_LISTENERS)); + + ASSERT_EQ(0, listener_set(attrs, off)); + EXPECT_GT(rpcb_conns(), conns); +} + +TEST_F(nfsd_listener, func_ipv6) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off, s; + + s = socket(AF_INET6, SOCK_STREAM, 0); + if (s < 0) + SKIP(return, "IPv6 unavailable: %s", strerror(errno)); + close(s); + + off = put_listener_af(attrs, 0, "tcp", AF_INET6, TEST_PORT); + ASSERT_EQ(0, listener_set(attrs, off)); + ASSERT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET6, TEST_PORT)); +} + +/* ===================== rpcbind registration ===================== */ + +/* + * A rpcbind that refuses the registration takes the listener down with it. + * svc_register() fails, so svc_setup_socket() fails, so no listener is + * created. -EACCES alone does not show that, since a bind can return it + * too, so read the listener set back as well. + */ +TEST_F(nfsd_listener, sem_register_refused) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + + rpcb_stub_set_mode(RPCB_STUB_REFUSE); + + EXPECT_EQ(-EACCES, listener_set(attrs, off)); + EXPECT_STRNE("", last_extack); + EXPECT_EQ(0, listener_get(got, MAX_LISTENERS)); +} + +/* + * A listener that cannot be created reports which one it was: the errno + * alone does not name the entry in a multi-listener request. + */ +TEST_F(nfsd_listener, sem_create_failure_extack) +{ + struct sockaddr_in s4 = { .sin_family = AF_INET, + .sin_port = htons(TEST_PORT), + .sin_addr.s_addr = htonl(INADDR_LOOPBACK) }; + struct listener_ent got[MAX_LISTENERS]; + char attrs[64]; + int off = put_listener(attrs, 0, "tcp", TEST_PORT); + int s; + + /* squat on the port so the listener cannot bind */ + s = socket(AF_INET, SOCK_STREAM, 0); + ASSERT_GE(s, 0); + ASSERT_EQ(0, bind(s, (struct sockaddr *)&s4, sizeof(s4))); + + EXPECT_EQ(-EADDRINUSE, listener_set(attrs, off)); + EXPECT_STRNE("", last_extack); + EXPECT_EQ(0, listener_get(got, MAX_LISTENERS)); + close(s); +} + +/* ============ one rpcbind attempt for each request ============ */ + +/* + * Every listener used to register on its own, so a rpcbind that never + * answers cost one timeout for each entry. Ask for one listener, then for + * three, and compare what the stub saw. Three entries must not cost three + * times as much. + * + * The stub has to stay silent rather than refuse. A refusal is an answer, + * and rpcbind refuses one entry at a time, so the count ignores it. + */ +TEST_F(nfsd_listener, rpcb_stop_after_failure) +{ + int before, one, three, off; + char attrs[192]; + + rpcb_stub_set_mode(RPCB_STUB_SILENT); + + before = rpcb_calls(); + off = put_listener(attrs, 0, "tcp", TEST_PORT); + listener_set(attrs, off); + one = rpcb_calls() - before; + ASSERT_GT(one, 0); + + ASSERT_EQ(0, listener_set(attrs, 0)); + + before = rpcb_calls(); + off = put_listener(attrs, 0, "tcp", TEST_PORT); + off = put_listener(attrs, off, "tcp", TEST_PORT + 1); + off = put_listener(attrs, off, "tcp", TEST_PORT + 2); + listener_set(attrs, off); + three = rpcb_calls() - before; + + /* the second and third entries must not reach rpcbind at all */ + EXPECT_LE(three, one); +} + +/* + * The entry that finds rpcbind silent is the one that pays for the + * discovery, and v3 has no vs_rpcb_optnl to discard the error, so it is the + * only entry whose listener would be lost. Nothing distinguishes it from the + * rest of the request, and a retry of the same request would fail the same + * entry again, so the set would stay short for as long as rpcbind was quiet. + * + * Ask for three listeners against a silent stub and require the whole set, + * a success, and a warning that says why. + */ +TEST_F(nfsd_listener, rpcb_silent_set_complete) +{ + struct listener_ent got[MAX_LISTENERS]; + char attrs[192]; + int off; + + rpcb_stub_set_mode(RPCB_STUB_SILENT); + + off = put_listener(attrs, 0, "tcp", TEST_PORT); + off = put_listener(attrs, off, "tcp", TEST_PORT + 1); + off = put_listener(attrs, off, "tcp", TEST_PORT + 2); + EXPECT_EQ(0, listener_set(attrs, off)); + + /* the first entry is not the odd one out */ + EXPECT_EQ(3, listener_get(got, MAX_LISTENERS)); + /* no errno reports this, so the ack has to */ + EXPECT_STRNE("", last_extack); +} + +/* + * The case that needs the count rather than a failed listener. NFSv4 sets + * vs_rpcb_optnl, so svc_generic_rpcbind_set() discards the error, every + * listener comes up, and nothing reports a failure. Without the fix each + * entry still waits for rpcbind on its own. + * + * Make the server v4-only, answer no SET, and require three things: the + * listeners come up, the ack warns that they are not registered, and the + * stub does not see one round trip for each entry. + */ +TEST_F(nfsd_listener, rpcb_v4_only_bounded) +{ + struct listener_ent got[MAX_LISTENERS]; + int before, one, three, off; + char attrs[192]; + + /* refuses once a serv exists, so this has to come first */ + ASSERT_EQ(0, version_set_only(4, 1)); + rpcb_stub_set_mode(RPCB_STUB_SILENT); + + before = rpcb_calls(); + off = put_listener(attrs, 0, "tcp", TEST_PORT); + ASSERT_EQ(0, listener_set(attrs, off)); + one = rpcb_calls() - before; + ASSERT_GT(one, 0); + + /* start over, so the second measurement also builds a serv */ + ASSERT_EQ(0, listener_set(attrs, 0)); + + before = rpcb_calls(); + off = put_listener(attrs, 0, "tcp", TEST_PORT); + off = put_listener(attrs, off, "tcp", TEST_PORT + 1); + off = put_listener(attrs, off, "tcp", TEST_PORT + 2); + ASSERT_EQ(0, listener_set(attrs, off)); + three = rpcb_calls() - before; + + /* the listeners are up even though rpcbind never answered */ + EXPECT_EQ(3, listener_get(got, MAX_LISTENERS)); + /* and the ack says they are unregistered, since no errno can */ + EXPECT_STRNE("", last_extack); + EXPECT_LE(three, one); +} + +/* + * The stop applies to one request only. After rpcbind starts answering, + * the next request must register without any other step. + */ +TEST_F(nfsd_listener, rpcb_retry_next_request) +{ + int before, after, off; + char attrs[192]; + + rpcb_stub_set_mode(RPCB_STUB_SILENT); + + off = put_listener(attrs, 0, "tcp", TEST_PORT); + off = put_listener(attrs, off, "tcp", TEST_PORT + 1); + listener_set(attrs, off); + ASSERT_EQ(0, listener_set(attrs, 0)); + + /* rpcbind recovers */ + rpcb_stub_set_mode(RPCB_STUB_ACCEPT); + + before = rpcb_calls(); + off = put_listener(attrs, 0, "tcp", TEST_PORT); + EXPECT_EQ(0, listener_set(attrs, off)); + after = rpcb_calls(); + + /* a fresh request starts from a fresh reading and tries again */ + EXPECT_GT(after, before); + EXPECT_STREQ("", last_extack); +} + +/* + * The same rule on the way out. Removing a listener unregisters it, so a + * rpcbind that stops answering used to cost one timeout for each listener + * removed. Register one listener while the stub answers, silence the stub, + * remove it and count; then do the same with three. + * + * Both measurements also pay the svc_unregister() sweep that + * nfsd_destroy_serv() runs once the last listener is gone, so that cancels + * out of the comparison. + */ +TEST_F(nfsd_listener, rpcb_unreg_stop_after_failure) +{ + int before, one, three, off; + char attrs[192]; + + off = put_listener(attrs, 0, "tcp", TEST_PORT); + ASSERT_EQ(0, listener_set(attrs, off)); + + rpcb_stub_set_mode(RPCB_STUB_SILENT); + before = rpcb_calls(); + ASSERT_EQ(0, listener_set(NULL, 0)); + one = rpcb_calls() - before; + ASSERT_GT(one, 0); + + rpcb_stub_set_mode(RPCB_STUB_ACCEPT); + off = put_listener(attrs, 0, "tcp", TEST_PORT); + off = put_listener(attrs, off, "tcp", TEST_PORT + 1); + off = put_listener(attrs, off, "tcp", TEST_PORT + 2); + ASSERT_EQ(0, listener_set(attrs, off)); + + rpcb_stub_set_mode(RPCB_STUB_SILENT); + before = rpcb_calls(); + ASSERT_EQ(0, listener_set(NULL, 0)); + three = rpcb_calls() - before; + + /* the second and third removals must not reach rpcbind at all */ + EXPECT_LE(three, one); +} + +/* ===================== threads / -EBUSY semantics ===================== */ + +TEST_F(nfsd_listener, sem_busy_on_change) +{ + struct listener_ent got[MAX_LISTENERS]; + char one[64], two[128]; + int o1 = put_listener(one, 0, "tcp", TEST_PORT); + int o2 = put_listener(two, 0, "tcp", TEST_PORT); + + o2 = put_listener(two, o2, "udp", TEST_PORT); + ASSERT_EQ(0, listener_set(one, o1)); + ASSERT_EQ(0, threads_set(1)); /* threads now running */ + EXPECT_EQ(-EBUSY, listener_set(two, o2)); /* add refused */ + + /* refused means refused: the udp listener must not have been added */ + EXPECT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT)); + + threads_set(0); /* stop before netns exit */ +} + +TEST_F(nfsd_listener, sem_busy_on_remove) +{ + struct listener_ent got[MAX_LISTENERS]; + char one[64]; + int o1 = put_listener(one, 0, "tcp", TEST_PORT); + + ASSERT_EQ(0, listener_set(one, o1)); + ASSERT_EQ(0, threads_set(1)); + EXPECT_EQ(-EBUSY, listener_set(NULL, 0)); /* remove refused */ + + /* the doit moves the permsocks to a temp list before it can fail */ + EXPECT_EQ(1, listener_get(got, MAX_LISTENERS)); + EXPECT_NE(NULL, find_listener(got, 1, "tcp", AF_INET, TEST_PORT)); + + threads_set(0); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/nfsd/settings b/tools/testing/selftests/nfsd/settings new file mode 100644 index 000000000000..6091b45d226b --- /dev/null +++ b/tools/testing/selftests/nfsd/settings @@ -0,0 +1 @@ +timeout=120 diff --git a/tools/testing/selftests/nommu/Makefile b/tools/testing/selftests/nommu/Makefile new file mode 100644 index 000000000000..8e7cd7315c53 --- /dev/null +++ b/tools/testing/selftests/nommu/Makefile @@ -0,0 +1,8 @@ +# SPDX-License-Identifier: GPL-2.0 +# Makefile for nommu selftests + +TEST_GEN_PROGS += nommu_mmap_test +TEST_GEN_PROGS += nommu_mremap_test + +include ../lib.mk +include local.mk diff --git a/tools/testing/selftests/nommu/local.mk b/tools/testing/selftests/nommu/local.mk new file mode 100644 index 000000000000..0bd1300f00f4 --- /dev/null +++ b/tools/testing/selftests/nommu/local.mk @@ -0,0 +1,7 @@ +# detect if users request NOMMU build or not +# User can set NOMMU to 1 to build/test for NOMMU platforms +NOMMU ?= 0 +ifeq ($(NOMMU),1) +CFLAGS += -DNOMMU +export NOMMU +endif diff --git a/tools/testing/selftests/nommu/nommu_mmap_test.c b/tools/testing/selftests/nommu/nommu_mmap_test.c new file mode 100644 index 000000000000..a1f5fdda554a --- /dev/null +++ b/tools/testing/selftests/nommu/nommu_mmap_test.c @@ -0,0 +1,261 @@ +// SPDX-License-Identifier: GPL-2.0 +#define _GNU_SOURCE +#include <stdio.h> +#include <stdlib.h> +#include <sys/mman.h> +#include <unistd.h> +#include <fcntl.h> +#include <errno.h> +#include <string.h> +#include <limits.h> +#include "kselftest.h" + +#include <sys/vfs.h> +#ifndef RAMFS_MAGIC +#define RAMFS_MAGIC 0x858458f6 +#endif + +static size_t ps; + +struct test_case_t { + const char *name; + const char *pathname; + int open_flags; + int mmap_prot; + int mmap_flags; + int exp_err; + int (*resolve_exp_err)(const char *path); +}; + +static int get_shm_expected_error(const char *path) +{ + struct statfs fs; + + if (statfs(path, &fs) == 0) { + if (fs.f_type == RAMFS_MAGIC) + return 0; /* ramfs succeed with contiguous memory */ + } + /* hostfs, etc returns ENODEV due to lack of contiguous allocation */ + return ENODEV; +} + +static struct test_case_t test_cases[] = { + { + .name = "anonymous private allocation", + .pathname = NULL, + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ | PROT_WRITE, + .mmap_flags = MAP_ANONYMOUS | MAP_PRIVATE, + .exp_err = 0, + .resolve_exp_err = NULL, + }, + { + .name = "non-anonymous private file mapping (rw-)", + .pathname = "/tmp/ksft.nommu-reg-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ | PROT_WRITE, + .mmap_flags = MAP_PRIVATE, + .exp_err = 0, + .resolve_exp_err = NULL, + }, + { + .name = "non-anonymous private file mapping (r--)", + .pathname = "/tmp/ksft.nommu-reg-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ, + .mmap_flags = MAP_PRIVATE, + .exp_err = 0, + .resolve_exp_err = NULL, + }, + { + .name = "non-anonymous shared file mapping (rw-)", + .pathname = "/tmp/ksft.nommu-shm-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ | PROT_WRITE, + .mmap_flags = MAP_SHARED, + .exp_err = 0, +#ifdef NOMMU + .resolve_exp_err = get_shm_expected_error, +#else + .resolve_exp_err = NULL, +#endif + }, + { + .name = "non-anonymous shared file mapping (r--)", + .pathname = "/tmp/ksft.nommu-shm-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ, + .mmap_flags = MAP_SHARED, + .exp_err = 0, +#ifdef NOMMU + .resolve_exp_err = get_shm_expected_error, +#else + .resolve_exp_err = 0, +#endif + }, +}; + +static int run_mapping_matrix_test(struct test_case_t *tcase) +{ + int fd; + void *ptr; + char path_buf[PATH_MAX]; + const char *path = tcase->pathname; + int rc = KSFT_PASS; + int expected_error; + + ksft_print_msg("[RUN] %s\n", tcase->name); + + if (tcase->pathname == NULL) { + fd = -1; + } else if (strstr(tcase->pathname, "XXXXXX")) { + strncpy(path_buf, tcase->pathname, sizeof(path_buf) - 1); + path_buf[sizeof(path_buf) - 1] = '\0'; + fd = mkstemp(path_buf); + if (fd < 0) { + ksft_print_msg("Failed to setup temp node: %s\n", + tcase->pathname); + ksft_test_result_skip("%s\n", tcase->name); + return KSFT_SKIP; + } + if (ftruncate(fd, ps) != 0) { + ksft_print_msg("ftruncate failed for: %s\n", tcase->pathname); + ksft_test_result_fail("%s\n", tcase->name); + close(fd); + unlink(path_buf); + return KSFT_FAIL; + } + path = path_buf; + } else { + fd = open(tcase->pathname, tcase->open_flags, 0600); + if (fd < 0) { + ksft_print_msg("Device node not accessible: %s\n", + tcase->pathname); + ksft_test_result_skip("%s\n", tcase->name); + return KSFT_SKIP; + } + } + + expected_error = tcase->exp_err; + if (tcase->resolve_exp_err && fd >= 0) + expected_error = tcase->resolve_exp_err(path); + + ptr = mmap(NULL, ps, tcase->mmap_prot, tcase->mmap_flags, fd, 0); + + if (expected_error != 0) { + if (ptr != MAP_FAILED) { + ksft_print_msg("mmap unexpectedly succeeded (exp error %d)\n", + expected_error); + ksft_test_result_fail("%s\n", tcase->name); + munmap(ptr, ps); + rc = KSFT_FAIL; + goto cleanup; + } + if (errno != expected_error) { + ksft_print_msg("mmap failed with %d (%s), but expected %d\n", + errno, strerror(errno), expected_error); + ksft_test_result_fail("%s\n", tcase->name); + rc = KSFT_FAIL; + goto cleanup; + } + ksft_print_msg("Correctly rejected with expected error %s(%d)\n", + strerror(expected_error), expected_error); + ksft_test_result_pass("%s\n", tcase->name); + rc = KSFT_PASS; + goto cleanup; + } + + if (ptr == MAP_FAILED) { + ksft_print_msg("mmap failed unexpectedly: %s\n", strerror(errno)); + ksft_test_result_fail("%s\n", tcase->name); + rc = KSFT_FAIL; + goto cleanup; + } + + ksft_test_result_pass("%s\n", tcase->name); + munmap(ptr, ps); + +cleanup: + if (fd >= 0) { + close(fd); + if (tcase->pathname && strstr(tcase->pathname, "XXXXXX")) + unlink(path_buf); + } + return rc; +} + +static int test_map_fixed(void) +{ + void *fixed_addr; + void *ptr; + + ksft_print_msg("[RUN] %s\n", __func__); + + fixed_addr = mmap(NULL, ps, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (fixed_addr == MAP_FAILED) { + ksft_print_msg("Unable to reserve test address: %s\n", + strerror(errno)); + ksft_test_result_skip("MAP_FIXED behavior\n"); + return KSFT_SKIP; + } + + if (munmap(fixed_addr, ps)) { + ksft_print_msg("Unable to release test address: %s\n", + strerror(errno)); + ksft_test_result_fail("MAP_FIXED behavior\n"); + return KSFT_FAIL; + } + + ptr = mmap(fixed_addr, ps, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + +#ifdef NOMMU + if (ptr == MAP_FAILED && (errno == ENODEV || errno == EINVAL)) { + ksft_print_msg("MAP_FIXED correctly rejected under nommu\n"); + ksft_test_result_pass("MAP_FIXED behavior\n"); + return KSFT_PASS; + } + if (ptr != MAP_FAILED) { + ksft_print_msg("MAP_FIXED unexpectedly allowed under nommu\n"); + ksft_test_result_fail("MAP_FIXED behavior\n"); + munmap(ptr, ps); + return KSFT_FAIL; + } + ksft_print_msg("MAP_FIXED failed under NOMMU: %s\n", + strerror(errno)); + ksft_test_result_fail("MAP_FIXED behavior\n"); + return KSFT_FAIL; +#else + if (ptr != MAP_FAILED) { + ksft_print_msg("MAP_FIXED successfully allocated under MMU\n"); + ksft_test_result_pass("MAP_FIXED behavior\n"); + munmap(ptr, ps); + return KSFT_PASS; + } + ksft_print_msg("MAP_FIXED failed allocation under MMU\n"); + ksft_test_result_fail("MAP_FIXED behavior\n"); + return KSFT_FAIL; +#endif +} + +int main(int argc, char **argv) +{ + int i; + + ps = sysconf(_SC_PAGESIZE); + ksft_print_header(); + ksft_set_plan(ARRAY_SIZE(test_cases) + 1); + +#ifdef NOMMU + ksft_print_msg("Running strict MMAP test criteria under nommu architecture\n"); +#else + ksft_print_msg("Running MMAP test criteria under MMU architecture\n"); +#endif + + test_map_fixed(); + for (i = 0; i < (int)ARRAY_SIZE(test_cases); i++) + run_mapping_matrix_test(&test_cases[i]); + + ksft_finished(); +} diff --git a/tools/testing/selftests/nommu/nommu_mremap_test.c b/tools/testing/selftests/nommu/nommu_mremap_test.c new file mode 100644 index 000000000000..7ccdf65b675f --- /dev/null +++ b/tools/testing/selftests/nommu/nommu_mremap_test.c @@ -0,0 +1,366 @@ +// SPDX-License-Identifier: GPL-2.0 +#define _GNU_SOURCE +#include <stdio.h> +#include <stdlib.h> +#include <sys/mman.h> +#include <unistd.h> +#include <fcntl.h> +#include <errno.h> +#include <string.h> +#include <limits.h> +#include "kselftest.h" + +#include <sys/vfs.h> +#ifndef RAMFS_MAGIC +#define RAMFS_MAGIC 0x858458f6 +#endif + +static size_t ps; + +static long get_fs_type(const char *path) +{ + struct statfs fs; + + if (statfs(path, &fs) == 0) + return fs.f_type; + + return 0; +} + +static void munmap_shrink_test(void) +{ + void *addr; + int ret; + + /* munmap shrink test */ + for (int i = 0; i < 4; i++) { + addr = mmap(NULL, ps * 4, PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE, -1, 0); + if (addr == MAP_FAILED) { + ksft_print_msg("mmap failed: %s(%d)\n", strerror(errno), errno); + ksft_test_result_fail("munmap shrink\n"); + return; + } + ret = munmap((char *)addr + ps * i, ps); + if (ret != 0) { + ksft_print_msg("memory %p isn't unmapped at %p\n", + addr, (char *)addr + ps * i); + ksft_test_result_fail("munmap shrink\n"); + return; + } + + if (i == 0) { + if (munmap(addr + ps, ps * 3)) + goto error; + } else if (i == 1) { + if (munmap(addr, ps) || munmap(addr + (ps * 2), ps * 2)) + goto error; + } else if (i == 2) { + if (munmap(addr, ps * 2) || munmap(addr + (ps * 3), ps)) + goto error; + } else if (i == 3) { + if (munmap(addr, ps * 3)) + goto error; + } + } + + ksft_test_result_pass("munmap shrink\n"); + return; +error: + for (int j = 0; j < 4; j++) + munmap((char *)addr + j * ps, ps); + ksft_print_msg("clean up failures\n"); + ksft_test_result_fail("munmap shrink\n"); +} + +static size_t page_align(size_t len) +{ + return (len + ps - 1) / ps * ps; +} + +static void mremap_shrink_test(void) +{ + void *addr, *addr2; + size_t current_len; + size_t old_len, new_len; + struct param { + size_t old; + size_t new; + } params[] = { + /* should not happen any shrink */ + { .old = ps * 4 - 1, .new = ps * 4 - 2 }, + /* should not happen any shrink */ + { .old = ps * 4 - 1, .new = ps * 4 }, + { .old = ps * 4, .new = ps * 2 }, + /* should not happen any shrink */ + { .old = ps * 2, .new = ps * 2 - 2 }, + { .old = ps * 2 - 2, .new = ps * 1 }, + }; + + /* mremap shrink test */ + current_len = page_align(ps * 4 - 1); + addr = mmap(NULL, ps * 4 - 1, PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE, -1, 0); + if (addr == MAP_FAILED) { + ksft_print_msg("mmap failed: %s(%d)\n", strerror(errno), errno); + ksft_test_result_fail("mremap shrink\n"); + return; + } + + for (int i = 0; i < ARRAY_SIZE(params); i++) { + old_len = params[i].old; + new_len = params[i].new; + current_len = page_align(new_len); + addr2 = mremap(addr, old_len, new_len, MREMAP_MAYMOVE); + if (addr2 == MAP_FAILED) { + ksft_print_msg("memory %p isn't remapped at %p\n", addr, addr2); + ksft_test_result_fail("mremap shrink\n"); + munmap(addr, page_align(old_len)); + return; + } + + addr = addr2; + } + + if (munmap(addr, current_len)) { + ksft_print_msg("cleanup failed: %s\n", strerror(errno)); + ksft_test_result_fail("mremap shrink\n"); + return; + } + ksft_test_result_pass("mremap shrink\n"); +} + +static int get_shared_writable_file_expected_error(const char *path) +{ + if (get_fs_type(path) == RAMFS_MAGIC) + return EPERM; /* ramfs failed */ + + return 0; +} + +struct mremap_case_t { + const char *name; + const char *pathname; + int open_flags; + int mmap_prot; + int mmap_flags; + int exp_err; + int (*resolve_exp_err)(const char *path); + unsigned int old_pages; + unsigned int new_pages; +}; + +static struct mremap_case_t mremap_cases[] = { + { + .name = "anonymous shrink (r--)", + .pathname = NULL, + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ, + .mmap_flags = MAP_ANONYMOUS | MAP_PRIVATE, + .exp_err = 0, + .resolve_exp_err = 0, + }, + { + .name = "shared file shrink (r--)", + .pathname = "/tmp/ksft.nommu-remap-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ, + .mmap_flags = MAP_SHARED, + .exp_err = 0, +#ifdef NOMMU + .resolve_exp_err = get_shared_writable_file_expected_error, +#else + .resolve_exp_err = 0, +#endif + }, + { + .name = "private file unchanged length (r-)", + .pathname = "/tmp/ksft.nommu-remap-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ, + .mmap_flags = MAP_PRIVATE, +#ifdef NOMMU + .exp_err = EPERM, +#else + .exp_err = 0, +#endif + .resolve_exp_err = 0, + .old_pages = 4, + .new_pages = 4, + }, + { + .name = "private file unchanged length (rw-)", + .pathname = "/tmp/ksft.nommu-remap-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ | PROT_WRITE, + .mmap_flags = MAP_PRIVATE, + .exp_err = 0, + .resolve_exp_err = 0, + .old_pages = 4, + .new_pages = 4, + }, + { + .name = "private file growth (r-)", + .pathname = "/tmp/ksft.nommu-remap-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ, + .mmap_flags = MAP_PRIVATE, +#ifdef NOMMU + .exp_err = EPERM, +#else + .exp_err = 0, +#endif + .resolve_exp_err = 0, + .old_pages = 4, + .new_pages = 8, + }, + { + .name = "private file growth (rw-)", + .pathname = "/tmp/ksft.nommu-remap-XXXXXX", + .open_flags = O_CREAT | O_RDWR | O_EXCL, + .mmap_prot = PROT_READ | PROT_WRITE, + .mmap_flags = MAP_PRIVATE, +#ifdef NOMMU + .exp_err = ENOMEM, +#else + .exp_err = 0, +#endif + .resolve_exp_err = 0, + .old_pages = 4, + .new_pages = 8, + }, +}; + +static int run_mremap_test(struct mremap_case_t *tcase) +{ + int fd = -1; + void *addr, *addr2; + char pb[PATH_MAX]; + const char *path = tcase->pathname; + int rc = KSFT_PASS; + int expected_error; + unsigned int old_pages = tcase->old_pages ?: 4; + unsigned int new_pages = tcase->new_pages ?: 2; + unsigned int file_pages = old_pages > new_pages ? + old_pages : new_pages; + + ksft_print_msg("[RUN] Testing mremap: %s\n", tcase->name); + + if (tcase->pathname && strstr(tcase->pathname, "XXXXXX")) { + strncpy(pb, tcase->pathname, sizeof(pb) - 1); + pb[sizeof(pb) - 1] = '\0'; + fd = mkstemp(pb); + if (fd < 0) { + ksft_print_msg("Failed to setup file backing\n"); + ksft_test_result_skip("%s\n", tcase->name); + return KSFT_SKIP; + } + if (ftruncate(fd, ps * file_pages) != 0) { + ksft_print_msg("Failed to setup file backing\n"); + ksft_test_result_fail("%s\n", tcase->name); + close(fd); + unlink(pb); + return KSFT_FAIL; + } + +#ifdef NOMMU + if ((tcase->mmap_flags & MAP_SHARED) && get_fs_type(pb) != RAMFS_MAGIC) { + ksft_print_msg("Skip the test under non-ramfs filesystem (%s)\n", + pb); + ksft_test_result_skip("%s\n", tcase->name); + close(fd); + unlink(pb); + return KSFT_SKIP; + } +#endif + path = pb; + } else if (tcase->pathname) { + fd = open(tcase->pathname, tcase->open_flags, 0600); + if (fd < 0) { + ksft_print_msg("Backing node not accessible\n"); + ksft_test_result_skip("%s\n", tcase->name); + return KSFT_SKIP; + } + +#ifdef NOMMU + if ((tcase->mmap_flags & MAP_SHARED) && + get_fs_type(tcase->pathname) != RAMFS_MAGIC) { + ksft_print_msg("Skip the test under non-ramfs filesystem (%s)\n", + tcase->pathname); + ksft_test_result_skip("%s\n", tcase->name); + close(fd); + return KSFT_SKIP; + } +#endif + } + + addr = mmap(NULL, ps * old_pages, tcase->mmap_prot, + tcase->mmap_flags, fd, 0); + if (addr == MAP_FAILED) { + ksft_print_msg("mmap mapping failed %s(%d)\n", strerror(errno), errno); + rc = KSFT_FAIL; + goto out; + } + + expected_error = tcase->exp_err; + if (tcase->resolve_exp_err && fd >= 0) + expected_error = tcase->resolve_exp_err(path); + + addr2 = mremap(addr, ps * old_pages, ps * new_pages, + MREMAP_MAYMOVE); + + if (expected_error != 0) { + if (addr2 != MAP_FAILED) { + ksft_print_msg("Expected error %d, but mremap unexpectedly succeeded\n", + expected_error); + rc = KSFT_FAIL; + } else if (errno != expected_error) { + ksft_print_msg("Expected error %d, got %s(%d)\n", + expected_error, strerror(errno), errno); + rc = KSFT_FAIL; + } else { + ksft_print_msg("%s: Handled expected error path (errno=%d)\n", + tcase->name, expected_error); + } + } else if (addr2 == MAP_FAILED) { + ksft_print_msg("mremap shrink failed unexpectedly: %s\n", + strerror(errno)); + rc = KSFT_FAIL; + } else { + ksft_print_msg("%s step successful\n", tcase->name); + } + + /* clean up */ + if (munmap(addr2 == MAP_FAILED ? addr : addr2, + addr2 == MAP_FAILED ? ps * old_pages : ps * new_pages)) { + ksft_print_msg("munmap failed: %s\n", strerror(errno)); + rc = KSFT_FAIL; + } + +out: + if (fd >= 0) { + close(fd); + if (tcase->pathname && strstr(tcase->pathname, "XXXXXX")) + unlink(pb); + } + + ksft_test_result_report(rc, "%s\n", tcase->name); + return rc; +} + +int main(int argc, char **argv) +{ + int i; + + ps = sysconf(_SC_PAGESIZE); + ksft_print_header(); + ksft_set_plan(ARRAY_SIZE(mremap_cases) + 2); + + munmap_shrink_test(); + mremap_shrink_test(); + + for (i = 0; i < (int)ARRAY_SIZE(mremap_cases); i++) + run_mremap_test(&mremap_cases[i]); + + ksft_finished(); +} diff --git a/tools/testing/selftests/seccomp/seccomp_bpf.c b/tools/testing/selftests/seccomp/seccomp_bpf.c index 891383a161d0..794336aaabb5 100644 --- a/tools/testing/selftests/seccomp/seccomp_bpf.c +++ b/tools/testing/selftests/seccomp/seccomp_bpf.c @@ -307,6 +307,10 @@ struct seccomp_notif_addfd_big { #define SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV (1UL << 5) #endif +#ifndef SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV +#define SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV (1UL << 6) +#endif + #ifndef seccomp int seccomp(unsigned int op, unsigned int flags, void *args) { @@ -4298,6 +4302,65 @@ TEST(user_notification_addfd) close(memfd); } +TEST(user_notification_addfd_opath) +{ + struct seccomp_notif req = {}; + struct seccomp_notif_addfd addfd = {}; + struct seccomp_notif_resp resp = {}; + pid_t pid; + int listener, pathfd, fd, status; + + pathfd = open("/dev/null", O_PATH | O_CLOEXEC); + ASSERT_GE(pathfd, 0); + ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0); + listener = user_notif_syscall(__NR_getppid, + SECCOMP_FILTER_FLAG_NEW_LISTENER); + ASSERT_GE(listener, 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + int flags; + char byte; + + close(pathfd); + fd = syscall(__NR_getppid); + if (fd < 0) + _exit(1); + flags = fcntl(fd, F_GETFL); + if (flags < 0 || !(flags & O_PATH)) + _exit(2); + flags = fcntl(fd, F_GETFD); + if (flags < 0 || !(flags & FD_CLOEXEC)) + _exit(3); + errno = 0; + if (read(fd, &byte, 1) != -1 || errno != EBADF) + _exit(4); + _exit(0); + } + + ASSERT_EQ(ioctl(listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0); + addfd.id = req.id; + addfd.flags = SECCOMP_ADDFD_FLAG_SEND; + addfd.srcfd = pathfd; + addfd.newfd_flags = O_CLOEXEC; + fd = ioctl(listener, SECCOMP_IOCTL_NOTIF_ADDFD, &addfd); + if (fd < 0) { + resp.id = req.id; + resp.error = -errno; + TH_LOG("ADDFD failed with errno %d", -resp.error); + ASSERT_EQ(ioctl(listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0); + } + + ASSERT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_GE(fd, 0); + EXPECT_EQ(true, WIFEXITED(status)); + if (WIFEXITED(status)) + EXPECT_EQ(0, WEXITSTATUS(status)); + close(pathfd); + close(listener); +} + TEST(user_notification_addfd_rlimit) { pid_t pid; @@ -4818,6 +4881,356 @@ static long get_proc_syscall(struct __test_metadata *_metadata, int pid) return ret; } + +static void notification_restart_handler(int sig) +{ + char c; + int saved_errno = errno; + + if (write(handled, "s", 1) != 1 || read(handled, &c, 1) != 1) + _exit(1); + errno = saved_errno; +} + +FIXTURE(notification_restart) { + int listener; + int sync[2]; + pid_t pid; +}; + +FIXTURE_VARIANT(notification_restart) { + bool restart; + bool killable; +}; + +FIXTURE_VARIANT_ADD(notification_restart, neither) { + .restart = false, .killable = false, +}; +FIXTURE_VARIANT_ADD(notification_restart, restart) { + .restart = true, .killable = false, +}; +FIXTURE_VARIANT_ADD(notification_restart, killable) { + .restart = false, .killable = true, +}; +FIXTURE_VARIANT_ADD(notification_restart, both) { + .restart = true, .killable = true, +}; + +FIXTURE_SETUP(notification_restart) +{ + unsigned int flags = SECCOMP_FILTER_FLAG_NEW_LISTENER; + + self->pid = -1; + self->listener = -1; + self->sync[0] = self->sync[1] = -1; + ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0); + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM, 0, self->sync), 0); + if (variant->restart) + flags |= SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV; + if (variant->killable) + flags |= SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV; + self->listener = user_notif_syscall(__NR_getppid, flags); + ASSERT_GE(self->listener, 0); +} + +FIXTURE_TEARDOWN(notification_restart) +{ + if (self->pid > 0) { + kill(self->pid, SIGKILL); + waitpid(self->pid, NULL, 0); + } + close(self->listener); + close(self->sync[0]); + close(self->sync[1]); +} + +static void notification_restart_child(struct __test_metadata *_metadata, + struct _test_data_notification_restart *self) +{ + struct sigaction action = { .sa_handler = notification_restart_handler }; + long result[2]; + + self->pid = fork(); + ASSERT_GE(self->pid, 0); + if (self->pid) + return; + + close(self->listener); + close(self->sync[0]); + handled = self->sync[1]; + if (sigemptyset(&action.sa_mask) || sigaction(SIGUSR1, &action, NULL)) + _exit(1); + result[0] = syscall(__NR_getppid); + result[1] = errno; + if (write(handled, result, sizeof(result)) != sizeof(result)) + _exit(1); + _exit(0); +} + +static void notification_pending(struct __test_metadata *_metadata, int fd) +{ + struct pollfd pfd = { .fd = fd, .events = POLLIN }; + + ASSERT_EQ(poll(&pfd, 1, 5000), 1); + ASSERT_TRUE(pfd.revents & POLLIN); +} + +static void notification_signal(struct __test_metadata *_metadata, + struct _test_data_notification_restart *self) +{ + struct pollfd pfd = { .fd = self->sync[0], .events = POLLIN }; + char c; + + ASSERT_EQ(kill(self->pid, SIGUSR1), 0); + ASSERT_EQ(poll(&pfd, 1, 5000), 1); + ASSERT_EQ(read(self->sync[0], &c, 1), 1); + ASSERT_EQ(c, 's'); + /* The handler holds the task until the abandoned request is checked. */ + pfd.fd = self->listener; + ASSERT_EQ(poll(&pfd, 1, 0), 0); + ASSERT_EQ(write(self->sync[0], "r", 1), 1); +} + +static void notification_result(struct __test_metadata *_metadata, + struct _test_data_notification_restart *self, + long value, int error) +{ + long result[2]; + int status; + + ASSERT_EQ(read(self->sync[0], result, sizeof(result)), sizeof(result)); + EXPECT_EQ(result[0], value); + if (value == -1) + EXPECT_EQ(result[1], error); + ASSERT_EQ(waitpid(self->pid, &status, 0), self->pid); + self->pid = -1; + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(WEXITSTATUS(status), 0); +} + +TEST_F(notification_restart, before_receive) +{ + struct seccomp_notif req = {}; + struct seccomp_notif_resp resp = {}; + int i; + + notification_restart_child(_metadata, self); + for (i = 0; i < 3; i++) { + notification_pending(_metadata, self->listener); + notification_signal(_metadata, self); + if (!variant->restart) { + notification_result(_metadata, self, -1, EINTR); + return; + } + } + notification_pending(_metadata, self->listener); + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0); + resp.id = req.id; + resp.flags = SECCOMP_USER_NOTIF_FLAG_CONTINUE; + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0); + notification_result(_metadata, self, getpid(), 0); +} + +TEST_F(notification_restart, failed_receive) +{ + struct seccomp_notif_resp resp = {}; + struct seccomp_notif req = {}; + void *buf; + + buf = mmap(NULL, sizeof(req), PROT_READ, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + ASSERT_NE(buf, MAP_FAILED); + notification_restart_child(_metadata, self); + notification_pending(_metadata, self->listener); + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, buf), -1); + ASSERT_EQ(errno, EFAULT); + ASSERT_EQ(munmap(buf, sizeof(req)), 0); + notification_signal(_metadata, self); + if (!variant->restart) { + notification_result(_metadata, self, -1, EINTR); + return; + } + notification_pending(_metadata, self->listener); + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0); + resp.id = req.id; + resp.error = -EAGAIN; + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0); + notification_result(_metadata, self, -1, EAGAIN); +} + +TEST_F(notification_restart, after_receive) +{ + struct seccomp_notif req = {}; + struct seccomp_notif_resp resp = {}; + char c; + + notification_restart_child(_metadata, self); + notification_pending(_metadata, self->listener); + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0); + if (!variant->killable) { + notification_signal(_metadata, self); + notification_result(_metadata, self, -1, EINTR); + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_ID_VALID, &req.id), -1); + EXPECT_EQ(errno, ENOENT); + return; + } + ASSERT_EQ(kill(self->pid, SIGUSR1), 0); + /* Either ordering of signal delivery and reply must preserve the response. */ + resp.id = req.id; + resp.val = USER_NOTIF_MAGIC; + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0); + ASSERT_EQ(read(self->sync[0], &c, 1), 1); + ASSERT_EQ(c, 's'); + ASSERT_EQ(write(self->sync[0], "r", 1), 1); + notification_result(_metadata, self, USER_NOTIF_MAGIC, 0); +} + +TEST_F(notification_restart, fork_and_close) +{ + struct sigaction action = { .sa_handler = notification_restart_handler }; + struct sock_filter filter[] = { + BPF_STMT(BPF_LD | BPF_W | BPF_ABS, offsetof(struct seccomp_data, nr)), +#ifdef __NR_fork + BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_fork, 0, 1), + BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF), +#endif +#ifdef __NR_clone + BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_clone, 0, 1), + BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF), +#endif +#ifdef __NR_clone3 + BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_clone3, 0, 1), + BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF), +#endif + BPF_JUMP(BPF_JMP | BPF_JEQ | BPF_K, __NR_close, 0, 1), + BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_USER_NOTIF), + BPF_STMT(BPF_RET | BPF_K, SECCOMP_RET_ALLOW), + }; + struct sock_fprog prog = { .len = ARRAY_SIZE(filter), .filter = filter }; + char control[CMSG_SPACE(sizeof(int))] = {}; + char c = 'f'; + struct iovec iov = { .iov_base = &c, .iov_len = 1 }; + struct msghdr msg = { + .msg_iov = &iov, .msg_iovlen = 1, + .msg_control = control, .msg_controllen = sizeof(control), + }; + struct cmsghdr *cmsg; + unsigned int flags = SECCOMP_FILTER_FLAG_NEW_LISTENER; + int i, fd, listener, status; + long result[2] = {}; + pid_t child; + + if (variant->restart) + flags |= SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV; + if (variant->killable) + flags |= SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV; + ASSERT_EQ(close(self->listener), 0); + self->listener = -1; + self->pid = fork(); + ASSERT_GE(self->pid, 0); + if (!self->pid) { + close(self->sync[0]); + handled = self->sync[1]; + ASSERT_EQ(sigemptyset(&action.sa_mask), 0); + ASSERT_EQ(sigaction(SIGUSR1, &action, NULL), 0); + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0); + listener = seccomp(SECCOMP_SET_MODE_FILTER, flags, &prog); + ASSERT_GE(listener, 0); + cmsg = CMSG_FIRSTHDR(&msg); + cmsg->cmsg_level = SOL_SOCKET; + cmsg->cmsg_type = SCM_RIGHTS; + cmsg->cmsg_len = CMSG_LEN(sizeof(listener)); + memcpy(CMSG_DATA(cmsg), &listener, sizeof(listener)); + ASSERT_EQ(sendmsg(handled, &msg, 0), 1); + + child = fork(); + if (!child) + _exit(0); + if (variant->restart) { + ASSERT_GT(child, 0); + ASSERT_EQ(waitpid(child, &status, 0), child); + ASSERT_TRUE(WIFEXITED(status)); + ASSERT_EQ(WEXITSTATUS(status), 0); + ASSERT_EQ(waitpid(-1, &status, WNOHANG), -1); + ASSERT_EQ(errno, ECHILD); + ASSERT_EQ(close(fd), 0); + ASSERT_EQ(fcntl(fd, F_GETFD), -1); + ASSERT_EQ(errno, EBADF); + } else { + ASSERT_EQ(child, -1); + ASSERT_EQ(errno, EINTR); + ASSERT_EQ(close(fd), -1); + ASSERT_EQ(errno, EINTR); + ASSERT_GE(fcntl(fd, F_GETFD), 0); + } + + ASSERT_EQ(fork(), -1); + ASSERT_EQ(errno, EAGAIN); + ASSERT_EQ(write(handled, result, sizeof(result)), sizeof(result)); + _exit(0); + } + ASSERT_EQ(recvmsg(self->sync[0], &msg, 0), 1); + ASSERT_FALSE(msg.msg_flags & MSG_CTRUNC); + cmsg = CMSG_FIRSTHDR(&msg); + ASSERT_NE(cmsg, NULL); + ASSERT_EQ(cmsg->cmsg_level, SOL_SOCKET); + ASSERT_EQ(cmsg->cmsg_type, SCM_RIGHTS); + ASSERT_EQ(cmsg->cmsg_len, CMSG_LEN(sizeof(listener))); + memcpy(&self->listener, CMSG_DATA(cmsg), sizeof(self->listener)); + + for (i = 0; i < 3; i++) { + struct seccomp_notif req = {}; + struct seccomp_notif_resp resp = {}; + + notification_pending(_metadata, self->listener); + if (i < 2 || variant->restart) { + notification_signal(_metadata, self); + if (!variant->restart) + continue; + notification_pending(_metadata, self->listener); + } + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_RECV, &req), 0); + EXPECT_EQ(req.pid, self->pid); + resp.id = req.id; + if (i == 2) + resp.error = -EAGAIN; + else + resp.flags = SECCOMP_USER_NOTIF_FLAG_CONTINUE; + ASSERT_EQ(ioctl(self->listener, SECCOMP_IOCTL_NOTIF_SEND, &resp), 0); + } + notification_result(_metadata, self, 0, 0); +} + +TEST_F(notification_restart, fatal_signal) +{ + int status; + + notification_restart_child(_metadata, self); + notification_pending(_metadata, self->listener); + ASSERT_EQ(kill(self->pid, SIGKILL), 0); + ASSERT_EQ(waitpid(self->pid, &status, 0), self->pid); + self->pid = -1; + ASSERT_TRUE(WIFSIGNALED(status)); + EXPECT_EQ(WTERMSIG(status), SIGKILL); +} + +TEST_F(notification_restart, listener_closed) +{ + notification_restart_child(_metadata, self); + notification_pending(_metadata, self->listener); + ASSERT_EQ(close(self->listener), 0); + self->listener = -1; + notification_result(_metadata, self, -1, ENOSYS); +} + +TEST(user_notification_restart_requires_listener) +{ + ASSERT_EQ(prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0), 0); + EXPECT_EQ(user_notif_syscall(__NR_getppid, + SECCOMP_FILTER_FLAG_RESTART_BEFORE_RECV), -1); + EXPECT_EQ(errno, EINVAL); +} + /* Ensure non-fatal signals prior to receive are unmodified */ TEST(user_notification_wait_killable_pre_notification) { diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index a4e3d30b2fc1..e47be744a149 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1358,23 +1358,23 @@ static inline int vfs_mmap_prepare(struct file *file, struct vm_area_desc *desc) return file->f_op->mmap_prepare(desc); } -int mmap_prepare_validate(const struct vm_area_desc *prev_desc, +int mmap_prepare_validate(const struct vm_area_desc *orig_desc, const struct vm_area_desc *desc); static inline int __compat_vma_mmap(struct vm_area_desc *desc, struct vm_area_struct *vma) { - struct vm_area_desc prev_desc; + struct vm_area_desc orig_desc; int err; /* Derive state prior to mmap_prepare hook. */ - compat_set_desc_from_vma(&prev_desc, desc->file, vma); + compat_set_desc_from_vma(&orig_desc, desc->file, vma); /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); if (err) return err; /* Check the caller did nothing crazy. */ - err = mmap_prepare_validate(&prev_desc, desc); + err = mmap_prepare_validate(&orig_desc, desc); if (err) return err; /* Update the VMA from the descriptor. */ @@ -1657,14 +1657,14 @@ static inline bool file_is_dev_zero(const struct file *file) return file && file->f_op == &zero_fops; } -static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) +static inline bool vma_flags_is_mm_managed(const vma_flags_t *flags) { - return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); + return !vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } -static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) +static inline bool vma_is_mm_managed(const struct vm_area_struct *vma) { - return vma_flags_is_kernel_owned(&vma->flags); + return vma_flags_is_mm_managed(&vma->flags); } static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags) @@ -1687,13 +1687,13 @@ static inline bool vma_flags_can_merge(const vma_flags_t *flags) * VMA merging assumes that a VMA's flags and fields completely describe * its state. * - * However, kernel-owned mappings may have established state upon mapping - * not embodied in any attribute of the VMA. + * However, mappings which are not mm-managed may have established state + * upon mapping not embodied in any attribute of the VMA. * * Additionally, private (CoW) PFN maps encode the source PFN of the * range in vma->vm_pgoff, which may otherwise cause spurious merges. */ - if (vma_flags_is_kernel_owned(flags)) + if (!vma_flags_is_mm_managed(flags)) return false; /* VMA explicitly marked as being unmergeable. */ if (vma_flags_is_fixed_mapping(flags)) |
