Add trace tests for the landlock_enforce_domain event in trace_test.c, asserting field counts after the syscall returns rather than line ordering across per-CPU buffers. They cover single-threaded and TSYNC enforcement (complete and process_wide set), a multi-threaded non-TSYNC process (process_wide clear), the single-threaded non-leader edge case, the flags-only path that creates no domain, and a thread-sync abort that emits create_domain and free_domain but no enforce_domain.
landlock_enforce_domain is added to the fixture enable path and every disable list so its zero-events assertions cannot be tripped by a stray enforcement event. Test coverage for security/landlock is 91.6% of 2571 lines according to LLVM 22. Cc: Günther Noack <[email protected]> Cc: Tingmao Wang <[email protected]> Signed-off-by: Mickaël Salaün <[email protected]> --- Changes since v3: https://patch.msgid.link/[email protected] - Rework the enforce_domain field tests into a trace_enforce FIXTURE_VARIANT table (variants single, no_new_privs, tsync_single, tsync_multithread, tsync_no_new_privs, multithread_non_tsync), each checking the total/complete/process_wide/no_new_privs event counts for a (nthreads, flags) input. This covers the new no_new_privs field and both ways it is set: the NO_NEW_PRIVS flag (which makes child_enforce() skip prctl()) or a prior prctl(), each propagated to the siblings by TSYNC; count_enforce_matches() gains a no_new_privs regex field. enforce_flags_only, enforce_single_non_leader, and enforce_abort stay standalone (distinct scaffolding). - Repurpose enforce_single_non_leader: an un-reaped zombie group leader keeps get_nr_threads() at 2, so a single-live-thread group is unreachable and the test always skipped; assert the reachable behavior instead, a non-leader enforcing while the zombie leader is still counted reports process_wide=0. Drop the now-unused read_thread_count()/ENFORCE_SKIP_EXIT helpers. - Raise enforce_abort's sibling count to 200 (matching tsync_test's NUM_IDLE_THREADS) so the thread-sync abort race has a window to fire on multi-core hosts instead of almost never landing with a handful of threads; the test still skips gracefully where the race does not occur (e.g. single-CPU UML). - Drop the unused, stale REGEX_ENFORCE_DOMAIN macro, superseded by count_enforce_matches() (it also lacked the no_new_privs field). Changes since v2: - New patch. --- tools/testing/selftests/landlock/trace.h | 2 + tools/testing/selftests/landlock/trace_test.c | 519 ++++++++++++++++++ 2 files changed, 521 insertions(+) diff --git a/tools/testing/selftests/landlock/trace.h b/tools/testing/selftests/landlock/trace.h index 4c82ecffc235..ba0c5e92001f 100644 --- a/tools/testing/selftests/landlock/trace.h +++ b/tools/testing/selftests/landlock/trace.h @@ -25,6 +25,8 @@ TRACEFS_LANDLOCK_DIR "/landlock_create_ruleset/enable" #define TRACEFS_CREATE_DOMAIN_ENABLE \ TRACEFS_LANDLOCK_DIR "/landlock_create_domain/enable" +#define TRACEFS_ENFORCE_DOMAIN_ENABLE \ + TRACEFS_LANDLOCK_DIR "/landlock_enforce_domain/enable" #define TRACEFS_ADD_RULE_FS_ENABLE \ TRACEFS_LANDLOCK_DIR "/landlock_add_rule_fs/enable" #define TRACEFS_ADD_RULE_NET_ENABLE \ diff --git a/tools/testing/selftests/landlock/trace_test.c b/tools/testing/selftests/landlock/trace_test.c index a141f22ad98f..afdaf8511b3a 100644 --- a/tools/testing/selftests/landlock/trace_test.c +++ b/tools/testing/selftests/landlock/trace_test.c @@ -9,6 +9,7 @@ #include <errno.h> #include <fcntl.h> #include <linux/landlock.h> +#include <pthread.h> #include <sched.h> #include <stdio.h> #include <string.h> @@ -47,6 +48,7 @@ FIXTURE_SETUP(trace) ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, true)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NET_ENABLE, true)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, true)); @@ -69,6 +71,7 @@ FIXTURE_TEARDOWN(trace) set_cap(_metadata, CAP_SYS_ADMIN); tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false); tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false); + tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false); tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, false); tracefs_enable_event(TRACEFS_ADD_RULE_NET_ENABLE, false); tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false); @@ -98,6 +101,8 @@ TEST_F(trace, no_trace_when_disabled) ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false)); + ASSERT_EQ(0, + tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_FS_ENABLE, false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NET_ENABLE, false)); ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CHECK_RULE_FS_ENABLE, false)); @@ -1069,6 +1074,520 @@ TEST_F(trace, free_ruleset_on_close) free(buf); } +/* + * Counts landlock_enforce_domain lines, filtered by @domain (NULL matches any), + * @complete and @process_wide (a negative value matches any). Builds the + * anchored regex dynamically so a single helper covers every field assertion. + */ +static int count_enforce_matches(const char *buf, const char *domain, + int complete, int process_wide, + int no_new_privs) +{ + char pattern[512], dom[80], comp[8], pw[8], nnp[8]; + + if (domain) + snprintf(dom, sizeof(dom), "%s", domain); + else + snprintf(dom, sizeof(dom), "[0-9a-f]\\+"); + if (complete < 0) + snprintf(comp, sizeof(comp), "[01]"); + else + snprintf(comp, sizeof(comp), "%d", complete); + if (process_wide < 0) + snprintf(pw, sizeof(pw), "[01]"); + else + snprintf(pw, sizeof(pw), "%d", process_wide); + if (no_new_privs < 0) + snprintf(nnp, sizeof(nnp), "[01]"); + else + snprintf(nnp, sizeof(nnp), "%d", no_new_privs); + + snprintf(pattern, sizeof(pattern), + TRACE_PREFIX(TRACE_TASK) "landlock_enforce_domain: " + "domain=%s " + "complete=%s process_wide=%s " + "no_new_privs=%s$", + dom, comp, pw, nnp); + return tracefs_count_matches(buf, pattern); +} + +/* Idle sibling: waits on the barrier so it is a live thread, then sleeps. */ +static void *enforce_idle(void *arg) +{ + pthread_barrier_t *barrier = arg; + + pthread_barrier_wait(barrier); + while (true) + sleep(1); + return NULL; +} + +/* + * Child body: spawns @nthreads idle siblings (barrier-synchronized so they are + * live when the syscall runs), then enforces a domain with @flags. Returns 0 + * on success; the process exits afterwards, reaping the siblings. + */ +static int child_enforce(int nthreads, __u32 flags) +{ + pthread_t threads[8]; + pthread_barrier_t barrier; + int ruleset_fd, i; + + if (nthreads > 0) { + if (pthread_barrier_init(&barrier, NULL, nthreads + 1)) + return 1; + for (i = 0; i < nthreads; i++) + if (pthread_create(&threads[i], NULL, enforce_idle, + &barrier)) + return 1; + pthread_barrier_wait(&barrier); + } + + ruleset_fd = build_enforce_ruleset(); + if (ruleset_fd < 0) + return 1; + + /* + * LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS sets no_new_privs itself, so skip + * the prctl() to exercise that path; otherwise Landlock requires + * no_new_privs up front. + */ + if (!(flags & LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS)) + prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0); + if (landlock_restrict_self(ruleset_fd, flags)) + return 1; + close(ruleset_fd); + return 0; +} + +/* + * Runs in a spawned thread after the group leader called pthread_exit(). The + * leader lingers as an un-reaped zombie, so get_nr_threads() still counts it + * and this non-leader is not the only thread; enforcing here therefore reports + * process_wide=0. + */ +static void *enforce_nonleader(void *arg) +{ + int ruleset_fd; + + ruleset_fd = build_enforce_ruleset(); + if (ruleset_fd < 0) + _exit(1); + prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0); + if (landlock_restrict_self(ruleset_fd, 0)) + _exit(1); + _exit(0); +} + +/* + * Collapses the enforce_domain field cases into one parametrized test. Each + * variant runs child_enforce(nthreads, flags) and checks the resulting + * enforce_domain events. The flags column also selects how no_new_privs is + * set: with LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS child_enforce() skips the + * prctl() so the flag sets it (and, with TSYNC, propagates to the siblings); + * otherwise a prior prctl() sets it on the caller (and TSYNC propagates that). + */ + +/* clang-format off */ +FIXTURE(trace_enforce) { + /* clang-format on */ + int tracefs_ok; +}; + +FIXTURE_SETUP(trace_enforce) +{ + int ret; + + set_cap(_metadata, CAP_SYS_ADMIN); + ASSERT_EQ(0, unshare(CLONE_NEWNS)); + ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL)); + + ret = tracefs_fixture_setup(); + if (ret) { + clear_cap(_metadata, CAP_SYS_ADMIN); + self->tracefs_ok = 0; + SKIP(return, "tracefs not available"); + } + self->tracefs_ok = 1; + + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, true)); + ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, true)); + ASSERT_EQ(0, tracefs_clear()); + clear_cap(_metadata, CAP_SYS_ADMIN); +} + +FIXTURE_TEARDOWN(trace_enforce) +{ + if (!self->tracefs_ok) + return; + + set_cap(_metadata, CAP_SYS_ADMIN); + tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false); + tracefs_enable_event(TRACEFS_CREATE_DOMAIN_ENABLE, false); + tracefs_enable_event(TRACEFS_ENFORCE_DOMAIN_ENABLE, false); + tracefs_fixture_teardown(); + clear_cap(_metadata, CAP_SYS_ADMIN); +} + +/* clang-format off */ +FIXTURE_VARIANT(trace_enforce) { + /* clang-format on */ + /* Inputs to child_enforce(). */ + int nthreads; + __u32 flags; + /* Expected enforce_domain event counts. */ + int total; + int complete; + int process_wide; + int no_new_privs; +}; + +/* clang-format off */ + +/* Single thread, no flags: prctl-backed no_new_privs. */ +FIXTURE_VARIANT_ADD(trace_enforce, single) { + .nthreads = 0, .flags = 0, + .total = 1, .complete = 1, .process_wide = 1, .no_new_privs = 1, +}; + +/* Single thread: the NO_NEW_PRIVS flag sets no_new_privs (no prctl). */ +FIXTURE_VARIANT_ADD(trace_enforce, no_new_privs) { + .nthreads = 0, .flags = LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS, + .total = 1, .complete = 1, .process_wide = 1, .no_new_privs = 1, +}; + +/* TSYNC on a lone thread still concludes, process-wide. */ +FIXTURE_VARIANT_ADD(trace_enforce, tsync_single) { + .nthreads = 0, .flags = LANDLOCK_RESTRICT_SELF_TSYNC, + .total = 1, .complete = 1, .process_wide = 1, .no_new_privs = 1, +}; + +/* TSYNC sweeps N siblings; the caller's prctl-backed nnp propagates to all. */ +FIXTURE_VARIANT_ADD(trace_enforce, tsync_multithread) { + .nthreads = 3, .flags = LANDLOCK_RESTRICT_SELF_TSYNC, + .total = 4, .complete = 1, .process_wide = 4, .no_new_privs = 4, +}; + +/* TSYNC + NO_NEW_PRIVS flag sets nnp on the caller and every swept sibling. */ +FIXTURE_VARIANT_ADD(trace_enforce, tsync_no_new_privs) { + .nthreads = 3, + .flags = LANDLOCK_RESTRICT_SELF_TSYNC | LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS, + .total = 4, .complete = 1, .process_wide = 4, .no_new_privs = 4, +}; + +/* Non-TSYNC on a multi-threaded process enforces only the caller. */ +FIXTURE_VARIANT_ADD(trace_enforce, multithread_non_tsync) { + .nthreads = 3, .flags = 0, + .total = 1, .complete = 1, .process_wide = 0, .no_new_privs = 1, +}; + +/* clang-format on */ + +/* + * One create_domain and variant->total enforce_domain events sharing that + * domain ID; complete=1 marks the single concluding event, and the process_wide + * / no_new_privs counts match the variant. Counts are order-independent, + * evaluated after the syscall returns. + */ +TEST_F(trace_enforce, enforce) +{ + pid_t pid; + int status; + char *buf; + char domain[64]; + + ASSERT_EQ(0, tracefs_clear_buf()); + + pid = fork(); + ASSERT_LE(0, pid); + if (pid == 0) + _exit(child_enforce(variant->nthreads, variant->flags)); + + ASSERT_EQ(pid, waitpid(pid, &status, 0)); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + buf = tracefs_read_buf(); + ASSERT_NE(NULL, buf); + + EXPECT_EQ(1, + tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK))); + EXPECT_EQ(variant->total, count_enforce_matches(buf, NULL, -1, -1, -1)) + { + TH_LOG("Expected %d enforce_domain events\n%s", variant->total, + buf); + } + EXPECT_EQ(variant->complete, + count_enforce_matches(buf, NULL, 1, -1, -1)); + EXPECT_EQ(variant->total - variant->complete, + count_enforce_matches(buf, NULL, 0, -1, -1)); + EXPECT_EQ(variant->process_wide, + count_enforce_matches(buf, NULL, -1, 1, -1)); + EXPECT_EQ(variant->total - variant->process_wide, + count_enforce_matches(buf, NULL, -1, 0, -1)); + EXPECT_EQ(variant->no_new_privs, + count_enforce_matches(buf, NULL, -1, -1, 1)); + + ASSERT_EQ(0, tracefs_extract_field(buf, REGEX_CREATE_DOMAIN(TRACE_TASK), + "domain", domain, sizeof(domain))); + EXPECT_EQ(variant->total, + count_enforce_matches(buf, domain, -1, -1, -1)); + + free(buf); +} + +/* + * A non-leader thread enforcing a domain while the group leader lingers as an + * un-reaped zombie reports process_wide=0: get_nr_threads() counts the zombie + * leader, so the group is not single-threaded. This is the reachable half of + * the caveat that process_wide==0 never proves the process is multi-threaded + * (get_nr_threads(), unlike the leader-relative thread_group_empty(), counts + * the zombie leader). + */ +TEST_F(trace, enforce_single_non_leader) +{ + pid_t pid; + int status; + char *buf; + + ASSERT_EQ(0, tracefs_clear_buf()); + + pid = fork(); + ASSERT_LE(0, pid); + if (pid == 0) { + pthread_t worker; + + if (pthread_create(&worker, NULL, enforce_nonleader, NULL)) + _exit(1); + /* Leader leaves; the worker enforces as a non-leader. */ + pthread_exit(NULL); + } + + ASSERT_EQ(pid, waitpid(pid, &status, 0)); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + buf = tracefs_read_buf(); + ASSERT_NE(NULL, buf); + + EXPECT_EQ(1, + tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK))); + EXPECT_EQ(1, count_enforce_matches(buf, NULL, 1, 0, -1)) + { + TH_LOG("Expected complete=1 process_wide=0 for non-leader\n%s", + buf); + } + + free(buf); +} + +/* + * Verifies the flags-only path (ruleset_fd == -1) creates no domain and emits + * neither create_domain nor enforce_domain, with and without TSYNC. + */ +TEST_F(trace, enforce_flags_only) +{ + pid_t pid; + int status; + char *buf; + + ASSERT_EQ(0, tracefs_clear_buf()); + + pid = fork(); + ASSERT_LE(0, pid); + if (pid == 0) { + prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0); + if (landlock_restrict_self( + -1, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF)) + _exit(1); + if (landlock_restrict_self( + -1, LANDLOCK_RESTRICT_SELF_LOG_SUBDOMAINS_OFF | + LANDLOCK_RESTRICT_SELF_TSYNC)) + _exit(1); + _exit(0); + } + + ASSERT_EQ(pid, waitpid(pid, &status, 0)); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + buf = tracefs_read_buf(); + ASSERT_NE(NULL, buf); + + EXPECT_EQ(0, + tracefs_count_matches(buf, REGEX_CREATE_DOMAIN(TRACE_TASK))); + EXPECT_EQ(0, count_enforce_matches(buf, NULL, -1, -1, -1)) + { + TH_LOG("No enforce_domain expected on flags-only path\n%s", + buf); + } + + free(buf); +} + +static void enforce_nop_handler(int sig) +{ +} + +struct abort_signaler_data { + pthread_t target; + volatile bool stop; +}; + +/* + * Hammers the target thread with SIGUSR1 to interrupt the TSYNC prepare wait. + */ +static void *abort_signaler(void *arg) +{ + struct abort_signaler_data *data = arg; + + while (!data->stop) + pthread_kill(data->target, SIGUSR1); + return NULL; +} + +/* + * Child body for the abort test: with idle siblings and a signaler interrupting + * it, repeatedly enforces under TSYNC. An interrupted attempt aborts its + * just-created domain (create_domain + free_domain, zero enforce_domain) while + * -ERESTARTNOINTR transparently restarts the syscall, so a successful retry may + * add its own full lifecycle. + */ +static int child_abort(int nsiblings, int attempts) +{ + pthread_t threads[200]; + pthread_t signaler; + pthread_barrier_t barrier; + struct abort_signaler_data data = {}; + struct sigaction sa = {}; + int i; + + sa.sa_handler = enforce_nop_handler; + if (sigaction(SIGUSR1, &sa, NULL)) + return 1; + + if (pthread_barrier_init(&barrier, NULL, nsiblings + 1)) + return 1; + for (i = 0; i < nsiblings; i++) + if (pthread_create(&threads[i], NULL, enforce_idle, &barrier)) + return 1; + pthread_barrier_wait(&barrier); + + data.target = pthread_self(); + if (pthread_create(&signaler, NULL, abort_signaler, &data)) + return 1; + + prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0); + for (i = 0; i < attempts; i++) { + int ruleset_fd = build_enforce_ruleset(); + + if (ruleset_fd < 0) + break; + /* + * Ignore the result: an abort returns an error, that is fine. + */ + landlock_restrict_self(ruleset_fd, + LANDLOCK_RESTRICT_SELF_TSYNC); + close(ruleset_fd); + } + + data.stop = true; + pthread_join(signaler, NULL); + return 0; +} + +/* + * Verifies the abort contract: a domain aborted by a thread-sync failure emits + * create_domain and free_domain but zero enforce_domain. The signal race is + * probabilistic and -ERESTARTNOINTR may add a successful retry's lifecycle, so + * events are grouped by domain ID and the test SKIPs if no abort occurred. + */ +TEST_F(trace, enforce_abort) +{ + pid_t pid; + int status, retry; + char *buf = NULL; + const char *cursor; + char domain[64]; + bool abort_found = false; + + ASSERT_EQ(0, tracefs_clear_buf()); + + /* free_domain fires from a kworker, so widen the filter first. */ + set_cap(_metadata, CAP_SYS_ADMIN); + tracefs_clear_pid_filter(); + clear_cap(_metadata, CAP_SYS_ADMIN); + + pid = fork(); + ASSERT_LE(0, pid); + if (pid == 0) + /* + * Match tsync_test's NUM_IDLE_THREADS: enough siblings that + * credential preparation runs in several serialized waves, + * giving the signaler a window to interrupt the thread-sync + * wait and abort the operation. A handful of threads finishes + * in a single wave, leaving no window (the abort never fires). + */ + _exit(child_abort(200, 8)); + + ASSERT_EQ(pid, waitpid(pid, &status, 0)); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* Poll for the asynchronous free_domain events. */ + for (retry = 0; retry < 10; retry++) { + usleep(100000); + set_cap(_metadata, CAP_SYS_ADMIN); + free(buf); + buf = tracefs_read_trace(); + clear_cap(_metadata, CAP_SYS_ADMIN); + ASSERT_NE(NULL, buf); + } + + set_cap(_metadata, CAP_SYS_ADMIN); + ASSERT_EQ(0, tracefs_set_pid_filter(getpid())); + clear_cap(_metadata, CAP_SYS_ADMIN); + + /* + * Walk every create_domain and look for one whose domain ID has zero + * enforce_domain events but a matching free_domain: that is an aborted + * domain (created, never enforced, freed). + */ + cursor = buf; + while (tracefs_extract_field(cursor, REGEX_CREATE_DOMAIN(TRACE_TASK), + "domain", domain, sizeof(domain)) == 0) { + const char *cd, *nl; + char free_pattern[256]; + + if (count_enforce_matches(buf, domain, -1, -1, -1) == 0) { + snprintf( + free_pattern, sizeof(free_pattern), + TRACE_PREFIX( + KWORKER_TASK) "landlock_free_domain: " + "domain=%s denials=[0-9]\\+$", + domain); + if (tracefs_count_matches(buf, free_pattern) >= 1) + abort_found = true; + } + + cd = strstr(cursor, "landlock_create_domain:"); + if (!cd) + break; + nl = strchr(cd, '\n'); + if (!nl) + break; + cursor = nl + 1; + } + + if (!abort_found) { + free(buf); + SKIP(return, "signal race did not produce a thread-sync abort"); + } + + free(buf); +} + /* * The following tests are intentionally elided because the underlying kernel * mechanisms are already validated by audit tests: -- 2.54.0
