MREMAP_DONTUNMAP keeps the source VMA in place, but clears its mlock flags for the whole VMA while setting them on the destination VMA. Two cases leak mm->locked_vm as a result:
- an unfaulted mlock-on-fault VMA moved behind itself self-merges, so the single resulting VMA loses the flags without the accounting being dropped; - a partial mremap() moves only part of the range, leaving the pages which are not moved accounted as locked in a VMA whose flags were cleared. Add three cases to the MREMAP_DONTUNMAP selftest which mlock() the source VMA, with and without MLOCK_ONFAULT, perform the operation and check that VmLck comes back to zero once everything is unmapped. Each case runs in its own process, so it starts from a clean mm with VmLck at zero and a failure cannot propagate to the cases which follow. Verified on x86_64: the three cases fail on v7.3-rc5 and pass on mm-unstable with the fixes from the "mm/mremap: fix two issues with MREMAP_DONTUNMAP" series applied. Signed-off-by: Jose A. Perez de Azpillaga <[email protected]> --- tools/testing/selftests/mm/mremap_dontunmap.c | 273 +++++++++++++++++- 1 file changed, 272 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/mremap_dontunmap.c b/tools/testing/selftests/mm/mremap_dontunmap.c index 96ba537facf7..2ec58b6cdc9f 100644 --- a/tools/testing/selftests/mm/mremap_dontunmap.c +++ b/tools/testing/selftests/mm/mremap_dontunmap.c @@ -7,6 +7,8 @@ */ #define _GNU_SOURCE #include <sys/mman.h> +#include <sys/syscall.h> +#include <sys/wait.h> #include <linux/mman.h> #include <errno.h> #include <stdio.h> @@ -37,6 +39,61 @@ static void dump_maps(void) } \ } while (0) +/* + * Same as mlock2.h's, plus an ENOSYS fallback for libc headers without + * __NR_mlock2. It is not taken from the header because that also defines + * seek_to_smaps_entry(), which nothing here uses and which then warns + * (-Wunused-function). + */ +static int mlock2_(void *start, size_t len, int flags) +{ +#ifdef __NR_mlock2 + return syscall(__NR_mlock2, start, len, flags); +#else + errno = ENOSYS; + return -1; +#endif +} + +/* + * Locked memory size in kB, as reported by /proc/self/status, which is + * mm->locked_vm accounted in kB. Used to check that the mlock() accounting + * balances across a MREMAP_DONTUNMAP operation. + * + * Returns LOCKED_VM_UNKNOWN if it cannot be read: the callers run in a child + * whose exit status is the test result, so this must not exit the process or + * print anything the TAP output parser would act on. + */ +#define LOCKED_VM_UNKNOWN ((unsigned long)-1) + +static unsigned long get_proc_locked_vm_size(void) +{ + unsigned long lock_size; + char *line = NULL; + size_t size = 0; + FILE *f; + + f = fopen("/proc/self/status", "r"); + if (!f) { + fprintf(stderr, "cannot open /proc/self/status: %s\n", + strerror(errno)); + return LOCKED_VM_UNKNOWN; + } + + while (getline(&line, &size, f) != -1) { + if (sscanf(line, "VmLck:\t%8lu kB", &lock_size) == 1) { + free(line); + fclose(f); + return lock_size; + } + } + + free(line); + fclose(f); + fprintf(stderr, "cannot parse VmLck in /proc/self/status\n"); + return LOCKED_VM_UNKNOWN; +} + // Try a simple operation for to "test" for kernel support this prevents // reporting tests as failed when it's run on an older kernel. static int kernel_support_for_mremap_dontunmap() @@ -335,6 +392,216 @@ static void mremap_dontunmap_partial_mapping_overwrite(void) ksft_test_result_pass("%s\n", __func__); } +/* + * Child exit codes for the accounting cases: any other exit code, and any + * signal, is reported as an error rather than mistaken for a leak. + */ +#define CASE_LEAK 2 +#define CASE_SKIP 77 +#define CASE_SKIP_ENOSYS 78 +#define CASE_SETUP_ERROR 79 + +/* Report a setup failure from a case: the child's exit status carries it back. */ +static int case_failed(const char *where, const char *what) +{ + fprintf(stderr, "%s: %s: %s\n", where, what, strerror(errno)); + return CASE_SETUP_ERROR; +} + +/* Report a check which failed for a reason errno does not describe. */ +static int case_unexpected(const char *where, const char *what) +{ + fprintf(stderr, "%s: unexpected %s\n", where, what); + return CASE_SETUP_ERROR; +} + +/* + * Only EPERM/ENOMEM are expected with a small RLIMIT_MEMLOCK, and ENOSYS means + * the kernel has no mlock2(); anything else is a genuine setup failure. + */ +static int lock_failed(const char *where, const char *call) +{ + if (errno == EPERM || errno == ENOMEM) + return CASE_SKIP; + if (errno == ENOSYS) + return CASE_SKIP_ENOSYS; + + return case_failed(where, call); +} + +/* + * Run one accounting case in a child, so that it starts with a clean mm and an + * empty VmLck, and report its outcome. + */ +static void run_locked_case(const char *label, int (*fn)(void)) +{ + int status; + pid_t pid; + + /* do not let the child flush a copy of our TAP output */ + fflush(NULL); + + pid = fork(); + if (pid < 0) { + ksft_test_result_error("%s: fork: %s\n", label, + strerror(errno)); + return; + } + if (!pid) + _exit(fn()); + + if (waitpid(pid, &status, 0) == -1) { + ksft_test_result_error("%s: waitpid: %s\n", label, + strerror(errno)); + return; + } + + if (WIFSIGNALED(status)) { + ksft_test_result_error("%s: killed by signal %d\n", label, + WTERMSIG(status)); + return; + } + + if (!WIFEXITED(status)) { + ksft_test_result_error("%s: child did not exit\n", label); + return; + } + + switch (WEXITSTATUS(status)) { + case 0: + ksft_test_result_pass("%s: locked memory released\n", label); + break; + case CASE_LEAK: + ksft_test_result_fail("%s: locked memory leaked\n", label); + break; + case CASE_SKIP: + ksft_test_result_skip("%s: mlock not permitted\n", label); + break; + case CASE_SKIP_ENOSYS: + ksft_test_result_skip("%s: mlock2 not supported\n", label); + break; + default: + ksft_test_result_error("%s: child exited with %d (see stderr)\n", + label, WEXITSTATUS(status)); + break; + } +} + +/* + * An unfaulted mlock-on-fault VMA moved behind itself: the source and + * destination VMAs are adjacent and mergeable, and merging them clears the + * mlock flags of the single resulting VMA, leaking mm->locked_vm. + */ +static int case_mlock_onfault_self_merge(void) +{ + unsigned long locked; + void *source, *dest, *reserve; + + /* + * Two adjacent pages: the source VMA goes in the first, the + * destination in the second, so the two are mergeable. + */ + reserve = mmap(NULL, 2 * page_size, PROT_NONE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (reserve == MAP_FAILED) + return case_failed(__func__, "mmap reserve"); + if (munmap(reserve, 2 * page_size) == -1) + return case_failed(__func__, "munmap reserve"); + + source = mmap(reserve, page_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0); + if (source != reserve) + return case_unexpected(__func__, "source address"); + + /* Locked on fault, but deliberately left unfaulted. */ + if (mlock2_(source, page_size, MLOCK_ONFAULT)) + return lock_failed(__func__, "mlock2"); + + dest = mremap(source, page_size, page_size, + MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED, + source + page_size); + if (dest == MAP_FAILED) + return case_failed(__func__, "mremap"); + + if (munmap(dest, page_size) == -1) + return case_failed(__func__, "munmap destination"); + if (munmap(source, page_size) == -1) + return case_failed(__func__, "munmap source"); + + locked = get_proc_locked_vm_size(); + if (locked == LOCKED_VM_UNKNOWN) + return CASE_SETUP_ERROR; + + return locked ? CASE_LEAK : 0; +} + +/* + * A partial MREMAP_DONTUNMAP of a locked VMA: all but the last page is moved, + * leaving the source VMA mapped. Both the moved pages and the VMA left + * behind must give up their mlock accounting. + * + * The destination goes into a window with a guard page on either side, so that + * it is not adjacent to and cannot merge with the source VMA: this case must + * exercise the partial-copy accounting on its own. The window's hole is + * smaller than the source mapping, so the source cannot land in it. + */ +static int case_locked_partial(int onfault) +{ + unsigned long num_pages = 3; + unsigned long span = (num_pages - 1) * page_size; + unsigned long locked; + void *source, *guard, *dest, *moved; + + guard = mmap(NULL, span + 2 * page_size, PROT_NONE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (guard == MAP_FAILED) + return case_failed(__func__, "mmap guard"); + dest = guard + page_size; + if (munmap(dest, span) == -1) + return case_failed(__func__, "munmap destination window"); + + source = mmap(NULL, num_pages * page_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (source == MAP_FAILED) + return case_failed(__func__, "mmap"); + + if (onfault) { + if (mlock2_(source, num_pages * page_size, MLOCK_ONFAULT)) + return lock_failed(__func__, "mlock2"); + } else if (mlock(source, num_pages * page_size)) { + return lock_failed(__func__, "mlock"); + } + + /* Move all but the last page, leaving the source partially mapped. */ + moved = mremap(source, span, span, + MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED, dest); + if (moved == MAP_FAILED) + return case_failed(__func__, "mremap"); + if (moved != dest) + return case_unexpected(__func__, "destination address"); + + if (munmap(dest, span) == -1) + return case_failed(__func__, "munmap destination"); + if (munmap(source, num_pages * page_size) == -1) + return case_failed(__func__, "munmap source"); + + locked = get_proc_locked_vm_size(); + if (locked == LOCKED_VM_UNKNOWN) + return CASE_SETUP_ERROR; + + return locked ? CASE_LEAK : 0; +} + +static int case_locked_partial_mlock(void) +{ + return case_locked_partial(0); +} + +static int case_locked_partial_onfault(void) +{ + return case_locked_partial(1); +} + int main(void) { ksft_print_header(); @@ -348,7 +615,7 @@ int main(void) ksft_finished(); } - ksft_set_plan(5); + ksft_set_plan(8); // Keep a page sized buffer around for when we need it. page_buffer = @@ -361,6 +628,10 @@ int main(void) mremap_dontunmap_simple_fixed(); mremap_dontunmap_partial_mapping(); mremap_dontunmap_partial_mapping_overwrite(); + run_locked_case("mlock-onfault self-merge", + case_mlock_onfault_self_merge); + run_locked_case("mlock partial", case_locked_partial_mlock); + run_locked_case("mlock2 onfault partial", case_locked_partial_onfault); BUG_ON(munmap(page_buffer, page_size) == -1, "unable to unmap page buffer"); -- 2.55.0

