MREMAP_DONTUNMAP keeps the source VMA in place, but clears its mlock flags
for the whole VMA while setting them on the destination VMA.  Two cases
leak mm->locked_vm as a result:

 - an unfaulted mlock-on-fault VMA moved behind itself self-merges, so the
   single resulting VMA loses the flags without the accounting being
   dropped;

 - a partial mremap() moves only part of the range, leaving the pages which
   are not moved accounted as locked in a VMA whose flags were cleared.

Add three cases to the MREMAP_DONTUNMAP selftest which mlock() the source
VMA, with and without MLOCK_ONFAULT, perform the operation and check that
VmLck comes back to zero once everything is unmapped.  Each case runs in
its own process, so it starts from a clean mm with VmLck at zero and a
failure cannot propagate to the cases which follow.

Verified on x86_64: the three cases fail on v7.3-rc5 and pass on
mm-unstable with the fixes from the "mm/mremap: fix two issues with
MREMAP_DONTUNMAP" series applied.

Signed-off-by: Jose A. Perez de Azpillaga <[email protected]>
---
 tools/testing/selftests/mm/mremap_dontunmap.c | 273 +++++++++++++++++-
 1 file changed, 272 insertions(+), 1 deletion(-)

diff --git a/tools/testing/selftests/mm/mremap_dontunmap.c 
b/tools/testing/selftests/mm/mremap_dontunmap.c
index 96ba537facf7..2ec58b6cdc9f 100644
--- a/tools/testing/selftests/mm/mremap_dontunmap.c
+++ b/tools/testing/selftests/mm/mremap_dontunmap.c
@@ -7,6 +7,8 @@
  */
 #define _GNU_SOURCE
 #include <sys/mman.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
 #include <linux/mman.h>
 #include <errno.h>
 #include <stdio.h>
@@ -37,6 +39,61 @@ static void dump_maps(void)
                }                                                               
\
        } while (0)

+/*
+ * Same as mlock2.h's, plus an ENOSYS fallback for libc headers without
+ * __NR_mlock2.  It is not taken from the header because that also defines
+ * seek_to_smaps_entry(), which nothing here uses and which then warns
+ * (-Wunused-function).
+ */
+static int mlock2_(void *start, size_t len, int flags)
+{
+#ifdef __NR_mlock2
+       return syscall(__NR_mlock2, start, len, flags);
+#else
+       errno = ENOSYS;
+       return -1;
+#endif
+}
+
+/*
+ * Locked memory size in kB, as reported by /proc/self/status, which is
+ * mm->locked_vm accounted in kB.  Used to check that the mlock() accounting
+ * balances across a MREMAP_DONTUNMAP operation.
+ *
+ * Returns LOCKED_VM_UNKNOWN if it cannot be read: the callers run in a child
+ * whose exit status is the test result, so this must not exit the process or
+ * print anything the TAP output parser would act on.
+ */
+#define LOCKED_VM_UNKNOWN      ((unsigned long)-1)
+
+static unsigned long get_proc_locked_vm_size(void)
+{
+       unsigned long lock_size;
+       char *line = NULL;
+       size_t size = 0;
+       FILE *f;
+
+       f = fopen("/proc/self/status", "r");
+       if (!f) {
+               fprintf(stderr, "cannot open /proc/self/status: %s\n",
+                       strerror(errno));
+               return LOCKED_VM_UNKNOWN;
+       }
+
+       while (getline(&line, &size, f) != -1) {
+               if (sscanf(line, "VmLck:\t%8lu kB", &lock_size) == 1) {
+                       free(line);
+                       fclose(f);
+                       return lock_size;
+               }
+       }
+
+       free(line);
+       fclose(f);
+       fprintf(stderr, "cannot parse VmLck in /proc/self/status\n");
+       return LOCKED_VM_UNKNOWN;
+}
+
 // Try a simple operation for to "test" for kernel support this prevents
 // reporting tests as failed when it's run on an older kernel.
 static int kernel_support_for_mremap_dontunmap()
@@ -335,6 +392,216 @@ static void 
mremap_dontunmap_partial_mapping_overwrite(void)
        ksft_test_result_pass("%s\n", __func__);
 }

+/*
+ * Child exit codes for the accounting cases: any other exit code, and any
+ * signal, is reported as an error rather than mistaken for a leak.
+ */
+#define CASE_LEAK              2
+#define CASE_SKIP              77
+#define CASE_SKIP_ENOSYS       78
+#define CASE_SETUP_ERROR       79
+
+/* Report a setup failure from a case: the child's exit status carries it 
back. */
+static int case_failed(const char *where, const char *what)
+{
+       fprintf(stderr, "%s: %s: %s\n", where, what, strerror(errno));
+       return CASE_SETUP_ERROR;
+}
+
+/* Report a check which failed for a reason errno does not describe. */
+static int case_unexpected(const char *where, const char *what)
+{
+       fprintf(stderr, "%s: unexpected %s\n", where, what);
+       return CASE_SETUP_ERROR;
+}
+
+/*
+ * Only EPERM/ENOMEM are expected with a small RLIMIT_MEMLOCK, and ENOSYS means
+ * the kernel has no mlock2(); anything else is a genuine setup failure.
+ */
+static int lock_failed(const char *where, const char *call)
+{
+       if (errno == EPERM || errno == ENOMEM)
+               return CASE_SKIP;
+       if (errno == ENOSYS)
+               return CASE_SKIP_ENOSYS;
+
+       return case_failed(where, call);
+}
+
+/*
+ * Run one accounting case in a child, so that it starts with a clean mm and an
+ * empty VmLck, and report its outcome.
+ */
+static void run_locked_case(const char *label, int (*fn)(void))
+{
+       int status;
+       pid_t pid;
+
+       /* do not let the child flush a copy of our TAP output */
+       fflush(NULL);
+
+       pid = fork();
+       if (pid < 0) {
+               ksft_test_result_error("%s: fork: %s\n", label,
+                                      strerror(errno));
+               return;
+       }
+       if (!pid)
+               _exit(fn());
+
+       if (waitpid(pid, &status, 0) == -1) {
+               ksft_test_result_error("%s: waitpid: %s\n", label,
+                                      strerror(errno));
+               return;
+       }
+
+       if (WIFSIGNALED(status)) {
+               ksft_test_result_error("%s: killed by signal %d\n", label,
+                                      WTERMSIG(status));
+               return;
+       }
+
+       if (!WIFEXITED(status)) {
+               ksft_test_result_error("%s: child did not exit\n", label);
+               return;
+       }
+
+       switch (WEXITSTATUS(status)) {
+       case 0:
+               ksft_test_result_pass("%s: locked memory released\n", label);
+               break;
+       case CASE_LEAK:
+               ksft_test_result_fail("%s: locked memory leaked\n", label);
+               break;
+       case CASE_SKIP:
+               ksft_test_result_skip("%s: mlock not permitted\n", label);
+               break;
+       case CASE_SKIP_ENOSYS:
+               ksft_test_result_skip("%s: mlock2 not supported\n", label);
+               break;
+       default:
+               ksft_test_result_error("%s: child exited with %d (see 
stderr)\n",
+                                      label, WEXITSTATUS(status));
+               break;
+       }
+}
+
+/*
+ * An unfaulted mlock-on-fault VMA moved behind itself: the source and
+ * destination VMAs are adjacent and mergeable, and merging them clears the
+ * mlock flags of the single resulting VMA, leaking mm->locked_vm.
+ */
+static int case_mlock_onfault_self_merge(void)
+{
+       unsigned long locked;
+       void *source, *dest, *reserve;
+
+       /*
+        * Two adjacent pages: the source VMA goes in the first, the
+        * destination in the second, so the two are mergeable.
+        */
+       reserve = mmap(NULL, 2 * page_size, PROT_NONE,
+                      MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+       if (reserve == MAP_FAILED)
+               return case_failed(__func__, "mmap reserve");
+       if (munmap(reserve, 2 * page_size) == -1)
+               return case_failed(__func__, "munmap reserve");
+
+       source = mmap(reserve, page_size, PROT_READ | PROT_WRITE,
+                     MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+       if (source != reserve)
+               return case_unexpected(__func__, "source address");
+
+       /* Locked on fault, but deliberately left unfaulted. */
+       if (mlock2_(source, page_size, MLOCK_ONFAULT))
+               return lock_failed(__func__, "mlock2");
+
+       dest = mremap(source, page_size, page_size,
+                     MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED,
+                     source + page_size);
+       if (dest == MAP_FAILED)
+               return case_failed(__func__, "mremap");
+
+       if (munmap(dest, page_size) == -1)
+               return case_failed(__func__, "munmap destination");
+       if (munmap(source, page_size) == -1)
+               return case_failed(__func__, "munmap source");
+
+       locked = get_proc_locked_vm_size();
+       if (locked == LOCKED_VM_UNKNOWN)
+               return CASE_SETUP_ERROR;
+
+       return locked ? CASE_LEAK : 0;
+}
+
+/*
+ * A partial MREMAP_DONTUNMAP of a locked VMA: all but the last page is moved,
+ * leaving the source VMA mapped.  Both the moved pages and the VMA left
+ * behind must give up their mlock accounting.
+ *
+ * The destination goes into a window with a guard page on either side, so that
+ * it is not adjacent to and cannot merge with the source VMA: this case must
+ * exercise the partial-copy accounting on its own.  The window's hole is
+ * smaller than the source mapping, so the source cannot land in it.
+ */
+static int case_locked_partial(int onfault)
+{
+       unsigned long num_pages = 3;
+       unsigned long span = (num_pages - 1) * page_size;
+       unsigned long locked;
+       void *source, *guard, *dest, *moved;
+
+       guard = mmap(NULL, span + 2 * page_size, PROT_NONE,
+                    MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+       if (guard == MAP_FAILED)
+               return case_failed(__func__, "mmap guard");
+       dest = guard + page_size;
+       if (munmap(dest, span) == -1)
+               return case_failed(__func__, "munmap destination window");
+
+       source = mmap(NULL, num_pages * page_size, PROT_READ | PROT_WRITE,
+                     MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+       if (source == MAP_FAILED)
+               return case_failed(__func__, "mmap");
+
+       if (onfault) {
+               if (mlock2_(source, num_pages * page_size, MLOCK_ONFAULT))
+                       return lock_failed(__func__, "mlock2");
+       } else if (mlock(source, num_pages * page_size)) {
+               return lock_failed(__func__, "mlock");
+       }
+
+       /* Move all but the last page, leaving the source partially mapped. */
+       moved = mremap(source, span, span,
+                      MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED, dest);
+       if (moved == MAP_FAILED)
+               return case_failed(__func__, "mremap");
+       if (moved != dest)
+               return case_unexpected(__func__, "destination address");
+
+       if (munmap(dest, span) == -1)
+               return case_failed(__func__, "munmap destination");
+       if (munmap(source, num_pages * page_size) == -1)
+               return case_failed(__func__, "munmap source");
+
+       locked = get_proc_locked_vm_size();
+       if (locked == LOCKED_VM_UNKNOWN)
+               return CASE_SETUP_ERROR;
+
+       return locked ? CASE_LEAK : 0;
+}
+
+static int case_locked_partial_mlock(void)
+{
+       return case_locked_partial(0);
+}
+
+static int case_locked_partial_onfault(void)
+{
+       return case_locked_partial(1);
+}
+
 int main(void)
 {
        ksft_print_header();
@@ -348,7 +615,7 @@ int main(void)
                ksft_finished();
        }

-       ksft_set_plan(5);
+       ksft_set_plan(8);

        // Keep a page sized buffer around for when we need it.
        page_buffer =
@@ -361,6 +628,10 @@ int main(void)
        mremap_dontunmap_simple_fixed();
        mremap_dontunmap_partial_mapping();
        mremap_dontunmap_partial_mapping_overwrite();
+       run_locked_case("mlock-onfault self-merge",
+                       case_mlock_onfault_self_merge);
+       run_locked_case("mlock partial", case_locked_partial_mlock);
+       run_locked_case("mlock2 onfault partial", case_locked_partial_onfault);

        BUG_ON(munmap(page_buffer, page_size) == -1,
               "unable to unmap page buffer");
--
2.55.0


Reply via email to