When CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE is enabled, architectures can
allocate a VMAP stack that is smaller than the full THREAD_SIZE.

This introduces arch abstraction arch_vm_stack_pages() which defaults
to THREAD_SIZE / PAGE_SIZE.
During stack allocation, the kernel will map only the requested number
of pages at the top of the THREAD_SIZE virtual area (as the stack
grows down), leaving the bottom unmapped. This saves memory while
retaining the same virtual alignment and guard page overflow detection
characteristics.

Signed-off-by: Pasha Tatashin <[email protected]>
[Rebased, used vm_area->nr_pages directly in one instance]
[Depends on !PREEMPT_RT]
Signed-off-by: Linus Walleij <[email protected]>
[Fix races around accounting]
[Use GFP_ATOMIC when executing in the scheduler]
[Depend on INIT_STACK_ALL_* config]
[Fix bugs in some error paths and edge cases]
[Don't cache partially faulted stacks]
[Added out-var to tell if address is on target stack]
Signed-off-by: David Stevens <[email protected]>
[Remove dynamic stack related code, and update commit message]
[Fix memory leak in error path of alloc_vmap_stack()] and use
 VM_UNINITIALIZED before installing the pages]
Signed-off-by: Mostafa Saleh <[email protected]>
---
 include/linux/thread_info.h |  8 +++++
 kernel/fork.c               | 68 +++++++++++++++++++++++++++++++++++++
 2 files changed, 76 insertions(+)

diff --git a/include/linux/thread_info.h b/include/linux/thread_info.h
index 307b8390fc67..8f052ec82402 100644
--- a/include/linux/thread_info.h
+++ b/include/linux/thread_info.h
@@ -92,6 +92,14 @@ static inline long set_restart_fn(struct restart_block 
*restart,
 #define THREAD_ALIGN   THREAD_SIZE
 #endif
 
+/*
+ * Arch selecting CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE should override this
+ * with kernel stack size.
+ */
+#ifndef arch_vm_stack_pages
+#define arch_vm_stack_pages()  (THREAD_SIZE / PAGE_SIZE)
+#endif
+
 #define THREADINFO_GFP         (GFP_KERNEL_ACCOUNT | __GFP_ZERO | 
__GFP_SKIP_KASAN)
 
 /*
diff --git a/kernel/fork.c b/kernel/fork.c
index ca8e68882316..66fbad3d4cbd 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -119,6 +119,8 @@
 
 /* For dup_mmap(). */
 #include "../mm/internal.h"
+/* For clear_vm_uninitialized_flag(). */
+#include "../mm/vmalloc.h"
 
 #include <trace/events/sched.h>
 
@@ -272,6 +274,71 @@ static bool try_release_thread_stack_to_cache(struct 
vm_struct *vm_area)
        return false;
 }
 
+#ifdef CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE
+static struct vm_struct *alloc_vmap_stack(int node)
+{
+       gfp_t gfp = GFP_VMAP_STACK;
+       unsigned long addr, end;
+       struct vm_struct *vm_area;
+       int err, i;
+
+       vm_area = get_vm_area_node(THREAD_SIZE, THREAD_ALIGN,
+                                  VM_MAP | VM_UNINITIALIZED, node, gfp,
+                                  __builtin_return_address(0));
+       if (!vm_area)
+               return NULL;
+
+       vm_area->pages = kmalloc_node(sizeof(void *) * (THREAD_SIZE >> 
PAGE_SHIFT),
+                                     gfp, node);
+       if (!vm_area->pages)
+               goto cleanup_err;
+
+       for (i = 0; i < arch_vm_stack_pages(); i++) {
+               vm_area->pages[i] = alloc_pages_node(node, gfp, 0);
+               if (!vm_area->pages[i])
+                       goto cleanup_err;
+               vm_area->nr_pages++;
+               mod_lruvec_page_state(vm_area->pages[i], NR_VMALLOC, 1);
+       }
+
+       end = (unsigned long)kasan_reset_tag(vm_area->addr) + THREAD_SIZE;
+       addr = end - (arch_vm_stack_pages() * PAGE_SIZE);
+
+       err = vmap_pages_range(addr, end, PAGE_KERNEL, vm_area->pages, 
PAGE_SHIFT);
+       if (err)
+               goto cleanup_err;
+
+       clear_vm_uninitialized_flag(vm_area);
+       return vm_area;
+
+cleanup_err:
+       remove_vm_area(vm_area->addr);
+       if (vm_area->pages) {
+               for (i = 0; i < vm_area->nr_pages; i++) {
+                       mod_lruvec_page_state(vm_area->pages[i], NR_VMALLOC, 
-1);
+                       __free_page(vm_area->pages[i]);
+               }
+               kfree(vm_area->pages);
+       }
+       kfree(vm_area);
+       return NULL;
+}
+
+static void free_vmap_stack(struct vm_struct *vm_area)
+{
+       int i;
+
+       remove_vm_area(vm_area->addr);
+
+       for (i = 0; i < vm_area->nr_pages; i++) {
+               mod_lruvec_page_state(vm_area->pages[i], NR_VMALLOC, -1);
+               __free_page(vm_area->pages[i]);
+       }
+
+       kfree(vm_area->pages);
+       kfree(vm_area);
+}
+#else
 static inline struct vm_struct *alloc_vmap_stack(int node)
 {
        void *stack;
@@ -286,6 +353,7 @@ static inline void free_vmap_stack(struct vm_struct 
*vm_area)
 {
        vfree(vm_area->addr);
 }
+#endif /* CONFIG_ARCH_HAS_VARIABLE_STACK_SIZE */
 
 static void thread_stack_free_work(struct work_struct *work)
 {
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


Reply via email to