sched/fair: Don't free p->numa_faults with concurrent readers

author Jann Horn <jannh@google.com>

Tue, 16 Jul 2019 15:20:45 +0000 (17:20 +0200)

committer Ingo Molnar <mingo@kernel.org>

Thu, 25 Jul 2019 13:37:04 +0000 (15:37 +0200)
author Jann Horn <jannh@google.com>
Tue, 16 Jul 2019 15:20:45 +0000 (17:20 +0200)
committer Ingo Molnar <mingo@kernel.org>
Thu, 25 Jul 2019 13:37:04 +0000 (15:37 +0200)
diff --git a/fs/exec.c b/fs/exec.c

index c71cbfe6826a5165577057000393be299613df5b..f7f6a140856a8315c9403eeaac0c2376071e95ec 100644 (file)
--- a/fs/exec.c
+++ b/fs/exec.c
@@ -1828,7 +1828,7 @@ static int __do_execve_file(int fd, struct filename *filename,
         membarrier_execve(current);
         rseq_execve(current);
         acct_update_integrals(current);
-       task_numa_free(current);
+       task_numa_free(current, false);
         free_bprm(bprm);
         kfree(pathbuf);
         if (filename)
diff --git a/include/linux/sched/numa_balancing.h b/include/linux/sched/numa_balancing.h

index e7dd04a84ba89d79507d461cee5b839269ff0b8a..3988762efe15c0e5a80602e2c9acb6a5820a740e 100644 (file)
--- a/include/linux/sched/numa_balancing.h
+++ b/include/linux/sched/numa_balancing.h
@@ -19,7 +19,7 @@
  extern void task_numa_fault(int last_node, int node, int pages, int flags);
  extern pid_t task_numa_group_id(struct task_struct *p);
  extern void set_numabalancing_state(bool enabled);
-extern void task_numa_free(struct task_struct *p);
+extern void task_numa_free(struct task_struct *p, bool final);
  extern bool should_numa_migrate_memory(struct task_struct *p, struct page *page,
                                         int src_nid, int dst_cpu);
  #else
@@ -34,7 +34,7 @@ static inline pid_t task_numa_group_id(struct task_struct *p)
  static inline void set_numabalancing_state(bool enabled)
  {
  }
-static inline void task_numa_free(struct task_struct *p)
+static inline void task_numa_free(struct task_struct *p, bool final)
  {
  }
  static inline bool should_numa_migrate_memory(struct task_struct *p,
diff --git a/kernel/fork.c b/kernel/fork.c

index d8ae0f1b4148023c8c0d242b799039deb362ca09..2852d0e76ea3b905693a454f36e21524c14a8b54 100644 (file)
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -726,7 +726,7 @@ void __put_task_struct(struct task_struct *tsk)
         WARN_ON(tsk == current);
  
         cgroup_free(tsk);
-       task_numa_free(tsk);
+       task_numa_free(tsk, true);
         security_task_free(tsk);
         exit_creds(tsk);
         delayacct_tsk_free(tsk);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c

index 036be95a87e905a726ed9fa61369a8054b20733c..6adb0e0f5feb8b159007201381a09318c073e645 100644 (file)
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -2353,13 +2353,23 @@ no_join:
         return;
  }
  
-void task_numa_free(struct task_struct *p)
+/*
+ * Get rid of NUMA staticstics associated with a task (either current or dead).
+ * If @final is set, the task is dead and has reached refcount zero, so we can
+ * safely free all relevant data structures. Otherwise, there might be
+ * concurrent reads from places like load balancing and procfs, and we should
+ * reset the data back to default state without freeing ->numa_faults.
+ */
+void task_numa_free(struct task_struct *p, bool final)
  {
         struct numa_group *grp = p->numa_group;
-       void *numa_faults = p->numa_faults;
+       unsigned long *numa_faults = p->numa_faults;
         unsigned long flags;
         int i;
  
+       if (!numa_faults)
+               return;
+
         if (grp) {
                 spin_lock_irqsave(&grp->lock, flags);
                 for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
@@ -2372,8 +2382,14 @@ void task_numa_free(struct task_struct *p)
                 put_numa_group(grp);
         }
  
-       p->numa_faults = NULL;
-       kfree(numa_faults);
+       if (final) {
+               p->numa_faults = NULL;
+               kfree(numa_faults);
+       } else {
+               p->total_numa_faults = 0;
+               for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
+                       numa_faults[i] = 0;
+       }
  }
  
  /*
author	Jann Horn <jannh@google.com>
	Tue, 16 Jul 2019 15:20:45 +0000 (17:20 +0200)
committer	Ingo Molnar <mingo@kernel.org>
	Thu, 25 Jul 2019 13:37:04 +0000 (15:37 +0200)
fs/exec.c		patch \| blob \| history
include/linux/sched/numa_balancing.h		patch \| blob \| history
kernel/fork.c		patch \| blob \| history
kernel/sched/fair.c		patch \| blob \| history