sched/fair: Don't free p->numa_faults with concurrent readers

author Jann Horn <jannh@google.com>

Tue, 16 Jul 2019 15:20:45 +0000 (17:20 +0200)

committer Kleber Sacilotto de Souza <kleber.souza@canonical.com>

Wed, 14 Aug 2019 09:18:49 +0000 (11:18 +0200)
author Jann Horn <jannh@google.com>
Tue, 16 Jul 2019 15:20:45 +0000 (17:20 +0200)
committer Kleber Sacilotto de Souza <kleber.souza@canonical.com>
Wed, 14 Aug 2019 09:18:49 +0000 (11:18 +0200)
diff --git a/fs/exec.c b/fs/exec.c

index 0705accfffc2861d0cb31e33c5e5edb3c2df6a7b..916cd805d32d4bed3b4a005525534411fe486f07 100644 (file)
--- a/fs/exec.c
+++ b/fs/exec.c
@@ -1818,7 +1818,7 @@ static int do_execveat_common(int fd, struct filename *filename,
         current->in_execve = 0;
         membarrier_execve(current);
         acct_update_integrals(current);
-       task_numa_free(current);
+       task_numa_free(current, false);
         free_bprm(bprm);
         kfree(pathbuf);
         putname(filename);
diff --git a/include/linux/sched/numa_balancing.h b/include/linux/sched/numa_balancing.h

index e7dd04a84ba89d79507d461cee5b839269ff0b8a..3988762efe15c0e5a80602e2c9acb6a5820a740e 100644 (file)
--- a/include/linux/sched/numa_balancing.h
+++ b/include/linux/sched/numa_balancing.h
@@ -19,7 +19,7 @@
  extern void task_numa_fault(int last_node, int node, int pages, int flags);
  extern pid_t task_numa_group_id(struct task_struct *p);
  extern void set_numabalancing_state(bool enabled);
-extern void task_numa_free(struct task_struct *p);
+extern void task_numa_free(struct task_struct *p, bool final);
  extern bool should_numa_migrate_memory(struct task_struct *p, struct page *page,
                                         int src_nid, int dst_cpu);
  #else
@@ -34,7 +34,7 @@ static inline pid_t task_numa_group_id(struct task_struct *p)
  static inline void set_numabalancing_state(bool enabled)
  {
  }
-static inline void task_numa_free(struct task_struct *p)
+static inline void task_numa_free(struct task_struct *p, bool final)
  {
  }
  static inline bool should_numa_migrate_memory(struct task_struct *p,
diff --git a/kernel/fork.c b/kernel/fork.c

index 620c7e7a40e2998e70fb2710f0867da317c89894..2e0b4559acfaef44436a436a1df9151eb4066bb8 100644 (file)
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -447,7 +447,7 @@ void __put_task_struct(struct task_struct *tsk)
         WARN_ON(tsk == current);
  
         cgroup_free(tsk);
-       task_numa_free(tsk);
+       task_numa_free(tsk, true);
         security_task_free(tsk);
         exit_creds(tsk);
         delayacct_tsk_free(tsk);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c

index ed7839b4c20ca20147d40ed293bf17f88e478edf..5d84f601ac45c3dd347df1b146159f1b0b512bff 100644 (file)
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -2350,13 +2350,23 @@ no_join:
         return;
  }
  
-void task_numa_free(struct task_struct *p)
+/*
+ * Get rid of NUMA staticstics associated with a task (either current or dead).
+ * If @final is set, the task is dead and has reached refcount zero, so we can
+ * safely free all relevant data structures. Otherwise, there might be
+ * concurrent reads from places like load balancing and procfs, and we should
+ * reset the data back to default state without freeing ->numa_faults.
+ */
+void task_numa_free(struct task_struct *p, bool final)
  {
         struct numa_group *grp = p->numa_group;
-       void *numa_faults = p->numa_faults;
+       unsigned long *numa_faults = p->numa_faults;
         unsigned long flags;
         int i;
  
+       if (!numa_faults)
+               return;
+
         if (grp) {
                 spin_lock_irqsave(&grp->lock, flags);
                 for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
@@ -2369,8 +2379,14 @@ void task_numa_free(struct task_struct *p)
                 put_numa_group(grp);
         }
  
-       p->numa_faults = NULL;
-       kfree(numa_faults);
+       if (final) {
+               p->numa_faults = NULL;
+               kfree(numa_faults);
+       } else {
+               p->total_numa_faults = 0;
+               for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
+                       numa_faults[i] = 0;
+       }
  }
  
  /*
author	Jann Horn <jannh@google.com>
	Tue, 16 Jul 2019 15:20:45 +0000 (17:20 +0200)
committer	Kleber Sacilotto de Souza <kleber.souza@canonical.com>
	Wed, 14 Aug 2019 09:18:49 +0000 (11:18 +0200)
fs/exec.c		patch \| blob \| blame \| history
include/linux/sched/numa_balancing.h		patch \| blob \| blame \| history
kernel/fork.c		patch \| blob \| blame \| history
kernel/sched/fair.c		patch \| blob \| blame \| history