mm for fs: add truncate_pagecache_range()

[platform/adaptation/renesas_rcar/renesas_kernel.git] / mm / memcontrol.c
diff --git a/mm/memcontrol.c b/mm/memcontrol.c

index 8afed28..b2ee6df 100644 (file)
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -1306,8 +1306,13 @@ int mem_cgroup_swappiness(struct mem_cgroup *memcg)
   *                                              rcu_read_unlock()
   *         start move here.
   */
+
+/* for quick checking without looking up memcg */
+atomic_t memcg_moving __read_mostly;
+
  static void mem_cgroup_start_move(struct mem_cgroup *memcg)
  {
+       atomic_inc(&memcg_moving);
         atomic_inc(&memcg->moving_account);
         synchronize_rcu();
  }
@@ -1318,15 +1323,17 @@ static void mem_cgroup_end_move(struct mem_cgroup *memcg)
          * Now, mem_cgroup_clear_mc() may call this function with NULL.
          * We check NULL in callee rather than caller.
          */
-       if (memcg)
+       if (memcg) {
+               atomic_dec(&memcg_moving);
                 atomic_dec(&memcg->moving_account);
+       }
  }
  
  /*
   * 2 routines for checking "mem" is under move_account() or not.
   *
- * mem_cgroup_stealed() - checking a cgroup is mc.from or not. This is used
- *                       for avoiding race in accounting. If true,
+ * mem_cgroup_stolen() -  checking whether a cgroup is mc.from or not. This
+ *                       is used for avoiding races in accounting.  If true,
   *                       pc->mem_cgroup may be overwritten.
   *
   * mem_cgroup_under_move() - checking a cgroup is mc.from or mc.to or
@@ -1334,7 +1341,7 @@ static void mem_cgroup_end_move(struct mem_cgroup *memcg)
   *                       waiting at hith-memory prressure caused by "move".
   */
  
-static bool mem_cgroup_stealed(struct mem_cgroup *memcg)
+static bool mem_cgroup_stolen(struct mem_cgroup *memcg)
  {
         VM_BUG_ON(!rcu_read_lock_held());
         return atomic_read(&memcg->moving_account) > 0;
@@ -1382,7 +1389,7 @@ static bool mem_cgroup_wait_acct_move(struct mem_cgroup *memcg)
   * Take this lock when
   * - a code tries to modify page's memcg while it's USED.
   * - a code tries to modify page state accounting in a memcg.
- * see mem_cgroup_stealed(), too.
+ * see mem_cgroup_stolen(), too.
   */
  static void move_lock_mem_cgroup(struct mem_cgroup *memcg,
                                   unsigned long *flags)
@@ -1910,39 +1917,62 @@ bool mem_cgroup_handle_oom(struct mem_cgroup *memcg, gfp_t mask, int order)
   * If there is, we take a lock.
   */
  
+void __mem_cgroup_begin_update_page_stat(struct page *page,
+                               bool *locked, unsigned long *flags)
+{
+       struct mem_cgroup *memcg;
+       struct page_cgroup *pc;
+
+       pc = lookup_page_cgroup(page);
+again:
+       memcg = pc->mem_cgroup;
+       if (unlikely(!memcg || !PageCgroupUsed(pc)))
+               return;
+       /*
+        * If this memory cgroup is not under account moving, we don't
+        * need to take move_lock_page_cgroup(). Because we already hold
+        * rcu_read_lock(), any calls to move_account will be delayed until
+        * rcu_read_unlock() if mem_cgroup_stolen() == true.
+        */
+       if (!mem_cgroup_stolen(memcg))
+               return;
+
+       move_lock_mem_cgroup(memcg, flags);
+       if (memcg != pc->mem_cgroup || !PageCgroupUsed(pc)) {
+               move_unlock_mem_cgroup(memcg, flags);
+               goto again;
+       }
+       *locked = true;
+}
+
+void __mem_cgroup_end_update_page_stat(struct page *page, unsigned long *flags)
+{
+       struct page_cgroup *pc = lookup_page_cgroup(page);
+
+       /*
+        * It's guaranteed that pc->mem_cgroup never changes while
+        * lock is held because a routine modifies pc->mem_cgroup
+        * should take move_lock_page_cgroup().
+        */
+       move_unlock_mem_cgroup(pc->mem_cgroup, flags);
+}
+
  void mem_cgroup_update_page_stat(struct page *page,
                                  enum mem_cgroup_page_stat_item idx, int val)
  {
         struct mem_cgroup *memcg;
         struct page_cgroup *pc = lookup_page_cgroup(page);
-       bool need_unlock = false;
         unsigned long uninitialized_var(flags);
  
         if (mem_cgroup_disabled())
                 return;
-again:
-       rcu_read_lock();
+
         memcg = pc->mem_cgroup;
         if (unlikely(!memcg || !PageCgroupUsed(pc)))
-               goto out;
-       /* pc->mem_cgroup is unstable ? */
-       if (unlikely(mem_cgroup_stealed(memcg))) {
-               /* take a lock against to access pc->mem_cgroup */
-               move_lock_mem_cgroup(memcg, &flags);
-               if (memcg != pc->mem_cgroup || !PageCgroupUsed(pc)) {
-                       move_unlock_mem_cgroup(memcg, &flags);
-                       rcu_read_unlock();
-                       goto again;
-               }
-               need_unlock = true;
-       }
+               return;
  
         switch (idx) {
         case MEMCG_NR_FILE_MAPPED:
-               if (val > 0)
-                       SetPageCgroupFileMapped(pc);
-               else if (!page_mapped(page))
-                       ClearPageCgroupFileMapped(pc);
                 idx = MEM_CGROUP_STAT_FILE_MAPPED;
                 break;
         default:
@@ -1950,11 +1980,6 @@ again:
         }
  
         this_cpu_add(memcg->stat->count[idx], val);
-
-out:
-       if (unlikely(need_unlock))
-               move_unlock_mem_cgroup(memcg, &flags);
-       rcu_read_unlock();
  }
  
  /*
@@ -2595,7 +2620,7 @@ static int mem_cgroup_move_account(struct page *page,
  
         move_lock_mem_cgroup(from, &flags);
  
-       if (PageCgroupFileMapped(pc)) {
+       if (!anon && page_mapped(page)) {
                 /* Update mapped_file data for mem_cgroup */
                 preempt_disable();
                 __this_cpu_dec(from->stat->count[MEM_CGROUP_STAT_FILE_MAPPED]);
@@ -2960,6 +2985,11 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype)
  
         switch (ctype) {
         case MEM_CGROUP_CHARGE_TYPE_MAPPED:
+               /*
+                * Generally PageAnon tells if it's the anon statistics to be
+                * updated; but sometimes e.g. mem_cgroup_uncharge_page() is
+                * used before page reached the stage of being marked PageAnon.
+                */
                 anon = true;
                 /* fallthrough */
         case MEM_CGROUP_CHARGE_TYPE_DROP:
@@ -3872,7 +3902,6 @@ static u64 mem_cgroup_read(struct cgroup *cont, struct cftype *cft)
                 break;
         default:
                 BUG();
-               break;
         }
         return val;
  }
@@ -4437,12 +4466,6 @@ static void mem_cgroup_usage_unregister_event(struct cgroup *cgrp,
         else
                 BUG();
  
-       /*
-        * Something went wrong if we trying to unregister a threshold
-        * if we don't have thresholds
-        */
-       BUG_ON(!thresholds);
-
         if (!thresholds->primary)
                 goto unlock;
  
@@ -5087,7 +5110,7 @@ one_by_one:
  }
  
  /**
- * is_target_pte_for_mc - check a pte whether it is valid for move charge
+ * get_mctgt_type - get target type of moving charge
   * @vma: the vma the pte to be checked belongs
   * @addr: the address corresponding to the pte to be checked
   * @ptent: the pte to be checked
@@ -5110,7 +5133,7 @@ union mc_target {
  };
  
  enum mc_target_type {
-       MC_TARGET_NONE, /* not used */
+       MC_TARGET_NONE = 0,
         MC_TARGET_PAGE,
         MC_TARGET_SWAP,
  };
@@ -5191,12 +5214,12 @@ static struct page *mc_handle_file_pte(struct vm_area_struct *vma,
         return page;
  }
  
-static int is_target_pte_for_mc(struct vm_area_struct *vma,
+static enum mc_target_type get_mctgt_type(struct vm_area_struct *vma,
                 unsigned long addr, pte_t ptent, union mc_target *target)
  {
         struct page *page = NULL;
         struct page_cgroup *pc;
-       int ret = 0;
+       enum mc_target_type ret = MC_TARGET_NONE;
         swp_entry_t ent = { .val = 0 };
  
         if (pte_present(ptent))
@@ -5207,7 +5230,7 @@ static int is_target_pte_for_mc(struct vm_area_struct *vma,
                 page = mc_handle_file_pte(vma, addr, ptent, &ent);
  
         if (!page && !ent.val)
-               return 0;
+               return ret;
         if (page) {
                 pc = lookup_page_cgroup(page);
                 /*
@@ -5233,6 +5256,41 @@ static int is_target_pte_for_mc(struct vm_area_struct *vma,
         return ret;
  }
  
+#ifdef CONFIG_TRANSPARENT_HUGEPAGE
+/*
+ * We don't consider swapping or file mapped pages because THP does not
+ * support them for now.
+ * Caller should make sure that pmd_trans_huge(pmd) is true.
+ */
+static enum mc_target_type get_mctgt_type_thp(struct vm_area_struct *vma,
+               unsigned long addr, pmd_t pmd, union mc_target *target)
+{
+       struct page *page = NULL;
+       struct page_cgroup *pc;
+       enum mc_target_type ret = MC_TARGET_NONE;
+
+       page = pmd_page(pmd);
+       VM_BUG_ON(!page || !PageHead(page));
+       if (!move_anon())
+               return ret;
+       pc = lookup_page_cgroup(page);
+       if (PageCgroupUsed(pc) && pc->mem_cgroup == mc.from) {
+               ret = MC_TARGET_PAGE;
+               if (target) {
+                       get_page(page);
+                       target->page = page;
+               }
+       }
+       return ret;
+}
+#else
+static inline enum mc_target_type get_mctgt_type_thp(struct vm_area_struct *vma,
+               unsigned long addr, pmd_t pmd, union mc_target *target)
+{
+       return MC_TARGET_NONE;
+}
+#endif
+
  static int mem_cgroup_count_precharge_pte_range(pmd_t *pmd,
                                         unsigned long addr, unsigned long end,
                                         struct mm_walk *walk)
@@ -5241,13 +5299,16 @@ static int mem_cgroup_count_precharge_pte_range(pmd_t *pmd,
         pte_t *pte;
         spinlock_t *ptl;
  
-       split_huge_page_pmd(walk->mm, pmd);
-       if (pmd_trans_unstable(pmd))
+       if (pmd_trans_huge_lock(pmd, vma) == 1) {
+               if (get_mctgt_type_thp(vma, addr, *pmd, NULL) == MC_TARGET_PAGE)
+                       mc.precharge += HPAGE_PMD_NR;
+               spin_unlock(&vma->vm_mm->page_table_lock);
                 return 0;
+       }
  
         pte = pte_offset_map_lock(vma->vm_mm, pmd, addr, &ptl);
         for (; addr != end; pte++, addr += PAGE_SIZE)
-               if (is_target_pte_for_mc(vma, addr, *pte, NULL))
+               if (get_mctgt_type(vma, addr, *pte, NULL))
                         mc.precharge++; /* increment precharge temporarily */
         pte_unmap_unlock(pte - 1, ptl);
         cond_resched();
@@ -5402,25 +5463,55 @@ static int mem_cgroup_move_charge_pte_range(pmd_t *pmd,
         struct vm_area_struct *vma = walk->private;
         pte_t *pte;
         spinlock_t *ptl;
+       enum mc_target_type target_type;
+       union mc_target target;
+       struct page *page;
+       struct page_cgroup *pc;
  
-       split_huge_page_pmd(walk->mm, pmd);
-       if (pmd_trans_unstable(pmd))
+       /*
+        * We don't take compound_lock() here but no race with splitting thp
+        * happens because:
+        *  - if pmd_trans_huge_lock() returns 1, the relevant thp is not
+        *    under splitting, which means there's no concurrent thp split,
+        *  - if another thread runs into split_huge_page() just after we
+        *    entered this if-block, the thread must wait for page table lock
+        *    to be unlocked in __split_huge_page_splitting(), where the main
+        *    part of thp split is not executed yet.
+        */
+       if (pmd_trans_huge_lock(pmd, vma) == 1) {
+               if (!mc.precharge) {
+                       spin_unlock(&vma->vm_mm->page_table_lock);
+                       return 0;
+               }
+               target_type = get_mctgt_type_thp(vma, addr, *pmd, &target);
+               if (target_type == MC_TARGET_PAGE) {
+                       page = target.page;
+                       if (!isolate_lru_page(page)) {
+                               pc = lookup_page_cgroup(page);
+                               if (!mem_cgroup_move_account(page, HPAGE_PMD_NR,
+                                                            pc, mc.from, mc.to,
+                                                            false)) {
+                                       mc.precharge -= HPAGE_PMD_NR;
+                                       mc.moved_charge += HPAGE_PMD_NR;
+                               }
+                               putback_lru_page(page);
+                       }
+                       put_page(page);
+               }
+               spin_unlock(&vma->vm_mm->page_table_lock);
                 return 0;
+       }
+
  retry:
         pte = pte_offset_map_lock(vma->vm_mm, pmd, addr, &ptl);
         for (; addr != end; addr += PAGE_SIZE) {
                 pte_t ptent = *(pte++);
-               union mc_target target;
-               int type;
-               struct page *page;
-               struct page_cgroup *pc;
                 swp_entry_t ent;
  
                 if (!mc.precharge)
                         break;
  
-               type = is_target_pte_for_mc(vma, addr, ptent, &target);
-               switch (type) {
+               switch (get_mctgt_type(vma, addr, ptent, &target)) {
                 case MC_TARGET_PAGE:
                         page = target.page;
                         if (isolate_lru_page(page))
@@ -5433,7 +5524,7 @@ retry:
                                 mc.moved_charge++;
                         }
                         putback_lru_page(page);
-put:                   /* is_target_pte_for_mc() gets the page */
+put:                   /* get_mctgt_type() gets the page */
                         put_page(page);
                         break;
                 case MC_TARGET_SWAP: