hugetlb: prevent BUG_ON in hugetlb_fault() -> hugetlb_cow()

[android-x86/kernel.git] / mm / memory.c
diff --git a/mm/memory.c b/mm/memory.c

index 40b7531..d49b58a 100644 (file)
--- a/mm/memory.c
+++ b/mm/memory.c
@@ -305,6 +305,7 @@ int __tlb_remove_page(struct mmu_gather *tlb, struct page *page)
         if (batch->nr == batch->max) {
                 if (!tlb_next_batch(tlb))
                         return 0;
+               batch = tlb->active;
         }
         VM_BUG_ON(batch->nr > batch->max);
  
@@ -1227,16 +1228,24 @@ static inline unsigned long zap_pmd_range(struct mmu_gather *tlb,
         do {
                 next = pmd_addr_end(addr, end);
                 if (pmd_trans_huge(*pmd)) {
-                       if (next-addr != HPAGE_PMD_SIZE) {
+                       if (next - addr != HPAGE_PMD_SIZE) {
                                 VM_BUG_ON(!rwsem_is_locked(&tlb->mm->mmap_sem));
                                 split_huge_page_pmd(vma->vm_mm, pmd);
                         } else if (zap_huge_pmd(tlb, vma, pmd))
-                               continue;
+                               goto next;
                         /* fall through */
                 }
-               if (pmd_none_or_clear_bad(pmd))
-                       continue;
+               /*
+                * Here there can be other concurrent MADV_DONTNEED or
+                * trans huge page faults running, and if the pmd is
+                * none or trans huge it can change under us. This is
+                * because MADV_DONTNEED holds the mmap_sem in read
+                * mode.
+                */
+               if (pmd_none_or_trans_huge_or_clear_bad(pmd))
+                       goto next;
                 next = zap_pte_range(tlb, vma, pmd, addr, next, details);
+next:
                 cond_resched();
         } while (pmd++, addr = next, addr != end);
  
@@ -1513,7 +1522,7 @@ split_fallthrough:
         }
  
         if (flags & FOLL_GET)
-               get_page(page);
+               get_page_foll(page);
         if (flags & FOLL_TOUCH) {
                 if ((flags & FOLL_WRITE) &&
                     !pte_dirty(pte) && !PageDirty(page))
@@ -1815,7 +1824,63 @@ next_page:
  }
  EXPORT_SYMBOL(__get_user_pages);
  
-/**
+/*
+ * fixup_user_fault() - manually resolve a user page fault
+ * @tsk:       the task_struct to use for page fault accounting, or
+ *             NULL if faults are not to be recorded.
+ * @mm:                mm_struct of target mm
+ * @address:   user address
+ * @fault_flags:flags to pass down to handle_mm_fault()
+ *
+ * This is meant to be called in the specific scenario where for locking reasons
+ * we try to access user memory in atomic context (within a pagefault_disable()
+ * section), this returns -EFAULT, and we want to resolve the user fault before
+ * trying again.
+ *
+ * Typically this is meant to be used by the futex code.
+ *
+ * The main difference with get_user_pages() is that this function will
+ * unconditionally call handle_mm_fault() which will in turn perform all the
+ * necessary SW fixup of the dirty and young bits in the PTE, while
+ * handle_mm_fault() only guarantees to update these in the struct page.
+ *
+ * This is important for some architectures where those bits also gate the
+ * access permission to the page because they are maintained in software.  On
+ * such architectures, gup() will not be enough to make a subsequent access
+ * succeed.
+ *
+ * This should be called with the mm_sem held for read.
+ */
+int fixup_user_fault(struct task_struct *tsk, struct mm_struct *mm,
+                    unsigned long address, unsigned int fault_flags)
+{
+       struct vm_area_struct *vma;
+       int ret;
+
+       vma = find_extend_vma(mm, address);
+       if (!vma || address < vma->vm_start)
+               return -EFAULT;
+
+       ret = handle_mm_fault(mm, vma, address, fault_flags);
+       if (ret & VM_FAULT_ERROR) {
+               if (ret & VM_FAULT_OOM)
+                       return -ENOMEM;
+               if (ret & (VM_FAULT_HWPOISON | VM_FAULT_HWPOISON_LARGE))
+                       return -EHWPOISON;
+               if (ret & VM_FAULT_SIGBUS)
+                       return -EFAULT;
+               BUG();
+       }
+       if (tsk) {
+               if (ret & VM_FAULT_MAJOR)
+                       tsk->maj_flt++;
+               else
+                       tsk->min_flt++;
+       }
+       return 0;
+}
+
+/*
   * get_user_pages() - pin user pages in memory
   * @tsk:       the task_struct to use for page fault accounting, or
   *             NULL if faults are not to be recorded.