From 4a527b37031ef509d62324138baef7e8a062601f Mon Sep 17 00:00:00 2001 From: Michael Bestas Date: Wed, 7 May 2025 08:33:28 +0300 Subject: [PATCH 01/48] power: supply: qti_battery_charger: Fix charging_enabled node disabled state The previous logic was flawed, since it was limiting charge to 1A when charging control was disabled. Default to thermal limit to mimic what restrict_chg node does. Change-Id: I18fb4f18ade276b561171f3217ddafa0e48a4555 --- drivers/power/supply/qti_battery_charger.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/power/supply/qti_battery_charger.c b/drivers/power/supply/qti_battery_charger.c index 3f2993698fcc..3ba574ab29d9 100644 --- a/drivers/power/supply/qti_battery_charger.c +++ b/drivers/power/supply/qti_battery_charger.c @@ -1801,13 +1801,13 @@ static ssize_t charging_enabled_store(struct class *c, if (val) { /* * Enable charging, i.e. set the restricted current back to - * its default value and unset the restriction boolean flag. + * the thermal limit and unset the restriction boolean flag. */ rc = __battery_psy_set_charge_current(bcdev, - DEFAULT_RESTRICT_FCC_UA); + bcdev->thermal_fcc_ua); if (rc < 0) return rc; - bcdev->restrict_fcc_ua = DEFAULT_RESTRICT_FCC_UA; + bcdev->restrict_fcc_ua = bcdev->thermal_fcc_ua; bcdev->restrict_chg_en = 0; } else { /* From 1828476938584031f59b52781d736406b688fc7a Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Sun, 14 Jan 2024 13:05:20 +0000 Subject: [PATCH 02/48] Revert "BACKPORT: FROMLIST: mm: protect free_pgtables with mmap_lock write lock in exit_mmap" This reverts commit bb3bc96f354eda355b1606b360e77580eabb3129. Change-Id: Id31ad7085d20706bbc6dc12a05b7c1e31b3edc2a Signed-off-by: Alexander Winkowski --- mm/mmap.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/mm/mmap.c b/mm/mmap.c index fcc6f69fd4be..aa072afea2ee 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -3259,9 +3259,10 @@ void exit_mmap(struct mm_struct *mm) (void)__oom_reap_task_mm(mm); set_bit(MMF_OOM_SKIP, &mm->flags); + down_write(&mm->mmap_sem); + up_write(&mm->mmap_sem); } - down_write(&mm->mmap_sem); if (mm->locked_vm) { vma = mm->mmap; while (vma) { @@ -3274,11 +3275,8 @@ void exit_mmap(struct mm_struct *mm) arch_exit_mmap(mm); vma = mm->mmap; - if (!vma) { - /* Can happen if dup_mmap() received an OOM */ - up_write(&mm->mmap_sem);; + if (!vma) /* Can happen if dup_mmap() received an OOM */ return; - } lru_add_drain(); flush_cache_mm(mm); @@ -3289,14 +3287,16 @@ void exit_mmap(struct mm_struct *mm) free_pgtables(&tlb, vma, FIRST_USER_ADDRESS, USER_PGTABLES_CEILING); tlb_finish_mmu(&tlb, 0, -1); - /* Walk the list again, actually closing and freeing it. */ + /* + * Walk the list again, actually closing and freeing it, + * with preemption enabled, without holding any MM locks. + */ while (vma) { if (vma->vm_flags & VM_ACCOUNT) nr_accounted += vma_pages(vma); vma = remove_vma(vma); cond_resched(); } - up_write(&mm->mmap_sem); vm_unacct_memory(nr_accounted); } From 2673354abea4cd8c742b65b9e44becd930c01fb4 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:50:54 +0000 Subject: [PATCH 03/48] Revert "ANDROID: mm/filemap: Fix missing put_page() for speculative page fault" This reverts commit 290d702383e50bf509c89a7aeff60f2525c6cc86. Change-Id: I6d083e1e4a70a352cdf8162e72c1f3bfb1cc0b64 Signed-off-by: Alexander Winkowski --- mm/filemap.c | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/mm/filemap.c b/mm/filemap.c index 67419f9a2a97..623550f9a9d3 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -2663,14 +2663,11 @@ vm_fault_t filemap_fault(struct vm_fault *vmf) if (vmf->flags & FAULT_FLAG_SPECULATIVE) { page = find_get_page(mapping, offset); - if (unlikely(!page)) + if (unlikely(!page) || unlikely(PageReadahead(page))) return VM_FAULT_RETRY; - if (unlikely(PageReadahead(page))) - goto page_put; - if (!trylock_page(page)) - goto page_put; + return VM_FAULT_RETRY; if (unlikely(compound_head(page)->mapping != mapping)) goto page_unlock; @@ -2702,8 +2699,6 @@ vm_fault_t filemap_fault(struct vm_fault *vmf) return VM_FAULT_LOCKED; page_unlock: unlock_page(page); -page_put: - put_page(page); return VM_FAULT_RETRY; } From fa80be5b4ec9a00cd81c143e45b81481d3ed8235 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:51:26 +0000 Subject: [PATCH 04/48] Revert "ANDROID: Re-enable fast mremap and fix UAF with SPF" This reverts commit ea5f9d7e7ebd9b098d253f1f3bd20d9f335d9ab5. Change-Id: Ib977f22950887f417660b60738f26289d9422c39 Signed-off-by: Alexander Winkowski --- mm/mmap.c | 18 ++---------------- mm/mremap.c | 43 +++---------------------------------------- 2 files changed, 5 insertions(+), 56 deletions(-) diff --git a/mm/mmap.c b/mm/mmap.c index aa072afea2ee..eedd39382561 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -2347,22 +2347,8 @@ struct vm_area_struct *get_vma(struct mm_struct *mm, unsigned long addr) read_lock(&mm->mm_rb_lock); vma = __find_vma(mm, addr); - - /* - * If there is a concurrent fast mremap, bail out since the entire - * PMD/PUD subtree may have been remapped. - * - * This is usually safe for conventional mremap since it takes the - * PTE locks as does SPF. However fast mremap only takes the lock - * at the PMD/PUD level which is ok as it is done with the mmap - * write lock held. But since SPF, as the term implies forgoes, - * taking the mmap read lock and also cannot take PTL lock at the - * larger PMD/PUD granualrity, since it would introduce huge - * contention in the page fault path; fall back to regular fault - * handling. - */ - if (vma && !atomic_inc_unless_negative(&vma->vm_ref_count)) - vma = NULL; + if (vma) + atomic_inc(&vma->vm_ref_count); read_unlock(&mm->mm_rb_lock); return vma; diff --git a/mm/mremap.c b/mm/mremap.c index 4588c5d90884..f0064331dbd3 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -210,39 +210,11 @@ static void move_ptes(struct vm_area_struct *vma, pmd_t *old_pmd, drop_rmap_locks(vma); } -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -static inline bool trylock_vma_ref_count(struct vm_area_struct *vma) -{ - /* - * If we have the only reference, swap the refcount to -1. This - * will prevent other concurrent references by get_vma() for SPFs. - */ - return atomic_cmpxchg(&vma->vm_ref_count, 1, -1) == 1; -} - /* - * Restore the VMA reference count to 1 after a fast mremap. + * Speculative page fault handlers will not detect page table changes done + * without ptl locking. */ -static inline void unlock_vma_ref_count(struct vm_area_struct *vma) -{ - /* - * This should only be called after a corresponding, - * successful trylock_vma_ref_count(). - */ - VM_BUG_ON_VMA(atomic_cmpxchg(&vma->vm_ref_count, -1, 1) != -1, - vma); -} -#else /* !CONFIG_SPECULATIVE_PAGE_FAULT */ -static inline bool trylock_vma_ref_count(struct vm_area_struct *vma) -{ - return true; -} -static inline void unlock_vma_ref_count(struct vm_area_struct *vma) -{ -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - -#ifdef CONFIG_HAVE_MOVE_PMD +#if defined(CONFIG_HAVE_MOVE_PMD) && !defined(CONFIG_SPECULATIVE_PAGE_FAULT) static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, unsigned long new_addr, unsigned long old_end, pmd_t *old_pmd, pmd_t *new_pmd) @@ -262,14 +234,6 @@ static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, if (WARN_ON(!pmd_none(*new_pmd))) return false; - /* - * We hold both exclusive mmap_lock and rmap_lock at this point and - * cannot block. If we cannot immediately take exclusive ownership - * of the VMA fallback to the move_ptes(). - */ - if (!trylock_vma_ref_count(vma)) - return false; - /* * We don't have to worry about the ordering of src and dst * ptlocks because exclusive mmap_sem prevents deadlock. @@ -292,7 +256,6 @@ static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, spin_unlock(new_ptl); spin_unlock(old_ptl); - unlock_vma_ref_count(vma); return true; } #else From 12326201b5c5878ad5c469ea0183f084d813445f Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:52:03 +0000 Subject: [PATCH 05/48] Revert "ANDROID: mm: fix invalid backport in speculative page fault path" This reverts commit 3aa1fadec50385b8f7d0c43aadf7ce70a55e6330. Change-Id: I5dccfc1ccc41650c9d2e5f944d92c7773093524f Signed-off-by: Alexander Winkowski --- mm/memory.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 25aaec08c90f..e3a6880f7295 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4516,8 +4516,9 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, pol = __get_vma_policy(vmf.vma, address); if (!pol) pol = get_task_policy(current); - if (pol && pol->mode == MPOL_INTERLEAVE) - return VM_FAULT_RETRY; + if (!pol) + if (pol && pol->mode == MPOL_INTERLEAVE) + return VM_FAULT_RETRY; #endif /* From d45fba04cc0c430c824252c0f6d37414b7e81281 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:52:12 +0000 Subject: [PATCH 06/48] Revert "ANDROID: disable page table moves when speculative page faults are enabled" This reverts commit 78035f7a502da2bff28626bd891527aa6cbb55ce. Change-Id: I458224eceaa9c415120f91f64ec09a49ac6895a2 Signed-off-by: Alexander Winkowski --- mm/mremap.c | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index f0064331dbd3..56b2e64cd6e8 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -210,11 +210,7 @@ static void move_ptes(struct vm_area_struct *vma, pmd_t *old_pmd, drop_rmap_locks(vma); } -/* - * Speculative page fault handlers will not detect page table changes done - * without ptl locking. - */ -#if defined(CONFIG_HAVE_MOVE_PMD) && !defined(CONFIG_SPECULATIVE_PAGE_FAULT) +#ifdef CONFIG_HAVE_MOVE_PMD static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, unsigned long new_addr, unsigned long old_end, pmd_t *old_pmd, pmd_t *new_pmd) From 1185e5aca11c66840e37154dc3cf46f7d1a1e702 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:56:11 +0000 Subject: [PATCH 07/48] Revert "ANDROID: mm: assert that mmap_lock is taken exclusively in vm_write_begin" This reverts commit 51cfccaecd1813aff19b15f0ca65af12ddf36809. Change-Id: Iacc9012268a5e7aedfe38afd8766e84a042a30b5 Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 6 ------ 1 file changed, 6 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 1a67be54dc47..0304a5241875 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1613,12 +1613,6 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, #ifdef CONFIG_SPECULATIVE_PAGE_FAULT static inline void vm_write_begin(struct vm_area_struct *vma) { - /* - * Isolated vma might be freed without exclusive mmap_lock but - * speculative page fault handler still needs to know it was changed. - */ - if (!RB_EMPTY_NODE(&vma->vm_rb)) - WARN_ON_ONCE(!rwsem_is_locked(&(vma->vm_mm)->mmap_sem)); /* * The reads never spins and preemption * disablement is not required. From 924ed5256e1f6cf7faab7aceb91ac8fbab7f9fd5 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:56:23 +0000 Subject: [PATCH 08/48] Revert "ANDROID: mm: remove sequence counting when mmap_lock is not exclusively owned" This reverts commit ad939deb18f94f32016e156aff54c84d1519d1b5. Change-Id: I210d9aa077b3b68916166d2b5265ef0600c5e3bb Signed-off-by: Alexander Winkowski --- mm/madvise.c | 6 ++++++ mm/memory.c | 2 ++ mm/mempolicy.c | 2 ++ 3 files changed, 10 insertions(+) diff --git a/mm/madvise.c b/mm/madvise.c index 06072daa7b0b..023e11d4c8bf 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -501,9 +501,11 @@ static void madvise_cold_page_range(struct mmu_gather *tlb, .target_task = task, }; + vm_write_begin(vma); tlb_start_vma(tlb, vma); walk_page_range(vma->vm_mm, addr, end, &cold_walk_ops, &walk_private); tlb_end_vma(tlb, vma); + vm_write_end(vma); } static long madvise_cold(struct task_struct *task, @@ -537,9 +539,11 @@ static void madvise_pageout_page_range(struct mmu_gather *tlb, .target_task = task, }; + vm_write_begin(vma); tlb_start_vma(tlb, vma); walk_page_range(vma->vm_mm, addr, end, &cold_walk_ops, &walk_private); tlb_end_vma(tlb, vma); + vm_write_end(vma); } static inline bool can_do_pageout(struct vm_area_struct *vma) @@ -742,10 +746,12 @@ static int madvise_free_single_vma(struct vm_area_struct *vma, update_hiwater_rss(mm); mmu_notifier_invalidate_range_start(&range); + vm_write_begin(vma); tlb_start_vma(&tlb, vma); walk_page_range(vma->vm_mm, range.start, range.end, &madvise_free_walk_ops, &tlb); tlb_end_vma(&tlb, vma); + vm_write_end(vma); mmu_notifier_invalidate_range_end(&range); tlb_finish_mmu(&tlb, range.start, range.end); diff --git a/mm/memory.c b/mm/memory.c index e3a6880f7295..6ad73caf3c1c 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1293,6 +1293,7 @@ void unmap_page_range(struct mmu_gather *tlb, unsigned long next; BUG_ON(addr >= end); + vm_write_begin(vma); tlb_start_vma(tlb, vma); pgd = pgd_offset(vma->vm_mm, addr); do { @@ -1302,6 +1303,7 @@ void unmap_page_range(struct mmu_gather *tlb, next = zap_p4d_range(tlb, vma, pgd, addr, next, details); } while (pgd++, addr = next, addr != end); tlb_end_vma(tlb, vma); + vm_write_end(vma); } diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 05c02b191995..284eb71f8e75 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -600,9 +600,11 @@ unsigned long change_prot_numa(struct vm_area_struct *vma, { int nr_updated; + vm_write_begin(vma); nr_updated = change_protection(vma, addr, end, PAGE_NONE, 0, 1); if (nr_updated) count_vm_numa_events(NUMA_PTE_UPDATES, nr_updated); + vm_write_end(vma); return nr_updated; } From ee757276ff8fe191d8c6880b93b849191859ffd4 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:58:56 +0000 Subject: [PATCH 09/48] Revert "ANDROID: mm/khugepaged: add missing vm_write_{begin|end}" This reverts commit 365a5b7af5fa6b3706aa8204772921172addc714. Change-Id: Ifb555936b0887bab47590ccd77bdd435da10c179 Signed-off-by: Alexander Winkowski --- mm/khugepaged.c | 6 ------ 1 file changed, 6 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 2bde07dd5258..092bc7ae184e 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1345,8 +1345,6 @@ void collapse_pte_mapped_thp(struct mm_struct *mm, unsigned long addr) if (!pmd) goto drop_hpage; - vm_write_begin(vma); - /* * We need to lock the mapping so that from here on, only GUP-fast and * hardware page walks can access the parts of the page tables that @@ -1414,7 +1412,6 @@ void collapse_pte_mapped_thp(struct mm_struct *mm, unsigned long addr) haddr + HPAGE_PMD_SIZE); mmu_notifier_invalidate_range_start(&range); _pmd = pmdp_collapse_flush(vma, haddr, pmd); - vm_write_end(vma); mm_dec_nr_ptes(mm); tlb_remove_table_sync_one(); mmu_notifier_invalidate_range_end(&range); @@ -1431,7 +1428,6 @@ drop_hpage: abort: pte_unmap_unlock(start_pte, ptl); - vm_write_end(vma); i_mmap_unlock_write(vma->vm_file->f_mapping); goto drop_hpage; } @@ -1512,10 +1508,8 @@ static void retract_page_tables(struct address_space *mapping, pgoff_t pgoff) NULL, mm, addr, addr + HPAGE_PMD_SIZE); mmu_notifier_invalidate_range_start(&range); - vm_write_begin(vma); /* assume page table is clear */ _pmd = pmdp_collapse_flush(vma, addr, pmd); - vm_write_end(vma); mm_dec_nr_ptes(mm); tlb_remove_table_sync_one(); pte_free(mm, pmd_pgtable(_pmd)); From 9ef3adc15b3027ca21c70684dd0d5faabf1b6ce0 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:59:40 +0000 Subject: [PATCH 10/48] Revert "BACKPORT: FROMLIST: mm: implement speculative handling in filemap_fault()" This reverts commit 5b5bd362f1c527e1844c3c57f54fcb0736f998f9. Change-Id: I3197cb64766cd5e102794d24c946bb88a8e73652 Signed-off-by: Alexander Winkowski --- mm/filemap.c | 45 +-------------------------------------------- 1 file changed, 1 insertion(+), 44 deletions(-) diff --git a/mm/filemap.c b/mm/filemap.c index 623550f9a9d3..c8b28c6538a2 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -2634,9 +2634,7 @@ static struct file *do_async_mmap_readahead(struct vm_fault *vmf, * it in the page cache, and handles the special cases reasonably without * having a lot of duplicated code. * - * If FAULT_FLAG_SPECULATIVE is set, this function runs with elevated vma - * refcount and with mmap lock not held. - * Otherwise, vma->vm_mm->mmap_sem must be held on entry. + * vma->vm_mm->mmap_sem must be held on entry (except FAULT_FLAG_SPECULATIVE). * * If our return value has VM_FAULT_RETRY set, it's because the mmap_sem * may be dropped before doing I/O or by lock_page_maybe_drop_mmap(). @@ -2661,47 +2659,6 @@ vm_fault_t filemap_fault(struct vm_fault *vmf) struct page *page; vm_fault_t ret = 0; - if (vmf->flags & FAULT_FLAG_SPECULATIVE) { - page = find_get_page(mapping, offset); - if (unlikely(!page) || unlikely(PageReadahead(page))) - return VM_FAULT_RETRY; - - if (!trylock_page(page)) - return VM_FAULT_RETRY; - - if (unlikely(compound_head(page)->mapping != mapping)) - goto page_unlock; - VM_BUG_ON_PAGE(page_to_pgoff(page) != offset, page); - if (unlikely(!PageUptodate(page))) - goto page_unlock; - - max_off = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); - if (unlikely(offset >= max_off)) - goto page_unlock; - - /* - * Update readahead mmap_miss statistic. - * - * Note that we are not sure if finish_fault() will - * manage to complete the transaction. If it fails, - * we'll come back to filemap_fault() non-speculative - * case which will update mmap_miss a second time. - * This is not ideal, we would prefer to guarantee the - * update will happen exactly once. - */ - if (!(vmf->vma->vm_flags & VM_RAND_READ) && ra->ra_pages) { - unsigned int mmap_miss = READ_ONCE(ra->mmap_miss); - if (mmap_miss) - WRITE_ONCE(ra->mmap_miss, --mmap_miss); - } - - vmf->page = page; - return VM_FAULT_LOCKED; -page_unlock: - unlock_page(page); - return VM_FAULT_RETRY; - } - max_off = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); if (unlikely(offset >= max_off)) return VM_FAULT_SIGBUS; From 9c47c0190b0d7af4bc96b63fe568dadea7a5514e Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:59:50 +0000 Subject: [PATCH 11/48] Revert "ANDROID: mm: prevent reads of unstable pmd during speculation" This reverts commit f87e6b8d4578ec17a96571b76f1296d93c5972bd. Change-Id: I3a896b78e77f9db5a52bcf261a35c3ec4afefb9c Signed-off-by: Alexander Winkowski --- mm/memory.c | 25 ++++++++++++------------- 1 file changed, 12 insertions(+), 13 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 6ad73caf3c1c..5e8c197e214c 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -3467,10 +3467,6 @@ static vm_fault_t __do_fault(struct vm_fault *vmf) struct vm_area_struct *vma = vmf->vma; vm_fault_t ret; - /* Do not check unstable pmd, if it's changed will retry later */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto skip_pmd_checks; - /* * Preallocate pte before we take page_lock because this might lead to * deadlocks for memcg reclaim which waits for pages under writeback: @@ -3493,7 +3489,6 @@ static vm_fault_t __do_fault(struct vm_fault *vmf) smp_wmb(); /* See comment in __pte_alloc() */ } -skip_pmd_checks: ret = vma->vm_ops->fault(vmf); if (unlikely(ret & (VM_FAULT_ERROR | VM_FAULT_NOPAGE | VM_FAULT_RETRY | VM_FAULT_DONE_COW))) @@ -3867,8 +3862,7 @@ static vm_fault_t do_fault_around(struct vm_fault *vmf) end_pgoff = min3(end_pgoff, vma_data_pages(vmf->vma) + vmf->vma->vm_pgoff - 1, start_pgoff + nr_pages - 1); - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE) && - pmd_none(*vmf->pmd)) { + if (pmd_none(*vmf->pmd)) { vmf->prealloc_pte = pte_alloc_one(vmf->vma->vm_mm); if (!vmf->prealloc_pte) goto out; @@ -4237,11 +4231,16 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) pte_t entry; vm_fault_t ret = 0; - /* Do not check unstable pmd, if it's changed will retry later */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto skip_pmd_checks; - if (unlikely(pmd_none(*vmf->pmd))) { + /* + * In the case of the speculative page fault handler we abort + * the speculative path immediately as the pmd is probably + * in the way to be converted in a huge one. We will try + * again holding the mmap_sem (which implies that the collapse + * operation is done). + */ + if (vmf->flags & FAULT_FLAG_SPECULATIVE) + return VM_FAULT_RETRY; /* * Leave __pte_alloc() until later: because vm_ops->fault may * want to allocate huge page, and if we expose page table @@ -4249,7 +4248,8 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) * concurrent faults and from rmap lookups. */ vmf->pte = NULL; - } else { + } else if (!(vmf->flags & FAULT_FLAG_SPECULATIVE)) { + /* See comment in pte_alloc_one_map() */ if (pmd_devmap_trans_unstable(vmf->pmd)) return 0; /* @@ -4279,7 +4279,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) } } -skip_pmd_checks: if (!vmf->pte) { if (vma_is_anonymous(vmf->vma)) return do_anonymous_page(vmf); From a1d1eb8bd3f62d1a40dec0e7d40c5f7ee2b5aa5f Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 04:59:59 +0000 Subject: [PATCH 12/48] Revert "ANDROID: mm: prevent speculative page fault handling for in do_swap_page()" This reverts commit cb68c255f8dc8e960ccd9b61ff584ba419923880. Change-Id: I3deaf837842423b91b1cca32ec974d9612c882b7 Signed-off-by: Alexander Winkowski --- mm/memory.c | 5 ----- 1 file changed, 5 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 5e8c197e214c..1df7a5f0fec8 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -3090,11 +3090,6 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) int exclusive = 0; vm_fault_t ret; - if (vmf->flags & FAULT_FLAG_SPECULATIVE) { - pte_unmap(vmf->pte); - return VM_FAULT_RETRY; - } - ret = pte_unmap_same(vmf); if (ret) { /* From f3fcee6d90a11d138372f13d703552f9c6ae6d8b Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:00:10 +0000 Subject: [PATCH 13/48] Revert "ANDROID: mm: skip pte_alloc during speculative page fault" This reverts commit b12d6d9506435b10f588075a8bdf537004d1afbd. Change-Id: I0891749da9dc05c7f9a3ad909a9d054e74a45e2b Signed-off-by: Alexander Winkowski --- mm/memory.c | 5 ----- 1 file changed, 5 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 1df7a5f0fec8..2a059e328d0d 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -3336,10 +3336,6 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) if (vmf->vma_flags & VM_SHARED) return VM_FAULT_SIGBUS; - /* Do not check unstable pmd, if it's changed will retry later */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto skip_pmd_checks; - /* * Use pte_alloc() instead of pte_alloc_map(). We can't run * pte_offset_map() on pmds where a huge pmd might be created @@ -3357,7 +3353,6 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) if (unlikely(pmd_trans_unstable(vmf->pmd))) return 0; -skip_pmd_checks: /* Use the zero-page for reads */ if (!(vmf->flags & FAULT_FLAG_WRITE) && !mm_forbids_zeropage(vma->vm_mm)) { From d74f249592a6fce13adc6179e3acb1ba5538e0d0 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:01:30 +0000 Subject: [PATCH 14/48] Revert "ANDROID: mm: Fix page table lookup in speculative fault path" This reverts commit 1b3c72b43c03ef66265e94434869089d34345f77. Change-Id: I03356916c3b27db106e75ed07b3d53c2479926e7 Signed-off-by: Alexander Winkowski --- mm/memory.c | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 2a059e328d0d..c4a183d27e45 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4522,15 +4522,11 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, goto out_walk; p4d = p4d_offset(pgd, address); - if (pgd_val(READ_ONCE(*pgd)) != pgd_val(pgdval)) - goto out_walk; p4dval = READ_ONCE(*p4d); if (p4d_none(p4dval) || unlikely(p4d_bad(p4dval))) goto out_walk; vmf.pud = pud_offset(p4d, address); - if (p4d_val(READ_ONCE(*p4d)) != p4d_val(p4dval)) - goto out_walk; pudval = READ_ONCE(*vmf.pud); if (pud_none(pudval) || unlikely(pud_bad(pudval))) goto out_walk; @@ -4540,8 +4536,6 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, goto out_walk; vmf.pmd = pmd_offset(vmf.pud, address); - if (pud_val(READ_ONCE(*vmf.pud)) != pud_val(pudval)) - goto out_walk; vmf.orig_pmd = READ_ONCE(*vmf.pmd); /* * pmd_none could mean that a hugepage collapse is in progress @@ -4569,11 +4563,6 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, */ vmf.pte = pte_offset_map(vmf.pmd, address); - if (pmd_val(READ_ONCE(*vmf.pmd)) != pmd_val(vmf.orig_pmd)) { - pte_unmap(vmf.pte); - vmf.pte = NULL; - goto out_walk; - } vmf.orig_pte = READ_ONCE(*vmf.pte); barrier(); /* See comment in handle_pte_fault() */ if (pte_none(vmf.orig_pte)) { From 5f80d03c8ccefc18b587166cd95291621f5acfab Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:02:10 +0000 Subject: [PATCH 15/48] Revert "ANDROID: mm: use raw seqcount variants in vm_write_*" This reverts commit f2c1ef71ed4dcdf40fa3220837634d753dcc5fb8. Change-Id: I5dc242f3d5aa05eb4630426d7444c36412ed6191 Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 29 ++++++++++++++++++++++++----- mm/mmap.c | 38 +++++++++++++++++++++++++++++--------- mm/mremap.c | 8 ++++---- 3 files changed, 57 insertions(+), 18 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 0304a5241875..72d43be61a97 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1613,13 +1613,22 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, #ifdef CONFIG_SPECULATIVE_PAGE_FAULT static inline void vm_write_begin(struct vm_area_struct *vma) { - /* - * The reads never spins and preemption - * disablement is not required. - */ - raw_write_seqcount_begin(&vma->vm_sequence); + write_seqcount_begin(&vma->vm_sequence); +} +static inline void vm_write_begin_nested(struct vm_area_struct *vma, + int subclass) +{ + write_seqcount_begin_nested(&vma->vm_sequence, subclass); } static inline void vm_write_end(struct vm_area_struct *vma) +{ + write_seqcount_end(&vma->vm_sequence); +} +static inline void vm_raw_write_begin(struct vm_area_struct *vma) +{ + raw_write_seqcount_begin(&vma->vm_sequence); +} +static inline void vm_raw_write_end(struct vm_area_struct *vma) { raw_write_seqcount_end(&vma->vm_sequence); } @@ -1627,9 +1636,19 @@ static inline void vm_write_end(struct vm_area_struct *vma) static inline void vm_write_begin(struct vm_area_struct *vma) { } +static inline void vm_write_begin_nested(struct vm_area_struct *vma, + int subclass) +{ +} static inline void vm_write_end(struct vm_area_struct *vma) { } +static inline void vm_raw_write_begin(struct vm_area_struct *vma) +{ +} +static inline void vm_raw_write_end(struct vm_area_struct *vma) +{ +} #endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ extern void truncate_pagecache(struct inode *inode, loff_t new); diff --git a/mm/mmap.c b/mm/mmap.c index eedd39382561..bc234247a09f 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -767,9 +767,29 @@ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, long adjust_next = 0; int remove_next = 0; - vm_write_begin(vma); + /* + * Why using vm_raw_write*() functions here to avoid lockdep's warning ? + * + * Locked is complaining about a theoretical lock dependency, involving + * 3 locks: + * mapping->i_mmap_rwsem --> vma->vm_sequence --> fs_reclaim + * + * Here are the major path leading to this dependency : + * 1. __vma_adjust() mmap_sem -> vm_sequence -> i_mmap_rwsem + * 2. move_vmap() mmap_sem -> vm_sequence -> fs_reclaim + * 3. __alloc_pages_nodemask() fs_reclaim -> i_mmap_rwsem + * 4. unmap_mapping_range() i_mmap_rwsem -> vm_sequence + * + * So there is no way to solve this easily, especially because in + * unmap_mapping_range() the i_mmap_rwsem is grab while the impacted + * VMAs are not yet known. + * However, the way the vm_seq is used is guarantying that we will + * never block on it since we just check for its value and never wait + * for it to move, see vma_has_changed() and handle_speculative_fault(). + */ + vm_raw_write_begin(vma); if (next) - vm_write_begin(next); + vm_raw_write_begin(next); if (next && !insert) { struct vm_area_struct *exporter = NULL, *importer = NULL; @@ -853,8 +873,8 @@ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, error = anon_vma_clone(importer, exporter); if (error) { if (next && next != vma) - vm_write_end(next); - vm_write_end(vma); + vm_raw_write_end(next); + vm_raw_write_end(vma); return error; } } @@ -983,7 +1003,7 @@ again: if (next->anon_vma) anon_vma_merge(vma, next); mm->map_count--; - vm_write_end(next); + vm_raw_write_end(next); put_vma(next); /* * In mprotect's case 6 (see comments on vma_merge), @@ -999,7 +1019,7 @@ again: */ next = vma->vm_next; if (next) - vm_write_begin(next); + vm_raw_write_begin(next); } else { /* * For the scope of the comment "next" and @@ -1047,9 +1067,9 @@ again: uprobe_mmap(insert); if (next && next != vma) - vm_write_end(next); + vm_raw_write_end(next); if (!keep_locked) - vm_write_end(vma); + vm_raw_write_end(vma); validate_mm(mm); @@ -3409,7 +3429,7 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, * that we protect it right now, and let the caller unprotect * it once the move is done. */ - vm_write_begin(new_vma); + vm_raw_write_begin(new_vma); vma_link(mm, new_vma, prev, rb_link, rb_parent); *need_rmap_locks = false; } diff --git a/mm/mremap.c b/mm/mremap.c index 56b2e64cd6e8..0485add3399f 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -533,7 +533,7 @@ static unsigned long move_vma(struct vm_area_struct *vma, * to be mapped in our back while we are copying the PTEs. */ if (vma != new_vma) - vm_write_begin(vma); + vm_raw_write_begin(vma); moved_len = move_page_tables(vma, old_addr, new_vma, new_addr, old_len, need_rmap_locks); @@ -552,7 +552,7 @@ static unsigned long move_vma(struct vm_area_struct *vma, move_page_tables(new_vma, new_addr, vma, old_addr, moved_len, true); if (vma != new_vma) - vm_write_end(vma); + vm_raw_write_end(vma); vma = new_vma; old_len = new_len; old_addr = new_addr; @@ -562,9 +562,9 @@ static unsigned long move_vma(struct vm_area_struct *vma, arch_remap(mm, old_addr, old_addr + old_len, new_addr, new_addr + new_len); if (vma != new_vma) - vm_write_end(vma); + vm_raw_write_end(vma); } - vm_write_end(new_vma); + vm_raw_write_end(new_vma); /* Conceal VM_ACCOUNT so old reservation is not undone */ if (vm_flags & VM_ACCOUNT && !(flags & MREMAP_DONTUNMAP)) { From ec928f71b9cb2a9754951cd7f748444f5a76ced6 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:03:35 +0000 Subject: [PATCH 16/48] Revert "mm: fix non-anon COW fault" This reverts commit 1b7ca44c7114e98c299c94899ed6fd1ab2280ea0. Change-Id: I3feb16538efb3b7260cfac8404f9ad68b150a64c Signed-off-by: Alexander Winkowski --- mm/memory.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index c4a183d27e45..04906d1d7be7 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4461,7 +4461,7 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, if (unlikely((vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma) || (!vma_is_anonymous(vmf.vma) && !(vmf.vma->vm_flags & VM_SHARED) && - (flags & FAULT_FLAG_WRITE) && + (vmf.flags & FAULT_FLAG_WRITE) && !vmf.vma->anon_vma))) return VM_FAULT_RETRY; From 8b720d1a140fda6ecbf4eee3a82b2e0064330343 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:04:10 +0000 Subject: [PATCH 17/48] Revert "mm: skip speculative path for non-anonymous COW faults" This reverts commit d95ca82a5de955f2d17731d40deb55ebb9832131. Change-Id: I65cd50b4aab3578fdb2ae1b67d25a84775b19c4e Signed-off-by: Alexander Winkowski --- mm/memory.c | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 04906d1d7be7..bdeb832a7dc5 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4458,11 +4458,7 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * because vm_next and vm_prev must be safe. This can't be guaranteed * in the speculative path. */ - if (unlikely((vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma) || - (!vma_is_anonymous(vmf.vma) && - !(vmf.vma->vm_flags & VM_SHARED) && - (vmf.flags & FAULT_FLAG_WRITE) && - !vmf.vma->anon_vma))) + if (unlikely(vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma)) return VM_FAULT_RETRY; vmf.vma_flags = READ_ONCE(vmf.vma->vm_flags); From d4119e83817fc9bd308f36fa09adc75826d0d608 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:10:42 +0000 Subject: [PATCH 18/48] Revert "mm: sync rss in speculative page fault path" This reverts commit 6011ee7ef26289bc13117e00d62704549524c078. Change-Id: I9192d3ad2a32e2f284248d6ccdc936eed48a7a61 Signed-off-by: Alexander Winkowski --- mm/memory.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index bdeb832a7dc5..a4a1ea1d0587 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4588,7 +4588,6 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * If there is no need to retry, don't return the vma to the caller. */ if (ret != VM_FAULT_RETRY) { - check_sync_rss_stat(current); if (vma_is_anonymous(vmf.vma)) count_vm_event(SPECULATIVE_PGFAULT_ANON); else From a93728d35e0650f8301f8c7a7b13cc4a973dcdfd Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 05:11:47 +0000 Subject: [PATCH 19/48] Revert "mm: remove the speculative page fault traces" This reverts commit f96d3d9c9e8c8a08989edfe4d12c797595dd72b6. Change-Id: Ib7723f11c64b05a989113078b2f3c90b00d65946 Signed-off-by: Alexander Winkowski --- mm/memory.c | 56 +++++++++++++++++++++++++++++++++++++++++------------ 1 file changed, 44 insertions(+), 12 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index a4a1ea1d0587..bbd1616db8c9 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -85,6 +85,9 @@ #include "internal.h" +#define CREATE_TRACE_POINTS +#include + #if defined(LAST_CPUPID_NOT_IN_PAGE_FLAGS) && !defined(CONFIG_COMPILE_TEST) #warning Unfortunate NUMA and NUMA Balancing config, growing page-frame for last_cpupid. #endif @@ -2236,8 +2239,10 @@ static bool pte_spinlock(struct vm_fault *vmf) } local_irq_disable(); - if (vma_has_changed(vmf)) + if (vma_has_changed(vmf)) { + trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); goto out; + } #ifdef CONFIG_TRANSPARENT_HUGEPAGE /* @@ -2245,16 +2250,21 @@ static bool pte_spinlock(struct vm_fault *vmf) * is not a huge collapse operation in progress in our back. */ pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) + if (!pmd_same(pmdval, vmf->orig_pmd)) { + trace_spf_pmd_changed(_RET_IP_, vmf->vma, vmf->address); goto out; + } #endif vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - if (unlikely(!spin_trylock(vmf->ptl))) + if (unlikely(!spin_trylock(vmf->ptl))) { + trace_spf_pte_lock(_RET_IP_, vmf->vma, vmf->address); goto out; + } if (vma_has_changed(vmf)) { spin_unlock(vmf->ptl); + trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); goto out; } @@ -2287,8 +2297,10 @@ static bool pte_map_lock(struct vm_fault *vmf) * block on the PTL and thus we're safe. */ local_irq_disable(); - if (vma_has_changed(vmf)) + if (vma_has_changed(vmf)) { + trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); goto out; + } #ifdef CONFIG_TRANSPARENT_HUGEPAGE /* @@ -2296,8 +2308,10 @@ static bool pte_map_lock(struct vm_fault *vmf) * is not a huge collapse operation in progress in our back. */ pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) + if (!pmd_same(pmdval, vmf->orig_pmd)) { + trace_spf_pmd_changed(_RET_IP_, vmf->vma, vmf->address); goto out; + } #endif /* @@ -2311,11 +2325,13 @@ static bool pte_map_lock(struct vm_fault *vmf) pte = pte_offset_map(vmf->pmd, vmf->address); if (unlikely(!spin_trylock(ptl))) { pte_unmap(pte); + trace_spf_pte_lock(_RET_IP_, vmf->vma, vmf->address); goto out; } if (vma_has_changed(vmf)) { pte_unmap_unlock(pte, ptl); + trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); goto out; } @@ -4450,35 +4466,45 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, /* rmb <-> seqlock,vma_rb_erase() */ seq = raw_read_seqcount(&vmf.vma->vm_sequence); - if (seq & 1) + if (seq & 1) { + trace_spf_vma_changed(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } /* * __anon_vma_prepare() requires the mmap_sem to be held * because vm_next and vm_prev must be safe. This can't be guaranteed * in the speculative path. */ - if (unlikely(vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma)) + if (unlikely(vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma)) { + trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } vmf.vma_flags = READ_ONCE(vmf.vma->vm_flags); vmf.vma_page_prot = READ_ONCE(vmf.vma->vm_page_prot); /* Can't call userland page fault handler in the speculative path */ - if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) + if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) { + trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } - if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) + if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) { /* * This could be detected by the check address against VMA's * boundaries but we want to trace it as not supported instead * of changed. */ + trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } if (address < READ_ONCE(vmf.vma->vm_start) - || READ_ONCE(vmf.vma->vm_end) <= address) + || READ_ONCE(vmf.vma->vm_end) <= address) { + trace_spf_vma_changed(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } if (!arch_vma_access_permitted(vmf.vma, flags & FAULT_FLAG_WRITE, flags & FAULT_FLAG_INSTRUCTION, @@ -4504,8 +4530,10 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, if (!pol) pol = get_task_policy(current); if (!pol) - if (pol && pol->mode == MPOL_INTERLEAVE) + if (pol && pol->mode == MPOL_INTERLEAVE) { + trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } #endif /* @@ -4577,8 +4605,10 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * We need to re-validate the VMA after checking the bounds, otherwise * we might have a false positive on the bounds. */ - if (read_seqcount_retry(&vmf.vma->vm_sequence, seq)) + if (read_seqcount_retry(&vmf.vma->vm_sequence, seq)) { + trace_spf_vma_changed(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; + } mem_cgroup_enter_user_fault(); ret = handle_pte_fault(&vmf); @@ -4607,10 +4637,12 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, return ret; out_walk: + trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); local_irq_enable(); return VM_FAULT_RETRY; out_segv: + trace_spf_vma_access(_RET_IP_, vmf.vma, address); /* * We don't return VM_FAULT_RETRY so the caller is not expected to * retrieve the fetched VMA. From a01be3aa0293bdaec084bc034668cdc27eccf7e3 Mon Sep 17 00:00:00 2001 From: Martin Liu Date: Sun, 12 Apr 2020 01:24:16 +0800 Subject: [PATCH 20/48] Revert "mm: allow vmas with vm_ops to be speculatively handled" This reverts commit b37bae60c52687eacf67169fbf5c6b675e8ce196. Reason for revert: remove SPF non upstream code Bug: 140544941 Test: boot Change-Id: I466435dabfed767085934109a43bf7ca3da855a3 Signed-off-by: Martin Liu [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/vm_event_item.h | 3 +-- mm/filemap.c | 8 ++++---- mm/memory.c | 24 +++++++++++++++--------- mm/shmem.c | 4 ++-- mm/vmstat.c | 3 +-- 5 files changed, 23 insertions(+), 19 deletions(-) diff --git a/include/linux/vm_event_item.h b/include/linux/vm_event_item.h index 7f221e8a87cc..0fa409e827bb 100644 --- a/include/linux/vm_event_item.h +++ b/include/linux/vm_event_item.h @@ -115,8 +115,7 @@ enum vm_event_item { PGPGIN, PGPGOUT, SWAP_RA_HIT, #endif #ifdef CONFIG_SPECULATIVE_PAGE_FAULT - SPECULATIVE_PGFAULT_ANON, - SPECULATIVE_PGFAULT_FILE, + SPECULATIVE_PGFAULT, #endif NR_VM_EVENT_ITEMS }; diff --git a/mm/filemap.c b/mm/filemap.c index c8b28c6538a2..3eff4f1388b2 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -2562,12 +2562,12 @@ static struct file *do_sync_mmap_readahead(struct vm_fault *vmf) pgoff_t offset = vmf->pgoff; /* If we don't want any read-ahead, don't bother */ - if (vmf->vma_flags & VM_RAND_READ) + if (vmf->vma->vm_flags & VM_RAND_READ) return fpin; if (!ra->ra_pages) return fpin; - if (vmf->vma_flags & VM_SEQ_READ) { + if (vmf->vma->vm_flags & VM_SEQ_READ) { fpin = maybe_unlock_mmap_for_io(vmf, fpin); page_cache_sync_readahead(mapping, ra, file, offset, ra->ra_pages); @@ -2611,7 +2611,7 @@ static struct file *do_async_mmap_readahead(struct vm_fault *vmf, pgoff_t offset = vmf->pgoff; /* If we don't want any read-ahead, don't bother */ - if (vmf->vma_flags & VM_RAND_READ || !ra->ra_pages) + if (vmf->vma->vm_flags & VM_RAND_READ || !ra->ra_pages) return fpin; if (ra->mmap_miss > 0) ra->mmap_miss--; @@ -2634,7 +2634,7 @@ static struct file *do_async_mmap_readahead(struct vm_fault *vmf, * it in the page cache, and handles the special cases reasonably without * having a lot of duplicated code. * - * vma->vm_mm->mmap_sem must be held on entry (except FAULT_FLAG_SPECULATIVE). + * vma->vm_mm->mmap_sem must be held on entry. * * If our return value has VM_FAULT_RETRY set, it's because the mmap_sem * may be dropped before doing I/O or by lock_page_maybe_drop_mmap(). diff --git a/mm/memory.c b/mm/memory.c index bbd1616db8c9..faffb6adb66f 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4235,7 +4235,6 @@ static vm_fault_t wp_huge_pud(struct vm_fault *vmf, pud_t orig_pud) static vm_fault_t handle_pte_fault(struct vm_fault *vmf) { pte_t entry; - vm_fault_t ret = 0; if (unlikely(pmd_none(*vmf->pmd))) { /* @@ -4288,6 +4287,8 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) if (!vmf->pte) { if (vma_is_anonymous(vmf->vma)) return do_anonymous_page(vmf); + else if (vmf->flags & FAULT_FLAG_SPECULATIVE) + return VM_FAULT_RETRY; else return do_fault(vmf); } @@ -4321,12 +4322,10 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) */ if (vmf->flags & FAULT_FLAG_WRITE) flush_tlb_fix_spurious_fault(vmf->vma, vmf->address); - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - ret = VM_FAULT_RETRY; } unlock: pte_unmap_unlock(vmf->pte, vmf->ptl); - return ret; + return 0; } /* @@ -4471,12 +4470,22 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, return VM_FAULT_RETRY; } + /* + * Can't call vm_ops service has we don't know what they would do + * with the VMA. + * This include huge page from hugetlbfs. + */ + if (vmf.vma->vm_ops) { + trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); + return VM_FAULT_RETRY; + } + /* * __anon_vma_prepare() requires the mmap_sem to be held * because vm_next and vm_prev must be safe. This can't be guaranteed * in the speculative path. */ - if (unlikely(vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma)) { + if (unlikely(!vmf.vma->anon_vma)) { trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); return VM_FAULT_RETRY; } @@ -4618,10 +4627,7 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * If there is no need to retry, don't return the vma to the caller. */ if (ret != VM_FAULT_RETRY) { - if (vma_is_anonymous(vmf.vma)) - count_vm_event(SPECULATIVE_PGFAULT_ANON); - else - count_vm_event(SPECULATIVE_PGFAULT_FILE); + count_vm_event(SPECULATIVE_PGFAULT); put_vma(vmf.vma); *vma = NULL; } diff --git a/mm/shmem.c b/mm/shmem.c index 9ae6831e7b07..e2f72b220988 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2072,10 +2072,10 @@ static vm_fault_t shmem_fault(struct vm_fault *vmf) sgp = SGP_CACHE; - if ((vmf->vma_flags & VM_NOHUGEPAGE) || + if ((vma->vm_flags & VM_NOHUGEPAGE) || test_bit(MMF_DISABLE_THP, &vma->vm_mm->flags)) sgp = SGP_NOHUGE; - else if (vmf->vma_flags & VM_HUGEPAGE) + else if (vma->vm_flags & VM_HUGEPAGE) sgp = SGP_HUGE; err = shmem_getpage_gfp(inode, vmf->pgoff, &vmf->page, sgp, diff --git a/mm/vmstat.c b/mm/vmstat.c index af290e42d497..8c7e6e718a5f 100644 --- a/mm/vmstat.c +++ b/mm/vmstat.c @@ -1301,8 +1301,7 @@ const char * const vmstat_text[] = { "swap_ra_hit", #endif #ifdef CONFIG_SPECULATIVE_PAGE_FAULT - "speculative_pgfault_anon", - "speculative_pgfault_file", + "speculative_pgfault" #endif #endif /* CONFIG_VM_EVENT_COUNTERS */ }; From 6c8feaacfcddfb8ce04547142b9252d49aeb6519 Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 16:41:21 +0000 Subject: [PATCH 21/48] Revert "mm: Fix sleeping while atomic during speculative page fault" This reverts commit 8fbac439e0b7a5b875c8e914a5649b4fc9bd7415. Change-Id: Ifa8e76de6c80a7453969df8bd431b94f9e66e547 Signed-off-by: Alexander Winkowski --- mm/memory.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index faffb6adb66f..3419f522ba85 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -3540,7 +3540,7 @@ static vm_fault_t pte_alloc_one_map(struct vm_fault *vmf) { struct vm_area_struct *vma = vmf->vma; - if (!pmd_none(*vmf->pmd) || (vmf->flags & FAULT_FLAG_SPECULATIVE)) + if (!pmd_none(*vmf->pmd)) goto map_pte; if (vmf->prealloc_pte) { vmf->ptl = pmd_lock(vma->vm_mm, vmf->pmd); From fb1f97dab890e46b74ad8e4df3505a8ebc678066 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:54:50 +0900 Subject: [PATCH 22/48] Revert "mm: don't do swap readahead during speculative page fault" This reverts commit c41742ec7ca571b523137e2811cefc0307d1539d. Bug: 128240262 Change-Id: Ide6a3bb060fd1792c1ac5880cfc2298931ad07ed Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- mm/memory.c | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 3419f522ba85..59885e20e165 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -3156,17 +3156,6 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) lru_cache_add_anon(page); swap_readpage(page, true); } - } else if (vmf->flags & FAULT_FLAG_SPECULATIVE) { - /* - * Don't try readahead during a speculative page fault - * as the VMA's boundaries may change in our back. - * If the page is not in the swap cache and synchronous - * read is disabled, fall back to the regular page fault - * mechanism. - */ - delayacct_clear_flag(DELAYACCT_PF_SWAPIN); - ret = VM_FAULT_RETRY; - goto out; } else { page = swapin_readahead(entry, GFP_HIGHUSER_MOVABLE, vmf); From 37d40ca0495ae531d94671e11f9db5cbb25f2e3e Mon Sep 17 00:00:00 2001 From: Martin Liu Date: Sun, 12 Apr 2020 01:32:15 +0800 Subject: [PATCH 23/48] Revert "mm: protect against PTE changes done by dup_mmap()" This reverts commit a9e3a1ab5dee4ecb0ec59426e10ebfcd48a54990. Reason for revert: remove SPF non upstream code Bug: 140544941 Test: boot Change-Id: I912a8891ac6cf3e72c7b7aa27df2922554b31491 Signed-off-by: Martin Liu [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- kernel/fork.c | 30 ++---------------------------- 1 file changed, 2 insertions(+), 28 deletions(-) diff --git a/kernel/fork.c b/kernel/fork.c index 74f6f5e26ee4..0db40e63ae28 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -486,7 +486,7 @@ EXPORT_SYMBOL(free_task); static __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm) { - struct vm_area_struct *mpnt, *tmp, *prev, **pprev, *last = NULL; + struct vm_area_struct *mpnt, *tmp, *prev, **pprev; struct rb_node **rb_link, *rb_parent; int retval; unsigned long charge; @@ -605,18 +605,8 @@ static __latent_entropy int dup_mmap(struct mm_struct *mm, rb_parent = &tmp->vm_rb; mm->map_count++; - if (!(tmp->vm_flags & VM_WIPEONFORK)) { - if (IS_ENABLED(CONFIG_SPECULATIVE_PAGE_FAULT)) { - /* - * Mark this VMA as changing to prevent the - * speculative page fault hanlder to process - * it until the TLB are flushed below. - */ - last = mpnt; - vm_write_begin(mpnt); - } + if (!(tmp->vm_flags & VM_WIPEONFORK)) retval = copy_page_range(mm, oldmm, mpnt); - } if (tmp->vm_ops && tmp->vm_ops->open) tmp->vm_ops->open(tmp); @@ -629,22 +619,6 @@ static __latent_entropy int dup_mmap(struct mm_struct *mm, out: up_write(&mm->mmap_sem); flush_tlb_mm(oldmm); - - if (IS_ENABLED(CONFIG_SPECULATIVE_PAGE_FAULT)) { - /* - * Since the TLB has been flush, we can safely unmark the - * copied VMAs and allows the speculative page fault handler to - * process them again. - * Walk back the VMA list from the last marked VMA. - */ - for (; last; last = last->vm_prev) { - if (last->vm_flags & VM_DONTCOPY) - continue; - if (!(last->vm_flags & VM_WIPEONFORK)) - vm_write_end(last); - } - } - up_write(&oldmm->mmap_sem); dup_userfaultfd_complete(&uf); fail_uprobe_end: From 561478b68519a3554e647658543cad5ca130422b Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:55:15 +0900 Subject: [PATCH 24/48] Revert "arm64/mm: add speculative page fault" This reverts commit ad3023acf917f89266afa98f0576fa5e3e187f9f. Bug: 128240262 Change-Id: I80f12ab9b25478a13b69d9c8fe7b46b3f65b1197 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- arch/arm64/mm/fault.c | 26 +++----------------------- 1 file changed, 3 insertions(+), 23 deletions(-) diff --git a/arch/arm64/mm/fault.c b/arch/arm64/mm/fault.c index 78fe15dd857d..8fcd1dbe1fb7 100644 --- a/arch/arm64/mm/fault.c +++ b/arch/arm64/mm/fault.c @@ -410,9 +410,10 @@ static void do_bad_area(unsigned long addr, unsigned int esr, struct pt_regs *re #define VM_FAULT_BADMAP ((__force vm_fault_t)0x010000) #define VM_FAULT_BADACCESS ((__force vm_fault_t)0x020000) -static int __do_page_fault(struct vm_area_struct *vma, unsigned long addr, +static vm_fault_t __do_page_fault(struct mm_struct *mm, unsigned long addr, unsigned int mm_flags, unsigned long vm_flags) { + struct vm_area_struct *vma = find_vma(mm, addr); if (unlikely(!vma)) return VM_FAULT_BADMAP; @@ -459,7 +460,6 @@ static int __kprobes do_page_fault(unsigned long addr, unsigned int esr, vm_fault_t fault, major = 0; unsigned long vm_flags = VM_READ | VM_WRITE | VM_EXEC; unsigned int mm_flags = FAULT_FLAG_DEFAULT; - struct vm_area_struct *vma = NULL; if (kprobe_page_fault(regs, esr)) return 0; @@ -499,14 +499,6 @@ static int __kprobes do_page_fault(unsigned long addr, unsigned int esr, perf_sw_event(PERF_COUNT_SW_PAGE_FAULTS, 1, regs, addr); - /* - * let's try a speculative page fault without grabbing the - * mmap_sem. - */ - fault = handle_speculative_fault(mm, addr, mm_flags, &vma); - if (fault != VM_FAULT_RETRY) - goto done; - /* * As per x86, we may deadlock here. However, since the kernel only * validly references user space from well defined areas of the code, @@ -531,10 +523,7 @@ retry: #endif } - if (!vma || !can_reuse_spf_vma(vma, addr)) - vma = find_vma(mm, addr); - - fault = __do_page_fault(vma, addr, mm_flags, vm_flags); + fault = __do_page_fault(mm, addr, mm_flags, vm_flags); major |= fault & VM_FAULT_MAJOR; /* Quick path to respond to signals */ @@ -547,20 +536,11 @@ retry: if (fault & VM_FAULT_RETRY) { if (mm_flags & FAULT_FLAG_ALLOW_RETRY) { mm_flags |= FAULT_FLAG_TRIED; - - /* - * Do not try to reuse this vma and fetch it - * again since we will release the mmap_sem. - */ - vma = NULL; - goto retry; } } up_read(&mm->mmap_sem); -done: - /* * Handle the "normal" (no error) case first. */ From 402afac2ab80e0f69f4291c5608ff0ae2849aa3f Mon Sep 17 00:00:00 2001 From: Martin Liu Date: Sun, 12 Apr 2020 01:31:51 +0800 Subject: [PATCH 25/48] Revert "arm64/mm: define ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT" This reverts commit bcca0756b31719a2d969642d65e12136f490c5f2. Reason for revert: remove SPF non upstream code Bug: 140544941 Test: boot Change-Id: Id97756b85be0a1690e000fd24f125f21915d20da Signed-off-by: Martin Liu [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- arch/arm64/Kconfig | 1 - 1 file changed, 1 deletion(-) diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index 3a942bbf01c6..bd9e46bbaa8d 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -191,7 +191,6 @@ config ARM64 select SYSCTL_EXCEPTION_TRACE select THREAD_INFO_IN_TASK select HAVE_ARCH_USERFAULTFD_MINOR if USERFAULTFD - select ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT help ARM 64-bit (AArch64) Linux support. From f2f6a4239e15fb990f4cb884c757973f151741f8 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:37:42 +0900 Subject: [PATCH 26/48] Revert "mm: add speculative page fault vmstats" This reverts commit 2842d9234e61dc7ae96500caa4ad48320ba60147. Bug: 128240262 Change-Id: I3ba5c0ab739af15061dd376fcb15366e9644ed4f Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/vm_event_item.h | 3 --- mm/memory.c | 1 - mm/vmstat.c | 5 +---- 3 files changed, 1 insertion(+), 8 deletions(-) diff --git a/include/linux/vm_event_item.h b/include/linux/vm_event_item.h index 0fa409e827bb..5a43dc19c317 100644 --- a/include/linux/vm_event_item.h +++ b/include/linux/vm_event_item.h @@ -113,9 +113,6 @@ enum vm_event_item { PGPGIN, PGPGOUT, #ifdef CONFIG_SWAP SWAP_RA, SWAP_RA_HIT, -#endif -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - SPECULATIVE_PGFAULT, #endif NR_VM_EVENT_ITEMS }; diff --git a/mm/memory.c b/mm/memory.c index 59885e20e165..b086424815fc 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4616,7 +4616,6 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * If there is no need to retry, don't return the vma to the caller. */ if (ret != VM_FAULT_RETRY) { - count_vm_event(SPECULATIVE_PGFAULT); put_vma(vmf.vma); *vma = NULL; } diff --git a/mm/vmstat.c b/mm/vmstat.c index 8c7e6e718a5f..10d281bfd71b 100644 --- a/mm/vmstat.c +++ b/mm/vmstat.c @@ -1300,10 +1300,7 @@ const char * const vmstat_text[] = { "swap_ra", "swap_ra_hit", #endif -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - "speculative_pgfault" -#endif -#endif /* CONFIG_VM_EVENT_COUNTERS */ +#endif /* CONFIG_VM_EVENTS_COUNTERS */ }; #endif /* CONFIG_PROC_FS || CONFIG_SYSFS || CONFIG_NUMA */ From a45b2e061ea5ebd9284bc7549993713f1dc37202 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:37:52 +0900 Subject: [PATCH 27/48] Revert "mm: speculative page fault handler return VMA" This reverts commit f556cd74a7ea038c67812b0d9b5d971d03da5372. Bug: 128240262 Change-Id: Icc3edff62de261aee10658b1567b85ff7b5d58dd Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 22 ++----- mm/memory.c | 139 ++++++++++++++++++--------------------------- 2 files changed, 59 insertions(+), 102 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 72d43be61a97..33f76e7aa7d7 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1666,37 +1666,25 @@ extern vm_fault_t handle_mm_fault(struct vm_area_struct *vma, #ifdef CONFIG_SPECULATIVE_PAGE_FAULT extern int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, - unsigned int flags, - struct vm_area_struct **vma); + unsigned int flags); static inline int handle_speculative_fault(struct mm_struct *mm, unsigned long address, - unsigned int flags, - struct vm_area_struct **vma) + unsigned int flags) { /* * Try speculative page fault for multithreaded user space task only. */ - if (!(flags & FAULT_FLAG_USER) || atomic_read(&mm->mm_users) == 1) { - *vma = NULL; + if (!(flags & FAULT_FLAG_USER) || atomic_read(&mm->mm_users) == 1) return VM_FAULT_RETRY; - } - return __handle_speculative_fault(mm, address, flags, vma); + return __handle_speculative_fault(mm, address, flags); } -extern bool can_reuse_spf_vma(struct vm_area_struct *vma, - unsigned long address); #else static inline int handle_speculative_fault(struct mm_struct *mm, unsigned long address, - unsigned int flags, - struct vm_area_struct **vma) + unsigned int flags) { return VM_FAULT_RETRY; } -static inline bool can_reuse_spf_vma(struct vm_area_struct *vma, - unsigned long address) -{ - return false; -} #endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ extern int fixup_user_fault(struct task_struct *tsk, struct mm_struct *mm, diff --git a/mm/memory.c b/mm/memory.c index b086424815fc..05e1c2be9cc8 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4418,22 +4418,13 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, /* This is required by vm_normal_page() */ #error "Speculative page fault handler requires CONFIG_ARCH_HAS_PTE_SPECIAL" #endif + /* * vm_normal_page() adds some processing which should be done while * hodling the mmap_sem. */ - -/* - * Tries to handle the page fault in a speculative way, without grabbing the - * mmap_sem. - * When VM_FAULT_RETRY is returned, the vma pointer is valid and this vma must - * be checked later when the mmap_sem has been grabbed by calling - * can_reuse_spf_vma(). - * This is needed as the returned vma is kept in memory until the call to - * can_reuse_spf_vma() is made. - */ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, - unsigned int flags, struct vm_area_struct **vma) + unsigned int flags) { struct vm_fault vmf = { .address = address, @@ -4441,22 +4432,22 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, pgd_t *pgd, pgdval; p4d_t *p4d, p4dval; pud_t pudval; - int seq, ret; + int seq, ret = VM_FAULT_RETRY; + struct vm_area_struct *vma; /* Clear flags that may lead to release the mmap_sem to retry */ flags &= ~(FAULT_FLAG_ALLOW_RETRY|FAULT_FLAG_KILLABLE); flags |= FAULT_FLAG_SPECULATIVE; - *vma = get_vma(mm, address); - if (!*vma) - return VM_FAULT_RETRY; - vmf.vma = *vma; + vma = get_vma(mm, address); + if (!vma) + return ret; /* rmb <-> seqlock,vma_rb_erase() */ - seq = raw_read_seqcount(&vmf.vma->vm_sequence); + seq = raw_read_seqcount(&vma->vm_sequence); if (seq & 1) { - trace_spf_vma_changed(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + trace_spf_vma_changed(_RET_IP_, vma, address); + goto out_put; } /* @@ -4464,9 +4455,9 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * with the VMA. * This include huge page from hugetlbfs. */ - if (vmf.vma->vm_ops) { - trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + if (vma->vm_ops) { + trace_spf_vma_notsup(_RET_IP_, vma, address); + goto out_put; } /* @@ -4474,18 +4465,18 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * because vm_next and vm_prev must be safe. This can't be guaranteed * in the speculative path. */ - if (unlikely(!vmf.vma->anon_vma)) { - trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + if (unlikely(!vma->anon_vma)) { + trace_spf_vma_notsup(_RET_IP_, vma, address); + goto out_put; } - vmf.vma_flags = READ_ONCE(vmf.vma->vm_flags); - vmf.vma_page_prot = READ_ONCE(vmf.vma->vm_page_prot); + vmf.vma_flags = READ_ONCE(vma->vm_flags); + vmf.vma_page_prot = READ_ONCE(vma->vm_page_prot); /* Can't call userland page fault handler in the speculative path */ if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) { - trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + trace_spf_vma_notsup(_RET_IP_, vma, address); + goto out_put; } if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) { @@ -4494,27 +4485,36 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * boundaries but we want to trace it as not supported instead * of changed. */ - trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + trace_spf_vma_notsup(_RET_IP_, vma, address); + goto out_put; } - if (address < READ_ONCE(vmf.vma->vm_start) - || READ_ONCE(vmf.vma->vm_end) <= address) { - trace_spf_vma_changed(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + if (address < READ_ONCE(vma->vm_start) + || READ_ONCE(vma->vm_end) <= address) { + trace_spf_vma_changed(_RET_IP_, vma, address); + goto out_put; } - if (!arch_vma_access_permitted(vmf.vma, flags & FAULT_FLAG_WRITE, + if (!arch_vma_access_permitted(vma, flags & FAULT_FLAG_WRITE, flags & FAULT_FLAG_INSTRUCTION, - flags & FAULT_FLAG_REMOTE)) - goto out_segv; + flags & FAULT_FLAG_REMOTE)) { + trace_spf_vma_access(_RET_IP_, vma, address); + ret = VM_FAULT_SIGSEGV; + goto out_put; + } /* This is one is required to check that the VMA has write access set */ if (flags & FAULT_FLAG_WRITE) { - if (unlikely(!(vmf.vma_flags & VM_WRITE))) - goto out_segv; - } else if (unlikely(!(vmf.vma_flags & (VM_READ|VM_EXEC|VM_WRITE)))) - goto out_segv; + if (unlikely(!(vmf.vma_flags & VM_WRITE))) { + trace_spf_vma_access(_RET_IP_, vma, address); + ret = VM_FAULT_SIGSEGV; + goto out_put; + } + } else if (unlikely(!(vmf.vma_flags & (VM_READ|VM_EXEC|VM_WRITE)))) { + trace_spf_vma_access(_RET_IP_, vma, address); + ret = VM_FAULT_SIGSEGV; + goto out_put; + } #ifdef CONFIG_NUMA struct mempolicy *pol; @@ -4524,13 +4524,13 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * mpol_misplaced() which are not compatible with the *speculative page fault processing. */ - pol = __get_vma_policy(vmf.vma, address); + pol = __get_vma_policy(vma, address); if (!pol) pol = get_task_policy(current); if (!pol) if (pol && pol->mode == MPOL_INTERLEAVE) { - trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + trace_spf_vma_notsup(_RET_IP_, vma, address); + goto out_put; } #endif @@ -4592,8 +4592,9 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, vmf.pte = NULL; } - vmf.pgoff = linear_page_index(vmf.vma, address); - vmf.gfp_mask = __get_fault_gfp_mask(vmf.vma); + vmf.vma = vma; + vmf.pgoff = linear_page_index(vma, address); + vmf.gfp_mask = __get_fault_gfp_mask(vma); vmf.sequence = seq; vmf.flags = flags; @@ -4603,22 +4604,16 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * We need to re-validate the VMA after checking the bounds, otherwise * we might have a false positive on the bounds. */ - if (read_seqcount_retry(&vmf.vma->vm_sequence, seq)) { - trace_spf_vma_changed(_RET_IP_, vmf.vma, address); - return VM_FAULT_RETRY; + if (read_seqcount_retry(&vma->vm_sequence, seq)) { + trace_spf_vma_changed(_RET_IP_, vma, address); + goto out_put; } mem_cgroup_enter_user_fault(); ret = handle_pte_fault(&vmf); mem_cgroup_exit_user_fault(); - /* - * If there is no need to retry, don't return the vma to the caller. - */ - if (ret != VM_FAULT_RETRY) { - put_vma(vmf.vma); - *vma = NULL; - } + put_vma(vma); /* * The task may have entered a memcg OOM situation but @@ -4631,35 +4626,9 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, return ret; out_walk: - trace_spf_vma_notsup(_RET_IP_, vmf.vma, address); + trace_spf_vma_notsup(_RET_IP_, vma, address); local_irq_enable(); - return VM_FAULT_RETRY; - -out_segv: - trace_spf_vma_access(_RET_IP_, vmf.vma, address); - /* - * We don't return VM_FAULT_RETRY so the caller is not expected to - * retrieve the fetched VMA. - */ - put_vma(vmf.vma); - *vma = NULL; - return VM_FAULT_SIGSEGV; -} - -/* - * This is used to know if the vma fetch in the speculative page fault handler - * is still valid when trying the regular fault path while holding the - * mmap_sem. - * The call to put_vma(vma) must be made after checking the vma's fields, as - * the vma may be freed by put_vma(). In such a case it is expected that false - * is returned. - */ -bool can_reuse_spf_vma(struct vm_area_struct *vma, unsigned long address) -{ - bool ret; - - ret = !RB_EMPTY_NODE(&vma->vm_rb) && - vma->vm_start <= address && address < vma->vm_end; +out_put: put_vma(vma); return ret; } From fb8981f8817898a702d6bca9b814287e988bcb33 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:38:15 +0900 Subject: [PATCH 28/48] Revert "mm: adding speculative page fault failure trace events" This reverts commit 52e2e1f805cb0136bdd6664817204e18bed95637. Bug: 128240262 Change-Id: I96fe4f7e02f5d6716e20b1f8bc03b686dd189f09 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/trace/events/pagefault.h | 88 -------------------------------- mm/memory.c | 65 +++++------------------ 2 files changed, 14 insertions(+), 139 deletions(-) delete mode 100644 include/trace/events/pagefault.h diff --git a/include/trace/events/pagefault.h b/include/trace/events/pagefault.h deleted file mode 100644 index a9643b3759f2..000000000000 --- a/include/trace/events/pagefault.h +++ /dev/null @@ -1,88 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#undef TRACE_SYSTEM -#define TRACE_SYSTEM pagefault - -#if !defined(_TRACE_PAGEFAULT_H) || defined(TRACE_HEADER_MULTI_READ) -#define _TRACE_PAGEFAULT_H - -#include -#include - -DECLARE_EVENT_CLASS(spf, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address), - - TP_STRUCT__entry( - __field(unsigned long, caller) - __field(unsigned long, vm_start) - __field(unsigned long, vm_end) - __field(unsigned long, address) - ), - - TP_fast_assign( - __entry->caller = caller; - __entry->vm_start = vma->vm_start; - __entry->vm_end = vma->vm_end; - __entry->address = address; - ), - - TP_printk("ip:%lx vma:%lx-%lx address:%lx", - __entry->caller, __entry->vm_start, __entry->vm_end, - __entry->address) -); - -DEFINE_EVENT(spf, spf_pte_lock, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_changed, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_noanon, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_notsup, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_access, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_pmd_changed, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -#endif /* _TRACE_PAGEFAULT_H */ - -/* This part must be outside protection */ -#include diff --git a/mm/memory.c b/mm/memory.c index 05e1c2be9cc8..dd9803c62184 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -85,9 +85,6 @@ #include "internal.h" -#define CREATE_TRACE_POINTS -#include - #if defined(LAST_CPUPID_NOT_IN_PAGE_FLAGS) && !defined(CONFIG_COMPILE_TEST) #warning Unfortunate NUMA and NUMA Balancing config, growing page-frame for last_cpupid. #endif @@ -2239,10 +2236,8 @@ static bool pte_spinlock(struct vm_fault *vmf) } local_irq_disable(); - if (vma_has_changed(vmf)) { - trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); + if (vma_has_changed(vmf)) goto out; - } #ifdef CONFIG_TRANSPARENT_HUGEPAGE /* @@ -2250,21 +2245,16 @@ static bool pte_spinlock(struct vm_fault *vmf) * is not a huge collapse operation in progress in our back. */ pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) { - trace_spf_pmd_changed(_RET_IP_, vmf->vma, vmf->address); + if (!pmd_same(pmdval, vmf->orig_pmd)) goto out; - } #endif vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - if (unlikely(!spin_trylock(vmf->ptl))) { - trace_spf_pte_lock(_RET_IP_, vmf->vma, vmf->address); + if (unlikely(!spin_trylock(vmf->ptl))) goto out; - } if (vma_has_changed(vmf)) { spin_unlock(vmf->ptl); - trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); goto out; } @@ -2297,10 +2287,8 @@ static bool pte_map_lock(struct vm_fault *vmf) * block on the PTL and thus we're safe. */ local_irq_disable(); - if (vma_has_changed(vmf)) { - trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); + if (vma_has_changed(vmf)) goto out; - } #ifdef CONFIG_TRANSPARENT_HUGEPAGE /* @@ -2308,10 +2296,8 @@ static bool pte_map_lock(struct vm_fault *vmf) * is not a huge collapse operation in progress in our back. */ pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) { - trace_spf_pmd_changed(_RET_IP_, vmf->vma, vmf->address); + if (!pmd_same(pmdval, vmf->orig_pmd)) goto out; - } #endif /* @@ -2325,13 +2311,11 @@ static bool pte_map_lock(struct vm_fault *vmf) pte = pte_offset_map(vmf->pmd, vmf->address); if (unlikely(!spin_trylock(ptl))) { pte_unmap(pte); - trace_spf_pte_lock(_RET_IP_, vmf->vma, vmf->address); goto out; } if (vma_has_changed(vmf)) { pte_unmap_unlock(pte, ptl); - trace_spf_vma_changed(_RET_IP_, vmf->vma, vmf->address); goto out; } @@ -4445,60 +4429,47 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, /* rmb <-> seqlock,vma_rb_erase() */ seq = raw_read_seqcount(&vma->vm_sequence); - if (seq & 1) { - trace_spf_vma_changed(_RET_IP_, vma, address); + if (seq & 1) goto out_put; - } /* * Can't call vm_ops service has we don't know what they would do * with the VMA. * This include huge page from hugetlbfs. */ - if (vma->vm_ops) { - trace_spf_vma_notsup(_RET_IP_, vma, address); + if (vma->vm_ops) goto out_put; - } /* * __anon_vma_prepare() requires the mmap_sem to be held * because vm_next and vm_prev must be safe. This can't be guaranteed * in the speculative path. */ - if (unlikely(!vma->anon_vma)) { - trace_spf_vma_notsup(_RET_IP_, vma, address); + if (unlikely(!vma->anon_vma)) goto out_put; - } vmf.vma_flags = READ_ONCE(vma->vm_flags); vmf.vma_page_prot = READ_ONCE(vma->vm_page_prot); /* Can't call userland page fault handler in the speculative path */ - if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) { - trace_spf_vma_notsup(_RET_IP_, vma, address); + if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) goto out_put; - } - if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) { + if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) /* * This could be detected by the check address against VMA's * boundaries but we want to trace it as not supported instead * of changed. */ - trace_spf_vma_notsup(_RET_IP_, vma, address); goto out_put; - } if (address < READ_ONCE(vma->vm_start) - || READ_ONCE(vma->vm_end) <= address) { - trace_spf_vma_changed(_RET_IP_, vma, address); + || READ_ONCE(vma->vm_end) <= address) goto out_put; - } if (!arch_vma_access_permitted(vma, flags & FAULT_FLAG_WRITE, flags & FAULT_FLAG_INSTRUCTION, flags & FAULT_FLAG_REMOTE)) { - trace_spf_vma_access(_RET_IP_, vma, address); ret = VM_FAULT_SIGSEGV; goto out_put; } @@ -4506,12 +4477,10 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, /* This is one is required to check that the VMA has write access set */ if (flags & FAULT_FLAG_WRITE) { if (unlikely(!(vmf.vma_flags & VM_WRITE))) { - trace_spf_vma_access(_RET_IP_, vma, address); ret = VM_FAULT_SIGSEGV; goto out_put; } } else if (unlikely(!(vmf.vma_flags & (VM_READ|VM_EXEC|VM_WRITE)))) { - trace_spf_vma_access(_RET_IP_, vma, address); ret = VM_FAULT_SIGSEGV; goto out_put; } @@ -4527,11 +4496,8 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, pol = __get_vma_policy(vma, address); if (!pol) pol = get_task_policy(current); - if (!pol) - if (pol && pol->mode == MPOL_INTERLEAVE) { - trace_spf_vma_notsup(_RET_IP_, vma, address); - goto out_put; - } + if (pol && pol->mode == MPOL_INTERLEAVE) + goto out_put; #endif /* @@ -4604,10 +4570,8 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, * We need to re-validate the VMA after checking the bounds, otherwise * we might have a false positive on the bounds. */ - if (read_seqcount_retry(&vma->vm_sequence, seq)) { - trace_spf_vma_changed(_RET_IP_, vma, address); + if (read_seqcount_retry(&vma->vm_sequence, seq)) goto out_put; - } mem_cgroup_enter_user_fault(); ret = handle_pte_fault(&vmf); @@ -4626,7 +4590,6 @@ int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, return ret; out_walk: - trace_spf_vma_notsup(_RET_IP_, vma, address); local_irq_enable(); out_put: put_vma(vma); From c0e5f947b4fe8b48cd9c87b79da26409d9165a2c Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:39:02 +0900 Subject: [PATCH 29/48] Revert "mm: provide speculative fault infrastructure" This reverts commit 6e9deb2ea7ea1d63a5c1516c9acc1efeb182977e. Bug: 128240262 Change-Id: I1060d8e43cebca1826a489fa3271279501c8647c Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/hugetlb_inline.h | 2 +- include/linux/mm.h | 31 --- include/linux/pagemap.h | 4 +- mm/internal.h | 16 +- mm/memory.c | 342 +-------------------------------- 5 files changed, 7 insertions(+), 388 deletions(-) diff --git a/include/linux/hugetlb_inline.h b/include/linux/hugetlb_inline.h index 9e25283d6fc9..0660a03d37d9 100644 --- a/include/linux/hugetlb_inline.h +++ b/include/linux/hugetlb_inline.h @@ -8,7 +8,7 @@ static inline bool is_vm_hugetlb_page(struct vm_area_struct *vma) { - return !!(READ_ONCE(vma->vm_flags) & VM_HUGETLB); + return !!(vma->vm_flags & VM_HUGETLB); } #else diff --git a/include/linux/mm.h b/include/linux/mm.h index 33f76e7aa7d7..a218570076aa 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -417,7 +417,6 @@ extern pgprot_t protection_map[16]; * @FAULT_FLAG_INSTRUCTION: The fault was during an instruction fetch. * @FAULT_FLAG_INTERRUPTIBLE: The fault can be interrupted by non-fatal signals. * @FAULT_FLAG_PREFAULT_OLD: Make faultaround ptes old. - * @FAULT_FLAG_SPECULATIVE: Speculative fault, not holding mmap_sem. * * About @FAULT_FLAG_ALLOW_RETRY and @FAULT_FLAG_TRIED: we can specify * whether we would allow page faults to retry by specifying these two @@ -449,7 +448,6 @@ extern pgprot_t protection_map[16]; #define FAULT_FLAG_INSTRUCTION 0x100 #define FAULT_FLAG_INTERRUPTIBLE 0x200 #define FAULT_FLAG_PREFAULT_OLD 0x400 -#define FAULT_FLAG_SPECULATIVE 0x800 /* * The default fault flags that should be used by most of the @@ -505,10 +503,6 @@ struct vm_fault { gfp_t gfp_mask; /* gfp mask to be used for allocations */ pgoff_t pgoff; /* Logical page offset based on vma */ unsigned long address; /* Faulting virtual address */ -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - unsigned int sequence; - pmd_t orig_pmd; /* value of PMD at the time of fault */ -#endif pmd_t *pmd; /* Pointer to pmd entry matching * the 'address' */ pud_t *pud; /* Pointer to pud entry matching @@ -1662,31 +1656,6 @@ int invalidate_inode_page(struct page *page); #ifdef CONFIG_MMU extern vm_fault_t handle_mm_fault(struct vm_area_struct *vma, unsigned long address, unsigned int flags); - -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -extern int __handle_speculative_fault(struct mm_struct *mm, - unsigned long address, - unsigned int flags); -static inline int handle_speculative_fault(struct mm_struct *mm, - unsigned long address, - unsigned int flags) -{ - /* - * Try speculative page fault for multithreaded user space task only. - */ - if (!(flags & FAULT_FLAG_USER) || atomic_read(&mm->mm_users) == 1) - return VM_FAULT_RETRY; - return __handle_speculative_fault(mm, address, flags); -} -#else -static inline int handle_speculative_fault(struct mm_struct *mm, - unsigned long address, - unsigned int flags) -{ - return VM_FAULT_RETRY; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - extern int fixup_user_fault(struct task_struct *tsk, struct mm_struct *mm, unsigned long address, unsigned int fault_flags, bool *unlocked); diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 5ccb6baf050f..207432ae6e19 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -494,8 +494,8 @@ static inline pgoff_t linear_page_index(struct vm_area_struct *vma, pgoff_t pgoff; if (unlikely(is_vm_hugetlb_page(vma))) return linear_hugepage_index(vma, address); - pgoff = (address - READ_ONCE(vma->vm_start)) >> PAGE_SHIFT; - pgoff += READ_ONCE(vma->vm_pgoff); + pgoff = (address - vma->vm_start) >> PAGE_SHIFT; + pgoff += vma->vm_pgoff; return pgoff; } diff --git a/mm/internal.h b/mm/internal.h index a3e960c45322..62bc7d64ca30 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -40,21 +40,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf); extern struct vm_area_struct *get_vma(struct mm_struct *mm, unsigned long addr); extern void put_vma(struct vm_area_struct *vma); - -static inline bool vma_has_changed(struct vm_fault *vmf) -{ - int ret = RB_EMPTY_NODE(&vmf->vma->vm_rb); - unsigned int seq = READ_ONCE(vmf->vma->vm_sequence.sequence); - - /* - * Matches both the wmb in write_seqlock_{begin,end}() and - * the wmb in vma_rb_erase(). - */ - smp_rmb(); - - return ret || seq != vmf->sequence; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ +#endif void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *start_vma, unsigned long floor, unsigned long ceiling); diff --git a/mm/memory.c b/mm/memory.c index dd9803c62184..01890bca9578 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -555,7 +555,7 @@ static void print_bad_pte(struct vm_area_struct *vma, unsigned long addr, if (page) dump_page(page, "bad pte"); pr_alert("addr:%px vm_flags:%08lx anon_vma:%px mapping:%px index:%lx\n", - (void *)addr, READ_ONCE(vma->vm_flags), vma->anon_vma, mapping, index); + (void *)addr, vma->vm_flags, vma->anon_vma, mapping, index); pr_alert("file:%pD fault:%ps mmap:%ps readpage:%ps\n", vma->vm_file, vma->vm_ops ? vma->vm_ops->fault : NULL, @@ -2220,113 +2220,6 @@ int apply_to_page_range(struct mm_struct *mm, unsigned long addr, } EXPORT_SYMBOL_GPL(apply_to_page_range); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -static bool pte_spinlock(struct vm_fault *vmf) -{ - bool ret = false; -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - pmd_t pmdval; -#endif - - /* Check if vma is still valid */ - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE)) { - vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - spin_lock(vmf->ptl); - return true; - } - - local_irq_disable(); - if (vma_has_changed(vmf)) - goto out; - -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - /* - * We check if the pmd value is still the same to ensure that there - * is not a huge collapse operation in progress in our back. - */ - pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) - goto out; -#endif - - vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - if (unlikely(!spin_trylock(vmf->ptl))) - goto out; - - if (vma_has_changed(vmf)) { - spin_unlock(vmf->ptl); - goto out; - } - - ret = true; -out: - local_irq_enable(); - return ret; -} - -static bool pte_map_lock(struct vm_fault *vmf) -{ - bool ret = false; - pte_t *pte; - spinlock_t *ptl; -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - pmd_t pmdval; -#endif - - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE)) { - vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, - vmf->address, &vmf->ptl); - return true; - } - - /* - * The first vma_has_changed() guarantees the page-tables are still - * valid, having IRQs disabled ensures they stay around, hence the - * second vma_has_changed() to make sure they are still valid once - * we've got the lock. After that a concurrent zap_pte_range() will - * block on the PTL and thus we're safe. - */ - local_irq_disable(); - if (vma_has_changed(vmf)) - goto out; - -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - /* - * We check if the pmd value is still the same to ensure that there - * is not a huge collapse operation in progress in our back. - */ - pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) - goto out; -#endif - - /* - * Same as pte_offset_map_lock() except that we call - * spin_trylock() in place of spin_lock() to avoid race with - * unmap path which may have the lock and wait for this CPU - * to invalidate TLB but this CPU has irq disabled. - * Since we are in a speculative patch, accept it could fail - */ - ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - pte = pte_offset_map(vmf->pmd, vmf->address); - if (unlikely(!spin_trylock(ptl))) { - pte_unmap(pte); - goto out; - } - - if (vma_has_changed(vmf)) { - pte_unmap_unlock(pte, ptl); - goto out; - } - - vmf->pte = pte; - vmf->ptl = ptl; - ret = true; -out: - local_irq_enable(); - return ret; -} -#else static inline bool pte_spinlock(struct vm_fault *vmf) { vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); @@ -2340,7 +2233,6 @@ static inline bool pte_map_lock(struct vm_fault *vmf) vmf->address, &vmf->ptl); return true; } -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ /* * handle_pte_fault chooses page fault handler according to an entry which was @@ -3354,14 +3246,6 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) ret = check_stable_address_space(vma->vm_mm); if (ret) goto unlock; - /* - * Don't call the userfaultfd during the speculative path. - * We already checked for the VMA to not be managed through - * userfaultfd, but it may be set in our back once we have lock - * the pte. In such a case we can ignore it this time. - */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto setpte; /* Deliver the page fault to userland, check inside PT lock */ if (userfaultfd_missing(vma)) { pte_unmap_unlock(vmf->pte, vmf->ptl); @@ -3404,8 +3288,7 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) goto unlock_and_release; /* Deliver the page fault to userland, check inside PT lock */ - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE) && - userfaultfd_missing(vma)) { + if (userfaultfd_missing(vma)) { pte_unmap_unlock(vmf->pte, vmf->ptl); mem_cgroup_cancel_charge(page, memcg, false); put_page(page); @@ -4210,15 +4093,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) pte_t entry; if (unlikely(pmd_none(*vmf->pmd))) { - /* - * In the case of the speculative page fault handler we abort - * the speculative path immediately as the pmd is probably - * in the way to be converted in a huge one. We will try - * again holding the mmap_sem (which implies that the collapse - * operation is done). - */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - return VM_FAULT_RETRY; /* * Leave __pte_alloc() until later: because vm_ops->fault may * want to allocate huge page, and if we expose page table @@ -4226,7 +4100,7 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) * concurrent faults and from rmap lookups. */ vmf->pte = NULL; - } else if (!(vmf->flags & FAULT_FLAG_SPECULATIVE)) { + } else { /* See comment in pte_alloc_one_map() */ if (pmd_devmap_trans_unstable(vmf->pmd)) return 0; @@ -4235,9 +4109,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) * pmd from under us anymore at this point because we hold the * mmap_sem read mode and khugepaged takes it in write mode. * So now it's safe to run pte_offset_map(). - * This is not applicable to the speculative page fault handler - * but in that case, the pte is fetched earlier in - * handle_speculative_fault(). */ vmf->pte = pte_offset_map(vmf->pmd, vmf->address); vmf->orig_pte = *vmf->pte; @@ -4260,8 +4131,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) if (!vmf->pte) { if (vma_is_anonymous(vmf->vma)) return do_anonymous_page(vmf); - else if (vmf->flags & FAULT_FLAG_SPECULATIVE) - return VM_FAULT_RETRY; else return do_fault(vmf); } @@ -4359,10 +4228,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, vmf.pmd = pmd_alloc(mm, vmf.pud, address); if (!vmf.pmd) return VM_FAULT_OOM; - -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - vmf.sequence = raw_read_seqcount(&vma->vm_sequence); -#endif if (pmd_none(*vmf.pmd) && __transparent_hugepage_enabled(vma)) { ret = create_huge_pmd(&vmf); if (!(ret & VM_FAULT_FALLBACK)) @@ -4396,207 +4261,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, return handle_pte_fault(&vmf); } -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - -#ifndef CONFIG_ARCH_HAS_PTE_SPECIAL -/* This is required by vm_normal_page() */ -#error "Speculative page fault handler requires CONFIG_ARCH_HAS_PTE_SPECIAL" -#endif - -/* - * vm_normal_page() adds some processing which should be done while - * hodling the mmap_sem. - */ -int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, - unsigned int flags) -{ - struct vm_fault vmf = { - .address = address, - }; - pgd_t *pgd, pgdval; - p4d_t *p4d, p4dval; - pud_t pudval; - int seq, ret = VM_FAULT_RETRY; - struct vm_area_struct *vma; - - /* Clear flags that may lead to release the mmap_sem to retry */ - flags &= ~(FAULT_FLAG_ALLOW_RETRY|FAULT_FLAG_KILLABLE); - flags |= FAULT_FLAG_SPECULATIVE; - - vma = get_vma(mm, address); - if (!vma) - return ret; - - /* rmb <-> seqlock,vma_rb_erase() */ - seq = raw_read_seqcount(&vma->vm_sequence); - if (seq & 1) - goto out_put; - - /* - * Can't call vm_ops service has we don't know what they would do - * with the VMA. - * This include huge page from hugetlbfs. - */ - if (vma->vm_ops) - goto out_put; - - /* - * __anon_vma_prepare() requires the mmap_sem to be held - * because vm_next and vm_prev must be safe. This can't be guaranteed - * in the speculative path. - */ - if (unlikely(!vma->anon_vma)) - goto out_put; - - vmf.vma_flags = READ_ONCE(vma->vm_flags); - vmf.vma_page_prot = READ_ONCE(vma->vm_page_prot); - - /* Can't call userland page fault handler in the speculative path */ - if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) - goto out_put; - - if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) - /* - * This could be detected by the check address against VMA's - * boundaries but we want to trace it as not supported instead - * of changed. - */ - goto out_put; - - if (address < READ_ONCE(vma->vm_start) - || READ_ONCE(vma->vm_end) <= address) - goto out_put; - - if (!arch_vma_access_permitted(vma, flags & FAULT_FLAG_WRITE, - flags & FAULT_FLAG_INSTRUCTION, - flags & FAULT_FLAG_REMOTE)) { - ret = VM_FAULT_SIGSEGV; - goto out_put; - } - - /* This is one is required to check that the VMA has write access set */ - if (flags & FAULT_FLAG_WRITE) { - if (unlikely(!(vmf.vma_flags & VM_WRITE))) { - ret = VM_FAULT_SIGSEGV; - goto out_put; - } - } else if (unlikely(!(vmf.vma_flags & (VM_READ|VM_EXEC|VM_WRITE)))) { - ret = VM_FAULT_SIGSEGV; - goto out_put; - } - -#ifdef CONFIG_NUMA - struct mempolicy *pol; - - /* - * MPOL_INTERLEAVE implies additional checks in - * mpol_misplaced() which are not compatible with the - *speculative page fault processing. - */ - pol = __get_vma_policy(vma, address); - if (!pol) - pol = get_task_policy(current); - if (pol && pol->mode == MPOL_INTERLEAVE) - goto out_put; -#endif - - /* - * Do a speculative lookup of the PTE entry. - */ - local_irq_disable(); - pgd = pgd_offset(mm, address); - pgdval = READ_ONCE(*pgd); - if (pgd_none(pgdval) || unlikely(pgd_bad(pgdval))) - goto out_walk; - - p4d = p4d_offset(pgd, address); - p4dval = READ_ONCE(*p4d); - if (p4d_none(p4dval) || unlikely(p4d_bad(p4dval))) - goto out_walk; - - vmf.pud = pud_offset(p4d, address); - pudval = READ_ONCE(*vmf.pud); - if (pud_none(pudval) || unlikely(pud_bad(pudval))) - goto out_walk; - - /* Huge pages at PUD level are not supported. */ - if (unlikely(pud_trans_huge(pudval))) - goto out_walk; - - vmf.pmd = pmd_offset(vmf.pud, address); - vmf.orig_pmd = READ_ONCE(*vmf.pmd); - /* - * pmd_none could mean that a hugepage collapse is in progress - * in our back as collapse_huge_page() mark it before - * invalidating the pte (which is done once the IPI is catched - * by all CPU and we have interrupt disabled). - * For this reason we cannot handle THP in a speculative way since we - * can't safely indentify an in progress collapse operation done in our - * back on that PMD. - * Regarding the order of the following checks, see comment in - * pmd_devmap_trans_unstable() - */ - if (unlikely(pmd_devmap(vmf.orig_pmd) || - pmd_none(vmf.orig_pmd) || pmd_trans_huge(vmf.orig_pmd) || - is_swap_pmd(vmf.orig_pmd))) - goto out_walk; - - /* - * The above does not allocate/instantiate page-tables because doing so - * would lead to the possibility of instantiating page-tables after - * free_pgtables() -- and consequently leaking them. - * - * The result is that we take at least one !speculative fault per PMD - * in order to instantiate it. - */ - - vmf.pte = pte_offset_map(vmf.pmd, address); - vmf.orig_pte = READ_ONCE(*vmf.pte); - barrier(); /* See comment in handle_pte_fault() */ - if (pte_none(vmf.orig_pte)) { - pte_unmap(vmf.pte); - vmf.pte = NULL; - } - - vmf.vma = vma; - vmf.pgoff = linear_page_index(vma, address); - vmf.gfp_mask = __get_fault_gfp_mask(vma); - vmf.sequence = seq; - vmf.flags = flags; - - local_irq_enable(); - - /* - * We need to re-validate the VMA after checking the bounds, otherwise - * we might have a false positive on the bounds. - */ - if (read_seqcount_retry(&vma->vm_sequence, seq)) - goto out_put; - - mem_cgroup_enter_user_fault(); - ret = handle_pte_fault(&vmf); - mem_cgroup_exit_user_fault(); - - put_vma(vma); - - /* - * The task may have entered a memcg OOM situation but - * if the allocation error was handled gracefully (no - * VM_FAULT_OOM), there is no need to kill anything. - * Just clean up the OOM state peacefully. - */ - if (task_in_memcg_oom(current) && !(ret & VM_FAULT_OOM)) - mem_cgroup_oom_synchronize(false); - return ret; - -out_walk: - local_irq_enable(); -out_put: - put_vma(vma); - return ret; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - /* * By the time we get here, we already hold the mm semaphore * From fa0483bf625e3ac032574d58a731415cdb672740 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:39:45 +0900 Subject: [PATCH 30/48] Revert "mm: protect mm_rb tree with a rwlock" This reverts commit 9db216b5c9b0f9199b904d1e8ccd72c36518607f. Bug: 128240262 Change-Id: I210789054d394b929de6d9444040f1d2aac14917 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 1 - include/linux/mm_types.h | 4 -- kernel/fork.c | 3 -- mm/init-mm.c | 3 -- mm/internal.h | 6 --- mm/mmap.c | 114 ++++++++++----------------------------- 6 files changed, 28 insertions(+), 103 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index a218570076aa..3c4c40f0ba39 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -623,7 +623,6 @@ static inline void INIT_VMA(struct vm_area_struct *vma) INIT_LIST_HEAD(&vma->anon_vma_chain); #ifdef CONFIG_SPECULATIVE_PAGE_FAULT seqcount_init(&vma->vm_sequence); - atomic_set(&vma->vm_ref_count, 1); #endif } diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index cc6185a0e05f..894a443f2a47 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -364,7 +364,6 @@ struct vm_area_struct { struct vm_userfaultfd_ctx vm_userfaultfd_ctx; #ifdef CONFIG_SPECULATIVE_PAGE_FAULT seqcount_t vm_sequence; - atomic_t vm_ref_count; /* see vma_get(), vma_put() */ #endif ANDROID_KABI_RESERVE(1); ANDROID_KABI_RESERVE(2); @@ -390,9 +389,6 @@ struct mm_struct { struct vm_area_struct *mmap; /* list of VMAs */ struct rb_root mm_rb; u64 vmacache_seqnum; /* per-thread vmacache */ -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - rwlock_t mm_rb_lock; -#endif #ifdef CONFIG_MMU unsigned long (*get_unmapped_area) (struct file *filp, unsigned long addr, unsigned long len, diff --git a/kernel/fork.c b/kernel/fork.c index 0db40e63ae28..fa92b2ab4081 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1036,9 +1036,6 @@ static struct mm_struct *mm_init(struct mm_struct *mm, struct task_struct *p, mm->mmap = NULL; mm->mm_rb = RB_ROOT; mm->vmacache_seqnum = 0; -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - rwlock_init(&mm->mm_rb_lock); -#endif atomic_set(&mm->mm_users, 1); atomic_set(&mm->mm_count, 1); init_rwsem(&mm->mmap_sem); diff --git a/mm/init-mm.c b/mm/init-mm.c index 0a4abdcf8886..19603302a77f 100644 --- a/mm/init-mm.c +++ b/mm/init-mm.c @@ -28,9 +28,6 @@ */ struct mm_struct init_mm = { .mm_rb = RB_ROOT, -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - .mm_rb_lock = __RW_LOCK_UNLOCKED(init_mm.mm_rb_lock), -#endif .pgd = swapper_pg_dir, .mm_users = ATOMIC_INIT(2), .mm_count = ATOMIC_INIT(1), diff --git a/mm/internal.h b/mm/internal.h index 62bc7d64ca30..698713321327 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -36,12 +36,6 @@ void page_writeback_init(void); vm_fault_t do_swap_page(struct vm_fault *vmf); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -extern struct vm_area_struct *get_vma(struct mm_struct *mm, - unsigned long addr); -extern void put_vma(struct vm_area_struct *vma); -#endif - void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *start_vma, unsigned long floor, unsigned long ceiling); diff --git a/mm/mmap.c b/mm/mmap.c index bc234247a09f..b4ba606a734c 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -166,27 +166,6 @@ void unlink_file_vma(struct vm_area_struct *vma) } } -static void __free_vma(struct vm_area_struct *vma) -{ - if (vma->vm_file) - fput(vma->vm_file); - mpol_put(vma_policy(vma)); - vm_area_free(vma); -} - -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -void put_vma(struct vm_area_struct *vma) -{ - if (atomic_dec_and_test(&vma->vm_ref_count)) - __free_vma(vma); -} -#else -static inline void put_vma(struct vm_area_struct *vma) -{ - __free_vma(vma); -} -#endif - /* * Close a vm structure and free it, returning the next. */ @@ -197,7 +176,10 @@ static struct vm_area_struct *remove_vma(struct vm_area_struct *vma) might_sleep(); if (vma->vm_ops && vma->vm_ops->close) vma->vm_ops->close(vma); - put_vma(vma); + if (vma->vm_file) + fput(vma->vm_file); + mpol_put(vma_policy(vma)); + vm_area_free(vma); return next; } @@ -450,13 +432,6 @@ static void validate_mm(struct mm_struct *mm) RB_DECLARE_CALLBACKS_MAX(static, vma_gap_callbacks, struct vm_area_struct, vm_rb, unsigned long, rb_subtree_gap, vma_compute_gap) -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -#define mm_rb_write_lock(mm) write_lock(&(mm)->mm_rb_lock) -#define mm_rb_write_unlock(mm) write_unlock(&(mm)->mm_rb_lock) -#else -#define mm_rb_write_lock(mm) do { } while (0) -#define mm_rb_write_unlock(mm) do { } while (0) -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ /* * Update augmented rbtree rb_subtree_gap values after vma->vm_start or @@ -473,37 +448,26 @@ static void vma_gap_update(struct vm_area_struct *vma) } static inline void vma_rb_insert(struct vm_area_struct *vma, - struct mm_struct *mm) + struct rb_root *root) { - struct rb_root *root = &mm->mm_rb; - /* All rb_subtree_gap values must be consistent prior to insertion */ validate_mm_rb(root, NULL); rb_insert_augmented(&vma->vm_rb, root, &vma_gap_callbacks); } -static void __vma_rb_erase(struct vm_area_struct *vma, struct mm_struct *mm) +static void __vma_rb_erase(struct vm_area_struct *vma, struct rb_root *root) { - struct rb_root *root = &mm->mm_rb; /* * Note rb_erase_augmented is a fairly large inline function, * so make sure we instantiate it only once with our desired * augmented rbtree callbacks. */ - mm_rb_write_lock(mm); rb_erase_augmented(&vma->vm_rb, root, &vma_gap_callbacks); - mm_rb_write_unlock(mm); /* wmb */ - - /* - * Ensure the removal is complete before clearing the node. - * Matched by vma_has_changed()/handle_speculative_fault(). - */ - RB_CLEAR_NODE(&vma->vm_rb); } static __always_inline void vma_rb_erase_ignore(struct vm_area_struct *vma, - struct mm_struct *mm, + struct rb_root *root, struct vm_area_struct *ignore) { /* @@ -511,21 +475,21 @@ static __always_inline void vma_rb_erase_ignore(struct vm_area_struct *vma, * with the possible exception of the "next" vma being erased if * next->vm_start was reduced. */ - validate_mm_rb(&mm->mm_rb, ignore); + validate_mm_rb(root, ignore); - __vma_rb_erase(vma, mm); + __vma_rb_erase(vma, root); } static __always_inline void vma_rb_erase(struct vm_area_struct *vma, - struct mm_struct *mm) + struct rb_root *root) { /* * All rb_subtree_gap values must be consistent prior to erase, * with the possible exception of the vma being erased. */ - validate_mm_rb(&mm->mm_rb, vma); + validate_mm_rb(root, vma); - __vma_rb_erase(vma, mm); + __vma_rb_erase(vma, root); } /* @@ -640,12 +604,10 @@ void __vma_link_rb(struct mm_struct *mm, struct vm_area_struct *vma, * immediately update the gap to the correct value. Finally we * rebalance the rbtree after all augmented values have been set. */ - mm_rb_write_lock(mm); rb_link_node(&vma->vm_rb, rb_parent, rb_link); vma->rb_subtree_gap = 0; vma_gap_update(vma); - vma_rb_insert(vma, mm); - mm_rb_write_unlock(mm); + vma_rb_insert(vma, &mm->mm_rb); } static void __vma_link_file(struct vm_area_struct *vma) @@ -721,7 +683,7 @@ static __always_inline void __vma_unlink_common(struct mm_struct *mm, { struct vm_area_struct *next; - vma_rb_erase_ignore(vma, mm, ignore); + vma_rb_erase_ignore(vma, &mm->mm_rb, ignore); next = vma->vm_next; if (has_prev) prev->vm_next = next; @@ -998,13 +960,16 @@ again: } if (remove_next) { - if (file) + if (file) { uprobe_munmap(next, next->vm_start, next->vm_end); + fput(file); + } if (next->anon_vma) anon_vma_merge(vma, next); mm->map_count--; + mpol_put(vma_policy(next)); vm_raw_write_end(next); - put_vma(next); + vm_area_free(next); /* * In mprotect's case 6 (see comments on vma_merge), * we must remove another next too. It would clutter @@ -2319,11 +2284,15 @@ get_unmapped_area(struct file *file, unsigned long addr, unsigned long len, EXPORT_SYMBOL(get_unmapped_area); /* Look up the first VMA which satisfies addr < vm_end, NULL if none. */ -static struct vm_area_struct *__find_vma(struct mm_struct *mm, - unsigned long addr) +struct vm_area_struct *find_vma(struct mm_struct *mm, unsigned long addr) { struct rb_node *rb_node; - struct vm_area_struct *vma = NULL; + struct vm_area_struct *vma; + + /* Check the cache first. */ + vma = vmacache_find(mm, addr); + if (likely(vma)) + return vma; rb_node = mm->mm_rb.rb_node; @@ -2341,40 +2310,13 @@ static struct vm_area_struct *__find_vma(struct mm_struct *mm, rb_node = rb_node->rb_right; } - return vma; -} - -struct vm_area_struct *find_vma(struct mm_struct *mm, unsigned long addr) -{ - struct vm_area_struct *vma; - - /* Check the cache first. */ - vma = vmacache_find(mm, addr); - if (likely(vma)) - return vma; - - vma = __find_vma(mm, addr); if (vma) vmacache_update(addr, vma); return vma; } + EXPORT_SYMBOL(find_vma); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -struct vm_area_struct *get_vma(struct mm_struct *mm, unsigned long addr) -{ - struct vm_area_struct *vma = NULL; - - read_lock(&mm->mm_rb_lock); - vma = __find_vma(mm, addr); - if (vma) - atomic_inc(&vma->vm_ref_count); - read_unlock(&mm->mm_rb_lock); - - return vma; -} -#endif - /* * Same as find_vma, but also return a pointer to the previous VMA in *pprev. */ @@ -2759,7 +2701,7 @@ detach_vmas_to_be_unmapped(struct mm_struct *mm, struct vm_area_struct *vma, insertion_point = (prev ? &prev->vm_next : &mm->mmap); vma->vm_prev = NULL; do { - vma_rb_erase(vma, mm); + vma_rb_erase(vma, &mm->mm_rb); mm->map_count--; tail_vma = vma; vma = vma->vm_next; From 7ec1f3792614c49288cc888cd83ebad828df4a15 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:40:10 +0900 Subject: [PATCH 31/48] Revert "mm: introduce __page_add_new_anon_rmap()" This reverts commit d56126780c1e35d3f022d0c46be099262589e618. Bug: 128240262 Change-Id: Ia5a417e52de006fba4f8b1b51d9ae4db36fd9035 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/rmap.h | 12 ++---------- mm/memory.c | 8 ++++---- mm/rmap.c | 5 +++-- 3 files changed, 9 insertions(+), 16 deletions(-) diff --git a/include/linux/rmap.h b/include/linux/rmap.h index 9821f6b6fd55..c5d593edb8fa 100644 --- a/include/linux/rmap.h +++ b/include/linux/rmap.h @@ -198,16 +198,8 @@ void page_add_anon_rmap(struct page *, struct vm_area_struct *, unsigned long, bool); void do_page_add_anon_rmap(struct page *, struct vm_area_struct *, unsigned long, int); -void __page_add_new_anon_rmap(struct page *page, struct vm_area_struct *vma, - unsigned long address, bool compound); -static inline void page_add_new_anon_rmap(struct page *page, - struct vm_area_struct *vma, - unsigned long address, bool compound) -{ - VM_BUG_ON_VMA(address < vma->vm_start || address >= vma->vm_end, vma); - __page_add_new_anon_rmap(page, vma, address, compound); -} - +void page_add_new_anon_rmap(struct page *, struct vm_area_struct *, + unsigned long, bool); void page_add_file_rmap(struct page *, bool); void page_remove_rmap(struct page *, bool); diff --git a/mm/memory.c b/mm/memory.c index 01890bca9578..06f79cd1d7e9 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2585,7 +2585,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) * thread doing COW. */ ptep_clear_flush_notify(vma, vmf->address, vmf->pte); - __page_add_new_anon_rmap(new_page, vma, vmf->address, false); + page_add_new_anon_rmap(new_page, vma, vmf->address, false); mem_cgroup_commit_charge(new_page, memcg, false, false); __lru_cache_add_active_or_unevictable(new_page, vmf->vma_flags); /* @@ -3145,7 +3145,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) /* ksm created a completely new copy */ if (unlikely(page != swapcache && swapcache)) { - __page_add_new_anon_rmap(page, vma, vmf->address, false); + page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); } else { @@ -3296,7 +3296,7 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) } inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); - __page_add_new_anon_rmap(page, vma, vmf->address, false); + page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); setpte: @@ -3575,7 +3575,7 @@ vm_fault_t alloc_set_pte(struct vm_fault *vmf, struct mem_cgroup *memcg, /* copy-on-write page */ if (write && !(vmf->vma_flags & VM_SHARED)) { inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); - __page_add_new_anon_rmap(page, vma, vmf->address, false); + page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); } else { diff --git a/mm/rmap.c b/mm/rmap.c index 2b5886600b38..ba7e3fe2ae9b 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -1150,7 +1150,7 @@ void do_page_add_anon_rmap(struct page *page, } /** - * __page_add_new_anon_rmap - add pte mapping to a new anonymous page + * page_add_new_anon_rmap - add pte mapping to a new anonymous page * @page: the page to add the mapping to * @vma: the vm area in which the mapping is added * @address: the user virtual address mapped @@ -1160,11 +1160,12 @@ void do_page_add_anon_rmap(struct page *page, * This means the inc-and-test can be bypassed. * Page does not have to be locked. */ -void __page_add_new_anon_rmap(struct page *page, +void page_add_new_anon_rmap(struct page *page, struct vm_area_struct *vma, unsigned long address, bool compound) { int nr = compound ? hpage_nr_pages(page) : 1; + VM_BUG_ON_VMA(address < vma->vm_start || address >= vma->vm_end, vma); __SetPageSwapBacked(page); if (compound) { VM_BUG_ON_PAGE(!PageTransHuge(page), page); From af3517b11de0d4f57d17a914911c7637c4c78669 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:39:59 +0900 Subject: [PATCH 32/48] Revert "mm: introduce __vm_normal_page()" This reverts commit ec5147dea1b8f02e576a06531d52098045513eda. Bug: 128240262 Change-Id: Iadc1c38e82431020b79194aff4bb25e7d9d66125 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 10 ++-------- mm/memory.c | 24 +++++++++--------------- 2 files changed, 11 insertions(+), 23 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 3c4c40f0ba39..bc7cc0e46144 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1567,14 +1567,8 @@ struct zap_details { struct page *single_page; /* Locked page to be unmapped */ }; -struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, - pte_t pte, unsigned long vma_flags); -static inline struct page *vm_normal_page(struct vm_area_struct *vma, - unsigned long addr, pte_t pte) -{ - return _vm_normal_page(vma, addr, pte, vma->vm_flags); -} - +struct page *vm_normal_page(struct vm_area_struct *vma, unsigned long addr, + pte_t pte); struct page *vm_normal_page_pmd(struct vm_area_struct *vma, unsigned long addr, pmd_t pmd); diff --git a/mm/memory.c b/mm/memory.c index 06f79cd1d7e9..682a98cd158b 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -566,8 +566,7 @@ static void print_bad_pte(struct vm_area_struct *vma, unsigned long addr, } /* - * __vm_normal_page -- This function gets the "struct page" associated with - * a pte. + * vm_normal_page -- This function gets the "struct page" associated with a pte. * * "Special" mappings do not wish to be associated with a "struct page" (either * it doesn't exist, or it exists but they don't want to touch it). In this @@ -608,8 +607,8 @@ static void print_bad_pte(struct vm_area_struct *vma, unsigned long addr, * PFNMAP mappings in order to support COWable mappings. * */ -struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, - pte_t pte, unsigned long vma_flags) +struct page *vm_normal_page(struct vm_area_struct *vma, unsigned long addr, + pte_t pte) { unsigned long pfn = pte_pfn(pte); @@ -618,7 +617,7 @@ struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, goto check_pfn; if (vma->vm_ops && vma->vm_ops->find_special_page) return vma->vm_ops->find_special_page(vma, addr); - if (vma_flags & (VM_PFNMAP | VM_MIXEDMAP)) + if (vma->vm_flags & (VM_PFNMAP | VM_MIXEDMAP)) return NULL; if (is_zero_pfn(pfn)) return NULL; @@ -630,13 +629,9 @@ struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, } /* !CONFIG_ARCH_HAS_PTE_SPECIAL case follows: */ - /* - * This part should never get called when CONFIG_SPECULATIVE_PAGE_FAULT - * is set. This is mainly because we can't rely on vm_start. - */ - if (unlikely(vma_flags & (VM_PFNMAP|VM_MIXEDMAP))) { - if (vma_flags & VM_MIXEDMAP) { + if (unlikely(vma->vm_flags & (VM_PFNMAP|VM_MIXEDMAP))) { + if (vma->vm_flags & VM_MIXEDMAP) { if (!pfn_valid(pfn)) return NULL; goto out; @@ -645,7 +640,7 @@ struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, off = (addr - vma->vm_start) >> PAGE_SHIFT; if (pfn == vma->vm_pgoff + off) return NULL; - if (!is_cow_mapping(vma_flags)) + if (!is_cow_mapping(vma->vm_flags)) return NULL; } } @@ -2781,8 +2776,7 @@ static vm_fault_t do_wp_page(struct vm_fault *vmf) mm_tlb_flush_pending(vmf->vma->vm_mm))) flush_tlb_page(vmf->vma, vmf->address); - vmf->page = _vm_normal_page(vma, vmf->address, vmf->orig_pte, - vmf->vma_flags); + vmf->page = vm_normal_page(vma, vmf->address, vmf->orig_pte); if (!vmf->page) { /* * VM_MIXEDMAP !pfn_valid() case, or VM_SOFTDIRTY clear on a @@ -3966,7 +3960,7 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) ptep_modify_prot_commit(vma, vmf->address, vmf->pte, old_pte, pte); update_mmu_cache(vma, vmf->address, vmf->pte); - page = _vm_normal_page(vma, vmf->address, pte, vmf->vma_flags); + page = vm_normal_page(vma, vmf->address, pte); if (!page) { pte_unmap_unlock(vmf->pte, vmf->ptl); return 0; From 9422831755f46c5cef5b57ff4af0d072d76136d5 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:40:19 +0900 Subject: [PATCH 33/48] Revert "mm: introduce __lru_cache_add_active_or_unevictable" This reverts commit 2536f11b7d9bb7a6974ab163b0c482bb689748e5. Bug: 128240262 Change-Id: I54aa09970c5967cfa7a93deca52f28773e27c041 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/swap.h | 10 ++-------- mm/memory.c | 8 ++++---- mm/swap.c | 6 +++--- 3 files changed, 9 insertions(+), 15 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index b407d3963649..136a929648ef 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -344,14 +344,8 @@ extern void deactivate_page(struct page *page); extern void mark_page_lazyfree(struct page *page); extern void swap_setup(void); -extern void __lru_cache_add_active_or_unevictable(struct page *page, - unsigned long vma_flags); - -static inline void lru_cache_add_active_or_unevictable(struct page *page, - struct vm_area_struct *vma) -{ - return __lru_cache_add_active_or_unevictable(page, vma->vm_flags); -} +extern void lru_cache_add_active_or_unevictable(struct page *page, + struct vm_area_struct *vma); /* linux/mm/vmscan.c */ extern unsigned long zone_reclaimable_pages(struct zone *zone); diff --git a/mm/memory.c b/mm/memory.c index 682a98cd158b..daf44ceb4409 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2582,7 +2582,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) ptep_clear_flush_notify(vma, vmf->address, vmf->pte); page_add_new_anon_rmap(new_page, vma, vmf->address, false); mem_cgroup_commit_charge(new_page, memcg, false, false); - __lru_cache_add_active_or_unevictable(new_page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(new_page, vma); /* * We call the notify macro here because, when using secondary * mmu page tables (such as kvm shadow page tables), we want the @@ -3141,7 +3141,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) if (unlikely(page != swapcache && swapcache)) { page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); - __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(page, vma); } else { do_page_add_anon_rmap(page, vma, vmf->address, exclusive); mem_cgroup_commit_charge(page, memcg, true, false); @@ -3292,7 +3292,7 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); - __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(page, vma); setpte: set_pte_at(vma->vm_mm, vmf->address, vmf->pte, entry); @@ -3571,7 +3571,7 @@ vm_fault_t alloc_set_pte(struct vm_fault *vmf, struct mem_cgroup *memcg, inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); - __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(page, vma); } else { inc_mm_counter_fast(vma->vm_mm, mm_counter_file(page)); page_add_file_rmap(page, false); diff --git a/mm/swap.c b/mm/swap.c index 159174565497..1c5021b57e2c 100644 --- a/mm/swap.c +++ b/mm/swap.c @@ -452,12 +452,12 @@ void lru_cache_add(struct page *page) * directly back onto it's zone's unevictable list, it does NOT use a * per cpu pagevec. */ -void __lru_cache_add_active_or_unevictable(struct page *page, - unsigned long vma_flags) +void lru_cache_add_active_or_unevictable(struct page *page, + struct vm_area_struct *vma) { VM_BUG_ON_PAGE(PageLRU(page), page); - if (likely((vma_flags & (VM_LOCKED | VM_SPECIAL)) != VM_LOCKED)) + if (likely((vma->vm_flags & (VM_LOCKED | VM_SPECIAL)) != VM_LOCKED)) SetPageActive(page); else if (!TestSetPageMlocked(page)) { /* From c1826f5b61e1590232452fec2b9435e4dfaefe6a Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:40:29 +0900 Subject: [PATCH 34/48] Revert "mm/migrate: Pass vm_fault pointer to migrate_misplaced_page()" This reverts commit b7efa36bf548cec5875e9caf56a24f59685deff4. Bug: 128240262 Change-Id: I21a63bb3e600beb422a8fc5240a4c3bed611fadc Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/migrate.h | 4 ++-- mm/memory.c | 2 +- mm/migrate.c | 4 ++-- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/include/linux/migrate.h b/include/linux/migrate.h index dff5b45f0e0e..400e635e4272 100644 --- a/include/linux/migrate.h +++ b/include/linux/migrate.h @@ -131,14 +131,14 @@ static inline void __ClearPageMovable(struct page *page) #ifdef CONFIG_NUMA_BALANCING extern bool pmd_trans_migrating(pmd_t pmd); extern int migrate_misplaced_page(struct page *page, - struct vm_fault *vmf, int node); + struct vm_area_struct *vma, int node); #else static inline bool pmd_trans_migrating(pmd_t pmd) { return false; } static inline int migrate_misplaced_page(struct page *page, - struct vm_fault *vmf, int node) + struct vm_area_struct *vma, int node) { return -EAGAIN; /* can't migrate now */ } diff --git a/mm/memory.c b/mm/memory.c index daf44ceb4409..b7e6cc034b08 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4001,7 +4001,7 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) } /* Migrate to the requested node */ - migrated = migrate_misplaced_page(page, vmf, target_nid); + migrated = migrate_misplaced_page(page, vma, target_nid); if (migrated) { page_nid = target_nid; flags |= TNF_MIGRATED; diff --git a/mm/migrate.c b/mm/migrate.c index 9a6bdbceec00..645599558446 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -1976,7 +1976,7 @@ bool pmd_trans_migrating(pmd_t pmd) * node. Caller is expected to have an elevated reference count on * the page that will be dropped by this function before returning. */ -int migrate_misplaced_page(struct page *page, struct vm_fault *vmf, +int migrate_misplaced_page(struct page *page, struct vm_area_struct *vma, int node) { pg_data_t *pgdat = NODE_DATA(node); @@ -1989,7 +1989,7 @@ int migrate_misplaced_page(struct page *page, struct vm_fault *vmf, * with execute permissions as they are probably shared libraries. */ if (page_mapcount(page) != 1 && page_is_file_cache(page) && - (vmf->vma_flags & VM_EXEC)) + (vma->vm_flags & VM_EXEC)) goto out; /* From 6d707879c3d36798fa4211d3b43e692754242349 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:41:29 +0900 Subject: [PATCH 35/48] Revert "mm: cache some VMA fields in the vm_fault structure" This reverts commit 5d1dccddd720cc20ebc8cb9c68fa4118181500f7. Bug: 128240262 Change-Id: Iea60ab5360ae83ca0efcbb47a101b3870691a93f Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 4 ++-- mm/huge_memory.c | 6 +++--- mm/khugepaged.c | 2 -- mm/memory.c | 53 ++++++++++++++++++++++------------------------ mm/migrate.c | 2 +- 5 files changed, 31 insertions(+), 36 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index bc7cc0e46144..07e4681afe9d 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -944,9 +944,9 @@ void free_compound_page(struct page *page); * pte_mkwrite. But get_user_pages can cause write faults for mappings * that do not have writing enabled, when used by access_process_vm. */ -static inline pte_t maybe_mkwrite(pte_t pte, unsigned long vma_flags) +static inline pte_t maybe_mkwrite(pte_t pte, struct vm_area_struct *vma) { - if (likely(vma_flags & VM_WRITE)) + if (likely(vma->vm_flags & VM_WRITE)) pte = pte_mkwrite(pte); return pte; } diff --git a/mm/huge_memory.c b/mm/huge_memory.c index e86ae9ed9685..3cfe90082085 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -1270,8 +1270,8 @@ static vm_fault_t do_huge_pmd_wp_page_fallback(struct vm_fault *vmf, for (i = 0; i < HPAGE_PMD_NR; i++, haddr += PAGE_SIZE) { pte_t entry; - entry = mk_pte(pages[i], vmf->vma_page_prot); - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = mk_pte(pages[i], vma->vm_page_prot); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); memcg = (void *)page_private(pages[i]); set_page_private(pages[i], 0); page_add_new_anon_rmap(pages[i], vmf->vma, haddr, false); @@ -2263,7 +2263,7 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd, entry = pte_swp_mksoft_dirty(entry); } else { entry = mk_pte(page + i, READ_ONCE(vma->vm_page_prot)); - entry = maybe_mkwrite(entry, vma->vm_flags); + entry = maybe_mkwrite(entry, vma); if (!write) entry = pte_wrprotect(entry); if (!young) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 092bc7ae184e..0f958b675586 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -918,8 +918,6 @@ static bool __collapse_huge_page_swapin(struct mm_struct *mm, .flags = FAULT_FLAG_ALLOW_RETRY, .pmd = pmd, .pgoff = linear_page_index(vma, address), - .vma_flags = vma->vm_flags, - .vma_page_prot = vma->vm_page_prot, }; /* we only decide to swapin, if there is enough young ptes */ diff --git a/mm/memory.c b/mm/memory.c index b7e6cc034b08..054da820d99a 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1667,8 +1667,7 @@ static vm_fault_t insert_pfn(struct vm_area_struct *vma, unsigned long addr, goto out_unlock; } entry = pte_mkyoung(*pte); - entry = maybe_mkwrite(pte_mkdirty(entry), - vma->vm_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); if (ptep_set_access_flags(vma, addr, pte, entry, 1)) update_mmu_cache(vma, addr, pte); } @@ -1683,7 +1682,7 @@ static vm_fault_t insert_pfn(struct vm_area_struct *vma, unsigned long addr, if (mkwrite) { entry = pte_mkyoung(entry); - entry = maybe_mkwrite(pte_mkdirty(entry), vma->vm_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); } set_pte_at(mm, addr, pte, entry); @@ -2481,7 +2480,7 @@ static inline void wp_page_reuse(struct vm_fault *vmf) flush_cache_page(vma, vmf->address, pte_pfn(vmf->orig_pte)); entry = pte_mkyoung(vmf->orig_pte); - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); if (ptep_set_access_flags(vma, vmf->address, vmf->pte, entry, 1)) update_mmu_cache(vma, vmf->address, vmf->pte); pte_unmap_unlock(vmf->pte, vmf->ptl); @@ -2571,8 +2570,8 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) inc_mm_counter_fast(mm, MM_ANONPAGES); } flush_cache_page(vma, vmf->address, pte_pfn(vmf->orig_pte)); - entry = mk_pte(new_page, vmf->vma_page_prot); - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = mk_pte(new_page, vma->vm_page_prot); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); /* * Clear the pte entry and flush it first, before updating the * pte with the new entry. This will avoid a race condition @@ -2637,7 +2636,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) * Don't let another task, with possibly unlocked vma, * keep the mlocked page. */ - if (page_copied && (vmf->vma_flags & VM_LOCKED)) { + if (page_copied && (vma->vm_flags & VM_LOCKED)) { lock_page(old_page); /* LRU manipulation */ if (PageMlocked(old_page)) munlock_vma_page(old_page); @@ -2674,7 +2673,7 @@ out: */ vm_fault_t finish_mkwrite_fault(struct vm_fault *vmf) { - WARN_ON_ONCE(!(vmf->vma_flags & VM_SHARED)); + WARN_ON_ONCE(!(vmf->vma->vm_flags & VM_SHARED)); if (!pte_map_lock(vmf)) return VM_FAULT_RETRY; /* @@ -2785,7 +2784,7 @@ static vm_fault_t do_wp_page(struct vm_fault *vmf) * We should not cow pages in a shared writeable mapping. * Just mark the pages writable and/or call ops->pfn_mkwrite. */ - if ((vmf->vma_flags & (VM_WRITE|VM_SHARED)) == + if ((vma->vm_flags & (VM_WRITE|VM_SHARED)) == (VM_WRITE|VM_SHARED)) return wp_pfn_shared(vmf); @@ -2817,7 +2816,7 @@ static vm_fault_t do_wp_page(struct vm_fault *vmf) unlock_page(page); wp_page_reuse(vmf); return VM_FAULT_WRITE; - } else if (unlikely((vmf->vma_flags & (VM_WRITE|VM_SHARED)) == + } else if (unlikely((vma->vm_flags & (VM_WRITE|VM_SHARED)) == (VM_WRITE|VM_SHARED))) { return wp_page_shared(vmf); } @@ -3123,9 +3122,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); dec_mm_counter_fast(vma->vm_mm, MM_SWAPENTS); - pte = mk_pte(page, vmf->vma_page_prot); + pte = mk_pte(page, vma->vm_page_prot); if ((vmf->flags & FAULT_FLAG_WRITE) && reuse_swap_page(page, NULL)) { - pte = maybe_mkwrite(pte_mkdirty(pte), vmf->vma_flags); + pte = maybe_mkwrite(pte_mkdirty(pte), vma); vmf->flags &= ~FAULT_FLAG_WRITE; ret |= VM_FAULT_WRITE; exclusive = RMAP_EXCLUSIVE; @@ -3150,7 +3149,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) swap_free(entry); if (mem_cgroup_swap_full(page) || - (vmf->vma_flags & VM_LOCKED) || PageMlocked(page)) + (vma->vm_flags & VM_LOCKED) || PageMlocked(page)) try_to_free_swap(page); unlock_page(page); if (page != swapcache && swapcache) { @@ -3208,7 +3207,7 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) pte_t entry; /* File mapping without ->vm_ops ? */ - if (vmf->vma_flags & VM_SHARED) + if (vma->vm_flags & VM_SHARED) return VM_FAULT_SIGBUS; /* @@ -3232,7 +3231,7 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) if (!(vmf->flags & FAULT_FLAG_WRITE) && !mm_forbids_zeropage(vma->vm_mm)) { entry = pte_mkspecial(pfn_pte(my_zero_pfn(vmf->address), - vmf->vma_page_prot)); + vma->vm_page_prot)); if (!pte_map_lock(vmf)) return VM_FAULT_RETRY; if (!pte_none(*vmf->pte)) @@ -3266,8 +3265,8 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) */ __SetPageUptodate(page); - entry = mk_pte(page, vmf->vma_page_prot); - if (vmf->vma_flags & VM_WRITE) + entry = mk_pte(page, vma->vm_page_prot); + if (vma->vm_flags & VM_WRITE) entry = pte_mkwrite(pte_mkdirty(entry)); if (!pte_map_lock(vmf)) { @@ -3483,7 +3482,7 @@ static vm_fault_t do_set_pmd(struct vm_fault *vmf, struct page *page) for (i = 0; i < HPAGE_PMD_NR; i++) flush_icache_page(vma, page + i); - entry = mk_huge_pmd(page, vmf->vma_page_prot); + entry = mk_huge_pmd(page, vma->vm_page_prot); if (write) entry = maybe_pmd_mkwrite(pmd_mkdirty(entry), vma); @@ -3559,15 +3558,15 @@ vm_fault_t alloc_set_pte(struct vm_fault *vmf, struct mem_cgroup *memcg, return VM_FAULT_NOPAGE; flush_icache_page(vma, page); - entry = mk_pte(page, vmf->vma_page_prot); + entry = mk_pte(page, vma->vm_page_prot); if (write) - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); if (vmf->flags & FAULT_FLAG_PREFAULT_OLD) entry = pte_mkold(entry); /* copy-on-write page */ - if (write && !(vmf->vma_flags & VM_SHARED)) { + if (write && !(vma->vm_flags & VM_SHARED)) { inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); @@ -3607,7 +3606,7 @@ vm_fault_t finish_fault(struct vm_fault *vmf) /* Did we COW the page? */ if ((vmf->flags & FAULT_FLAG_WRITE) && - !(vmf->vma_flags & VM_SHARED)) + !(vmf->vma->vm_flags & VM_SHARED)) page = vmf->cow_page; else page = vmf->page; @@ -3896,7 +3895,7 @@ static vm_fault_t do_fault(struct vm_fault *vmf) } } else if (!(vmf->flags & FAULT_FLAG_WRITE)) ret = do_read_fault(vmf); - else if (!(vmf->vma_flags & VM_SHARED)) + else if (!(vma->vm_flags & VM_SHARED)) ret = do_cow_fault(vmf); else ret = do_shared_fault(vmf); @@ -3953,7 +3952,7 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) * accessible ptes, some can allow access by kernel mode. */ old_pte = ptep_modify_prot_start(vma, vmf->address, vmf->pte); - pte = pte_modify(old_pte, vmf->vma_page_prot); + pte = pte_modify(old_pte, vma->vm_page_prot); pte = pte_mkyoung(pte); if (was_writable) pte = pte_mkwrite(pte); @@ -3987,7 +3986,7 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) * Flag if the page is shared between multiple address spaces. This * is later used when determining whether to group tasks together */ - if (page_mapcount(page) > 1 && (vmf->vma_flags & VM_SHARED)) + if (page_mapcount(page) > 1 && (vma->vm_flags & VM_SHARED)) flags |= TNF_SHARED; last_cpupid = page_cpupid_last(page); @@ -4032,7 +4031,7 @@ static inline vm_fault_t wp_huge_pmd(struct vm_fault *vmf, pmd_t orig_pmd) return vmf->vma->vm_ops->huge_fault(vmf, PE_SIZE_PMD); /* COW handled on pte level: split pmd */ - VM_BUG_ON_VMA(vmf->vma_flags & VM_SHARED, vmf->vma); + VM_BUG_ON_VMA(vmf->vma->vm_flags & VM_SHARED, vmf->vma); __split_huge_pmd(vmf->vma, vmf->pmd, vmf->address, false, NULL); return VM_FAULT_FALLBACK; @@ -4179,8 +4178,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, .flags = flags, .pgoff = linear_page_index(vma, address), .gfp_mask = __get_fault_gfp_mask(vma), - .vma_flags = vma->vm_flags, - .vma_page_prot = vma->vm_page_prot, }; unsigned int dirty = flags & FAULT_FLAG_WRITE; struct mm_struct *mm = vma->vm_mm; diff --git a/mm/migrate.c b/mm/migrate.c index 645599558446..1c58ab614c50 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -241,7 +241,7 @@ static bool remove_migration_pte(struct page *page, struct vm_area_struct *vma, */ entry = pte_to_swp_entry(*pvmw.pte); if (is_write_migration_entry(entry)) - pte = maybe_mkwrite(pte, vma->vm_flags); + pte = maybe_mkwrite(pte, vma); if (unlikely(is_zone_device_page(new))) { if (is_device_private_page(new)) { From a43fdd54d4cf806db1c058b4168d3d487c906f78 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:41:40 +0900 Subject: [PATCH 36/48] Revert "mm: protect SPF handler against anon_vma changes" This reverts commit ff6cfddf7a20ecb94397f8ebd925e06933945c10. Bug: 128240262 Change-Id: I317785d46d335aab974e194e5ea0d744513baecd Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- mm/memory.c | 4 ---- 1 file changed, 4 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 054da820d99a..e2f20c6bf83c 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -410,9 +410,7 @@ void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *vma, * Hide vma from rmap and truncate_pagecache before freeing * pgtables */ - vm_write_begin(vma); unlink_anon_vmas(vma); - vm_write_end(vma); unlink_file_vma(vma); if (is_vm_hugetlb_page(vma)) { @@ -426,9 +424,7 @@ void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *vma, && !is_vm_hugetlb_page(next)) { vma = next; next = vma->vm_next; - vm_write_begin(vma); unlink_anon_vmas(vma); - vm_write_end(vma); unlink_file_vma(vma); } free_pgd_range(tlb, addr, vma->vm_end, From 99a68bfb93fb4c00d9ba8bd04c3214ce5f18996a Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:41:50 +0900 Subject: [PATCH 37/48] Revert "mm: protect mremap() against SPF hanlder" This reverts commit 728a43dfff0cda8e95447709484b7d8f54a9a3ee. Bug: 128240262 Change-Id: Ida9fb7e41a7905755470e20f5c72867bb3dad03f Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 23 +++++--------------- mm/mmap.c | 53 +++++++++++----------------------------------- mm/mremap.c | 13 ------------ 3 files changed, 17 insertions(+), 72 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 07e4681afe9d..18ab714d4f00 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2451,29 +2451,16 @@ void anon_vma_interval_tree_verify(struct anon_vma_chain *node); extern int __vm_enough_memory(struct mm_struct *mm, long pages, int cap_sys_admin); extern int __vma_adjust(struct vm_area_struct *vma, unsigned long start, unsigned long end, pgoff_t pgoff, struct vm_area_struct *insert, - struct vm_area_struct *expand, bool keep_locked); + struct vm_area_struct *expand); static inline int vma_adjust(struct vm_area_struct *vma, unsigned long start, unsigned long end, pgoff_t pgoff, struct vm_area_struct *insert) { - return __vma_adjust(vma, start, end, pgoff, insert, NULL, false); + return __vma_adjust(vma, start, end, pgoff, insert, NULL); } - -extern struct vm_area_struct *__vma_merge(struct mm_struct *mm, +extern struct vm_area_struct *vma_merge(struct mm_struct *, struct vm_area_struct *prev, unsigned long addr, unsigned long end, - unsigned long vm_flags, struct anon_vma *anon, struct file *file, - pgoff_t pgoff, struct mempolicy *mpol, struct vm_userfaultfd_ctx uff, - const char __user *user, bool keep_locked); - -static inline struct vm_area_struct *vma_merge(struct mm_struct *mm, - struct vm_area_struct *prev, unsigned long addr, unsigned long end, - unsigned long vm_flags, struct anon_vma *anon, struct file *file, - pgoff_t off, struct mempolicy *pol, struct vm_userfaultfd_ctx uff, - const char __user *user) -{ - return __vma_merge(mm, prev, addr, end, vm_flags, anon, file, off, - pol, uff, user, false); -} - + unsigned long vm_flags, struct anon_vma *, struct file *, pgoff_t, + struct mempolicy *, struct vm_userfaultfd_ctx, const char __user *); extern struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *); extern int __split_vma(struct mm_struct *, struct vm_area_struct *, unsigned long addr, int new_below); diff --git a/mm/mmap.c b/mm/mmap.c index b4ba606a734c..bbe2ac280f0f 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -717,7 +717,7 @@ static inline void __vma_unlink_prev(struct mm_struct *mm, */ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, unsigned long end, pgoff_t pgoff, struct vm_area_struct *insert, - struct vm_area_struct *expand, bool keep_locked) + struct vm_area_struct *expand) { struct mm_struct *mm = vma->vm_mm; struct vm_area_struct *next = vma->vm_next, *orig_vma = vma; @@ -833,12 +833,8 @@ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, importer->anon_vma = exporter->anon_vma; error = anon_vma_clone(importer, exporter); - if (error) { - if (next && next != vma) - vm_raw_write_end(next); - vm_raw_write_end(vma); + if (error) return error; - } } } again: @@ -1033,8 +1029,7 @@ again: if (next && next != vma) vm_raw_write_end(next); - if (!keep_locked) - vm_raw_write_end(vma); + vm_raw_write_end(vma); validate_mm(mm); @@ -1177,13 +1172,13 @@ can_vma_merge_after(struct vm_area_struct *vma, unsigned long vm_flags, * parameter) may establish ptes with the wrong permissions of NNNN * instead of the right permissions of XXXX. */ -struct vm_area_struct *__vma_merge(struct mm_struct *mm, +struct vm_area_struct *vma_merge(struct mm_struct *mm, struct vm_area_struct *prev, unsigned long addr, unsigned long end, unsigned long vm_flags, struct anon_vma *anon_vma, struct file *file, pgoff_t pgoff, struct mempolicy *policy, struct vm_userfaultfd_ctx vm_userfaultfd_ctx, - const char __user *anon_name, bool keep_locked) + const char __user *anon_name) { pgoff_t pglen = (end - addr) >> PAGE_SHIFT; struct vm_area_struct *area, *next; @@ -1233,11 +1228,10 @@ struct vm_area_struct *__vma_merge(struct mm_struct *mm, /* cases 1, 6 */ err = __vma_adjust(prev, prev->vm_start, next->vm_end, prev->vm_pgoff, NULL, - prev, keep_locked); + prev); } else /* cases 2, 5, 7 */ err = __vma_adjust(prev, prev->vm_start, - end, prev->vm_pgoff, NULL, prev, - keep_locked); + end, prev->vm_pgoff, NULL, prev); if (err) return NULL; khugepaged_enter_vma_merge(prev, vm_flags); @@ -1255,12 +1249,10 @@ struct vm_area_struct *__vma_merge(struct mm_struct *mm, anon_name)) { if (prev && addr < prev->vm_end) /* case 4 */ err = __vma_adjust(prev, prev->vm_start, - addr, prev->vm_pgoff, NULL, next, - keep_locked); + addr, prev->vm_pgoff, NULL, next); else { /* cases 3, 8 */ err = __vma_adjust(area, addr, next->vm_end, - next->vm_pgoff - pglen, NULL, next, - keep_locked); + next->vm_pgoff - pglen, NULL, next); /* * In case 3 area is already equal to next and * this is a noop, but in case 8 "area" has @@ -3311,21 +3303,9 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, if (find_vma_links(mm, addr, addr + len, &prev, &rb_link, &rb_parent)) return NULL; /* should never get here */ - - /* There is 3 cases to manage here in - * AAAA AAAA AAAA AAAA - * PPPP.... PPPP......NNNN PPPP....NNNN PP........NN - * PPPPPPPP(A) PPPP..NNNNNNNN(B) PPPPPPPPPPPP(1) NULL - * PPPPPPPPNNNN(2) - * PPPPNNNNNNNN(3) - * - * new_vma == prev in case A,1,2 - * new_vma == next in case B,3 - */ - new_vma = __vma_merge(mm, prev, addr, addr + len, vma->vm_flags, - vma->anon_vma, vma->vm_file, pgoff, - vma_policy(vma), vma->vm_userfaultfd_ctx, - vma_get_anon_name(vma), true); + new_vma = vma_merge(mm, prev, addr, addr + len, vma->vm_flags, + vma->anon_vma, vma->vm_file, pgoff, vma_policy(vma), + vma->vm_userfaultfd_ctx, vma_get_anon_name(vma)); if (new_vma) { /* * Source vma may have been merged into new_vma @@ -3363,15 +3343,6 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, get_file(new_vma->vm_file); if (new_vma->vm_ops && new_vma->vm_ops->open) new_vma->vm_ops->open(new_vma); - /* - * As the VMA is linked right now, it may be hit by the - * speculative page fault handler. But we don't want it to - * to start mapping page in this area until the caller has - * potentially move the pte from the moved VMA. To prevent - * that we protect it right now, and let the caller unprotect - * it once the move is done. - */ - vm_raw_write_begin(new_vma); vma_link(mm, new_vma, prev, rb_link, rb_parent); *need_rmap_locks = false; } diff --git a/mm/mremap.c b/mm/mremap.c index 0485add3399f..8b1a5a6f7c50 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -527,14 +527,6 @@ static unsigned long move_vma(struct vm_area_struct *vma, return -ENOMEM; } - /* new_vma is returned protected by copy_vma, to prevent speculative - * page fault to be done in the destination area before we move the pte. - * Now, we must also protect the source VMA since we don't want pages - * to be mapped in our back while we are copying the PTEs. - */ - if (vma != new_vma) - vm_raw_write_begin(vma); - moved_len = move_page_tables(vma, old_addr, new_vma, new_addr, old_len, need_rmap_locks); if (moved_len < old_len) { @@ -551,8 +543,6 @@ static unsigned long move_vma(struct vm_area_struct *vma, */ move_page_tables(new_vma, new_addr, vma, old_addr, moved_len, true); - if (vma != new_vma) - vm_raw_write_end(vma); vma = new_vma; old_len = new_len; old_addr = new_addr; @@ -561,10 +551,7 @@ static unsigned long move_vma(struct vm_area_struct *vma, mremap_userfaultfd_prep(new_vma, uf); arch_remap(mm, old_addr, old_addr + old_len, new_addr, new_addr + new_len); - if (vma != new_vma) - vm_raw_write_end(vma); } - vm_raw_write_end(new_vma); /* Conceal VM_ACCOUNT so old reservation is not undone */ if (vm_flags & VM_ACCOUNT && !(flags & MREMAP_DONTUNMAP)) { From bb4e220b57a0aba7f2797ca02bb8768fd01f31bb Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:42:01 +0900 Subject: [PATCH 38/48] Revert "mm: protect VMA modifications using VMA sequence count" This reverts commit 06219e6c87ad8b309d0d8e010dbd06e0e74e889b. Bug: 128240262 Change-Id: If31a4c81badd891e6ca5740dbd022b5edbe47254 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- fs/proc/task_mmu.c | 5 +---- fs/userfaultfd.c | 17 ++++------------ mm/khugepaged.c | 3 --- mm/madvise.c | 10 +-------- mm/mempolicy.c | 51 ++++++++++++++++------------------------------ mm/mlock.c | 14 ++++++------- mm/mmap.c | 22 ++++++++------------ mm/mprotect.c | 4 +--- mm/swap_state.c | 10 +++------ 9 files changed, 42 insertions(+), 94 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 96062d525784..ef9a2d4d9200 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1296,11 +1296,8 @@ static ssize_t clear_refs_write(struct file *file, const char __user *buf, goto out_mm; } for (vma = mm->mmap; vma; vma = vma->vm_next) { - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, - vma->vm_flags & ~VM_SOFTDIRTY); + vma->vm_flags &= ~VM_SOFTDIRTY; vma_set_page_prot(vma); - vm_write_end(vma); } downgrade_write(&mm->mmap_sem); break; diff --git a/fs/userfaultfd.c b/fs/userfaultfd.c index 4a6719214dc4..de1203b48bb9 100644 --- a/fs/userfaultfd.c +++ b/fs/userfaultfd.c @@ -678,11 +678,8 @@ int dup_userfaultfd(struct vm_area_struct *vma, struct list_head *fcs) octx = vma->vm_userfaultfd_ctx.ctx; if (!octx || !(octx->features & UFFD_FEATURE_EVENT_FORK)) { - vm_write_begin(vma); vma->vm_userfaultfd_ctx = NULL_VM_UFFD_CTX; - WRITE_ONCE(vma->vm_flags, - vma->vm_flags & ~__VM_UFFD_FLAGS); - vm_write_end(vma); + vma->vm_flags &= ~__VM_UFFD_FLAGS; return 0; } @@ -924,10 +921,8 @@ static int userfaultfd_release(struct inode *inode, struct file *file) else prev = vma; } - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, new_flags); + vma->vm_flags = new_flags; vma->vm_userfaultfd_ctx = NULL_VM_UFFD_CTX; - vm_write_end(vma); } up_write(&mm->mmap_sem); mmput(mm); @@ -1499,10 +1494,8 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, * the next vma was merged into the current one and * the current one has not been updated yet. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, new_flags)); + vma->vm_flags = vma_pad_fixup_flags(vma, new_flags); vma->vm_userfaultfd_ctx.ctx = ctx; - vm_write_end(vma); if (is_vm_hugetlb_page(vma) && uffd_disable_huge_pmd_share(vma)) hugetlb_unshare_all_pmds(vma); @@ -1674,10 +1667,8 @@ static int userfaultfd_unregister(struct userfaultfd_ctx *ctx, * the next vma was merged into the current one and * the current one has not been updated yet. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, new_flags)); + vma->vm_flags = vma_pad_fixup_flags(vma, new_flags); vma->vm_userfaultfd_ctx = NULL_VM_UFFD_CTX; - vm_write_end(vma); skip: prev = vma; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 0f958b675586..f1f98305433e 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1041,7 +1041,6 @@ static void collapse_huge_page(struct mm_struct *mm, if (mm_find_pmd(mm, address) != pmd) goto out; - vm_write_begin(vma); anon_vma_lock_write(vma->anon_vma); mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, NULL, mm, @@ -1079,7 +1078,6 @@ static void collapse_huge_page(struct mm_struct *mm, pmd_populate(mm, pmd, pmd_pgtable(_pmd)); spin_unlock(pmd_ptl); anon_vma_unlock_write(vma->anon_vma); - vm_write_end(vma); result = SCAN_FAIL; goto out; } @@ -1115,7 +1113,6 @@ static void collapse_huge_page(struct mm_struct *mm, set_pmd_at(mm, address, pmd, _pmd); update_mmu_cache_pmd(vma, address, pmd); spin_unlock(pmd_ptl); - vm_write_end(vma); *hpage = NULL; diff --git a/mm/madvise.c b/mm/madvise.c index 023e11d4c8bf..c7b3b7e4ee7d 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -172,9 +172,7 @@ success: /* * vm_flags is protected by the mmap_sem held in write mode. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, new_flags)); - vm_write_end(vma); + vma->vm_flags = vma_pad_fixup_flags(vma, new_flags); out_convert_errno: /* @@ -501,11 +499,9 @@ static void madvise_cold_page_range(struct mmu_gather *tlb, .target_task = task, }; - vm_write_begin(vma); tlb_start_vma(tlb, vma); walk_page_range(vma->vm_mm, addr, end, &cold_walk_ops, &walk_private); tlb_end_vma(tlb, vma); - vm_write_end(vma); } static long madvise_cold(struct task_struct *task, @@ -539,11 +535,9 @@ static void madvise_pageout_page_range(struct mmu_gather *tlb, .target_task = task, }; - vm_write_begin(vma); tlb_start_vma(tlb, vma); walk_page_range(vma->vm_mm, addr, end, &cold_walk_ops, &walk_private); tlb_end_vma(tlb, vma); - vm_write_end(vma); } static inline bool can_do_pageout(struct vm_area_struct *vma) @@ -746,12 +740,10 @@ static int madvise_free_single_vma(struct vm_area_struct *vma, update_hiwater_rss(mm); mmu_notifier_invalidate_range_start(&range); - vm_write_begin(vma); tlb_start_vma(&tlb, vma); walk_page_range(vma->vm_mm, range.start, range.end, &madvise_free_walk_ops, &tlb); tlb_end_vma(&tlb, vma); - vm_write_end(vma); mmu_notifier_invalidate_range_end(&range); tlb_finish_mmu(&tlb, range.start, range.end); diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 284eb71f8e75..c3170460aa52 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -380,11 +380,8 @@ void mpol_rebind_mm(struct mm_struct *mm, nodemask_t *new) struct vm_area_struct *vma; down_write(&mm->mmap_sem); - for (vma = mm->mmap; vma; vma = vma->vm_next) { - vm_write_begin(vma); + for (vma = mm->mmap; vma; vma = vma->vm_next) mpol_rebind_policy(vma->vm_policy, new); - vm_write_end(vma); - } up_write(&mm->mmap_sem); } @@ -600,11 +597,9 @@ unsigned long change_prot_numa(struct vm_area_struct *vma, { int nr_updated; - vm_write_begin(vma); nr_updated = change_protection(vma, addr, end, PAGE_NONE, 0, 1); if (nr_updated) count_vm_numa_events(NUMA_PTE_UPDATES, nr_updated); - vm_write_end(vma); return nr_updated; } @@ -717,7 +712,6 @@ static int vma_replace_policy(struct vm_area_struct *vma, if (IS_ERR(new)) return PTR_ERR(new); - vm_write_begin(vma); if (vma->vm_ops && vma->vm_ops->set_policy) { err = vma->vm_ops->set_policy(vma, new); if (err) @@ -725,17 +719,11 @@ static int vma_replace_policy(struct vm_area_struct *vma, } old = vma->vm_policy; - /* - * The speculative page fault handler accesses this field without - * hodling the mmap_sem. - */ - WRITE_ONCE(vma->vm_policy, new); - vm_write_end(vma); + vma->vm_policy = new; /* protected by mmap_sem */ mpol_put(old); return 0; err_out: - vm_write_end(vma); mpol_put(new); return err; } @@ -1710,28 +1698,23 @@ COMPAT_SYSCALL_DEFINE4(migrate_pages, compat_pid_t, pid, struct mempolicy *__get_vma_policy(struct vm_area_struct *vma, unsigned long addr) { - struct mempolicy *pol; + struct mempolicy *pol = NULL; - if (!vma) - return NULL; + if (vma) { + if (vma->vm_ops && vma->vm_ops->get_policy) { + pol = vma->vm_ops->get_policy(vma, addr); + } else if (vma->vm_policy) { + pol = vma->vm_policy; - if (vma->vm_ops && vma->vm_ops->get_policy) - return vma->vm_ops->get_policy(vma, addr); - - /* - * This could be called without holding the mmap_sem in the - * speculative page fault handler's path. - */ - pol = READ_ONCE(vma->vm_policy); - if (pol) { - /* - * shmem_alloc_page() passes MPOL_F_SHARED policy with - * a pseudo vma whose vma->vm_ops=NULL. Take a reference - * count on these policies which will be dropped by - * mpol_cond_put() later - */ - if (mpol_needs_cond_ref(pol)) - mpol_get(pol); + /* + * shmem_alloc_page() passes MPOL_F_SHARED policy with + * a pseudo vma whose vma->vm_ops=NULL. Take a reference + * count on these policies which will be dropped by + * mpol_cond_put() later + */ + if (mpol_needs_cond_ref(pol)) + mpol_get(pol); + } } return pol; diff --git a/mm/mlock.c b/mm/mlock.c index 5805a284512d..ff9529310b3b 100644 --- a/mm/mlock.c +++ b/mm/mlock.c @@ -446,9 +446,7 @@ static unsigned long __munlock_pagevec_fill(struct pagevec *pvec, void munlock_vma_pages_range(struct vm_area_struct *vma, unsigned long start, unsigned long end) { - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma->vm_flags & VM_LOCKED_CLEAR_MASK); - vm_write_end(vma); + vma->vm_flags &= VM_LOCKED_CLEAR_MASK; while (start < end) { struct page *page; @@ -572,11 +570,11 @@ success: * It's okay if try_to_unmap_one unmaps a page just after we * set VM_LOCKED, populate_vma_page_range will bring it back. */ - if (lock) { - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, newflags)); - vm_write_end(vma); - } else + + if (lock) + vma->vm_flags = vma_pad_fixup_flags(vma, newflags); + + else munlock_vma_pages_range(vma, start, end); out: diff --git a/mm/mmap.c b/mm/mmap.c index bbe2ac280f0f..fae234a425c6 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -880,18 +880,17 @@ again: } if (start != vma->vm_start) { - WRITE_ONCE(vma->vm_start, start); + vma->vm_start = start; start_changed = true; } if (end != vma->vm_end) { - WRITE_ONCE(vma->vm_end, end); + vma->vm_end = end; end_changed = true; } - WRITE_ONCE(vma->vm_pgoff, pgoff); + vma->vm_pgoff = pgoff; if (adjust_next) { - WRITE_ONCE(next->vm_start, - next->vm_start + (adjust_next << PAGE_SHIFT)); - WRITE_ONCE(next->vm_pgoff, next->vm_pgoff + adjust_next); + next->vm_start += adjust_next << PAGE_SHIFT; + next->vm_pgoff += adjust_next; } if (root) { @@ -1876,14 +1875,12 @@ unsigned long mmap_region(struct file *file, unsigned long addr, out: perf_event_mmap(vma); - vm_write_begin(vma); vm_stat_account(mm, vm_flags, len >> PAGE_SHIFT); if (vm_flags & VM_LOCKED) { if ((vm_flags & VM_SPECIAL) || vma_is_dax(vma) || is_vm_hugetlb_page(vma) || vma == get_gate_vma(current->mm)) - WRITE_ONCE(vma->vm_flags, - vma->vm_flags & VM_LOCKED_CLEAR_MASK); + vma->vm_flags &= VM_LOCKED_CLEAR_MASK; else mm->locked_vm += (len >> PAGE_SHIFT); } @@ -1898,10 +1895,9 @@ out: * then new mapped in-place (which must be aimed as * a completely new data area). */ - WRITE_ONCE(vma->vm_flags, vma->vm_flags | VM_SOFTDIRTY); + vma->vm_flags |= VM_SOFTDIRTY; vma_set_page_prot(vma); - vm_write_end(vma); return addr; @@ -2529,8 +2525,8 @@ int expand_downwards(struct vm_area_struct *vma, mm->locked_vm += grow; vm_stat_account(mm, vma->vm_flags, grow); anon_vma_interval_tree_pre_update_vma(vma); - WRITE_ONCE(vma->vm_start, address); - WRITE_ONCE(vma->vm_pgoff, vma->vm_pgoff - grow); + vma->vm_start = address; + vma->vm_pgoff -= grow; anon_vma_interval_tree_post_update_vma(vma); vma_gap_update(vma); spin_unlock(&mm->page_table_lock); diff --git a/mm/mprotect.c b/mm/mprotect.c index c1e4a8da68b0..99cf8a97b2e4 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -455,14 +455,12 @@ success: * vm_flags and vm_page_prot are protected by the mmap_sem * held in write mode. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, newflags)); + vma->vm_flags = vma_pad_fixup_flags(vma, newflags); dirty_accountable = vma_wants_writenotify(vma, vma->vm_page_prot); vma_set_page_prot(vma); change_protection(vma, start, end, vma->vm_page_prot, dirty_accountable, 0); - vm_write_end(vma); /* * Private VM_LOCKED VMA becoming writable: trigger COW to avoid major diff --git a/mm/swap_state.c b/mm/swap_state.c index 260ac6a822b9..7c434fcfff0d 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -537,11 +537,7 @@ static unsigned long swapin_nr_pages(unsigned long offset) * This has been extended to use the NUMA policies from the mm triggering * the readahead. * - * Caller must hold down_read on the vma->vm_mm if vmf->vma is not NULL. - * This is needed to ensure the VMA will not be freed in our back. In the case - * of the speculative page fault handler, this cannot happen, even if we don't - * hold the mmap_sem. Callees are assumed to take care of reading VMA's fields - * using READ_ONCE() to read consistent values. + * Caller must hold read mmap_sem if vmf->vma is not NULL. */ struct page *swap_cluster_readahead(swp_entry_t entry, gfp_t gfp_mask, struct vm_fault *vmf) @@ -638,9 +634,9 @@ static inline void swap_ra_clamp_pfn(struct vm_area_struct *vma, unsigned long *start, unsigned long *end) { - *start = max3(lpfn, PFN_DOWN(READ_ONCE(vma->vm_start)), + *start = max3(lpfn, PFN_DOWN(vma->vm_start), PFN_DOWN(faddr & PMD_MASK)); - *end = min3(rpfn, PFN_DOWN(READ_ONCE(vma->vm_end)), + *end = min3(rpfn, PFN_DOWN(vma->vm_end), PFN_DOWN((faddr & PMD_MASK) + PMD_SIZE)); } From 6c8ee730de34fe8eec591d043dce592bc3721022 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:42:15 +0900 Subject: [PATCH 39/48] Revert "mm: VMA sequence count" This reverts commit 662540931e57cbfac485c1ba3c9363e15574be1b. Bug: 128240262 Change-Id: I7293939d08e6b9603d82464f4538a573b3407a88 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 44 ---------------------------------------- include/linux/mm_types.h | 4 +--- mm/memory.c | 2 -- mm/mmap.c | 31 ---------------------------- 4 files changed, 1 insertion(+), 80 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 18ab714d4f00..3821b83ddf02 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -621,9 +621,6 @@ struct vm_operations_struct { static inline void INIT_VMA(struct vm_area_struct *vma) { INIT_LIST_HEAD(&vma->anon_vma_chain); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - seqcount_init(&vma->vm_sequence); -#endif } static inline void vma_init(struct vm_area_struct *vma, struct mm_struct *mm) @@ -1597,47 +1594,6 @@ int follow_phys(struct vm_area_struct *vma, unsigned long address, int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, void *buf, int len, int write); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -static inline void vm_write_begin(struct vm_area_struct *vma) -{ - write_seqcount_begin(&vma->vm_sequence); -} -static inline void vm_write_begin_nested(struct vm_area_struct *vma, - int subclass) -{ - write_seqcount_begin_nested(&vma->vm_sequence, subclass); -} -static inline void vm_write_end(struct vm_area_struct *vma) -{ - write_seqcount_end(&vma->vm_sequence); -} -static inline void vm_raw_write_begin(struct vm_area_struct *vma) -{ - raw_write_seqcount_begin(&vma->vm_sequence); -} -static inline void vm_raw_write_end(struct vm_area_struct *vma) -{ - raw_write_seqcount_end(&vma->vm_sequence); -} -#else -static inline void vm_write_begin(struct vm_area_struct *vma) -{ -} -static inline void vm_write_begin_nested(struct vm_area_struct *vma, - int subclass) -{ -} -static inline void vm_write_end(struct vm_area_struct *vma) -{ -} -static inline void vm_raw_write_begin(struct vm_area_struct *vma) -{ -} -static inline void vm_raw_write_end(struct vm_area_struct *vma) -{ -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - extern void truncate_pagecache(struct inode *inode, loff_t new); extern void truncate_setsize(struct inode *inode, loff_t newsize); void pagecache_isize_extended(struct inode *inode, loff_t from, loff_t to); diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 894a443f2a47..41580bcd286a 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -362,9 +362,7 @@ struct vm_area_struct { struct mempolicy *vm_policy; /* NUMA policy for the VMA */ #endif struct vm_userfaultfd_ctx vm_userfaultfd_ctx; -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - seqcount_t vm_sequence; -#endif + ANDROID_KABI_RESERVE(1); ANDROID_KABI_RESERVE(2); ANDROID_KABI_RESERVE(3); diff --git a/mm/memory.c b/mm/memory.c index e2f20c6bf83c..658c0580a973 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1284,7 +1284,6 @@ void unmap_page_range(struct mmu_gather *tlb, unsigned long next; BUG_ON(addr >= end); - vm_write_begin(vma); tlb_start_vma(tlb, vma); pgd = pgd_offset(vma->vm_mm, addr); do { @@ -1294,7 +1293,6 @@ void unmap_page_range(struct mmu_gather *tlb, next = zap_p4d_range(tlb, vma, pgd, addr, next, details); } while (pgd++, addr = next, addr != end); tlb_end_vma(tlb, vma); - vm_write_end(vma); } diff --git a/mm/mmap.c b/mm/mmap.c index fae234a425c6..bcc12a4f0662 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -729,30 +729,6 @@ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, long adjust_next = 0; int remove_next = 0; - /* - * Why using vm_raw_write*() functions here to avoid lockdep's warning ? - * - * Locked is complaining about a theoretical lock dependency, involving - * 3 locks: - * mapping->i_mmap_rwsem --> vma->vm_sequence --> fs_reclaim - * - * Here are the major path leading to this dependency : - * 1. __vma_adjust() mmap_sem -> vm_sequence -> i_mmap_rwsem - * 2. move_vmap() mmap_sem -> vm_sequence -> fs_reclaim - * 3. __alloc_pages_nodemask() fs_reclaim -> i_mmap_rwsem - * 4. unmap_mapping_range() i_mmap_rwsem -> vm_sequence - * - * So there is no way to solve this easily, especially because in - * unmap_mapping_range() the i_mmap_rwsem is grab while the impacted - * VMAs are not yet known. - * However, the way the vm_seq is used is guarantying that we will - * never block on it since we just check for its value and never wait - * for it to move, see vma_has_changed() and handle_speculative_fault(). - */ - vm_raw_write_begin(vma); - if (next) - vm_raw_write_begin(next); - if (next && !insert) { struct vm_area_struct *exporter = NULL, *importer = NULL; @@ -963,7 +939,6 @@ again: anon_vma_merge(vma, next); mm->map_count--; mpol_put(vma_policy(next)); - vm_raw_write_end(next); vm_area_free(next); /* * In mprotect's case 6 (see comments on vma_merge), @@ -978,8 +953,6 @@ again: * "vma->vm_next" gap must be updated. */ next = vma->vm_next; - if (next) - vm_raw_write_begin(next); } else { /* * For the scope of the comment "next" and @@ -1026,10 +999,6 @@ again: if (insert && file) uprobe_mmap(insert); - if (next && next != vma) - vm_raw_write_end(next); - vm_raw_write_end(vma); - validate_mm(mm); return 0; From 35bafda6f54c4c89ad27e299587a6eefd29dbf3f Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:42:25 +0900 Subject: [PATCH 40/48] Revert "mm: introduce INIT_VMA()" This reverts commit f9eb248e24b322e81da8fb834350bfdaff14b341. Bug: 128240262 Change-Id: Ia6ca7bfc525b9e238d49df4812e1b66307674198 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 7 +------ kernel/fork.c | 2 +- 2 files changed, 2 insertions(+), 7 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 3821b83ddf02..3290675ac1d6 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -618,11 +618,6 @@ struct vm_operations_struct { ANDROID_KABI_RESERVE(4); }; -static inline void INIT_VMA(struct vm_area_struct *vma) -{ - INIT_LIST_HEAD(&vma->anon_vma_chain); -} - static inline void vma_init(struct vm_area_struct *vma, struct mm_struct *mm) { static const struct vm_operations_struct dummy_vm_ops = {}; @@ -630,7 +625,7 @@ static inline void vma_init(struct vm_area_struct *vma, struct mm_struct *mm) memset(vma, 0, sizeof(*vma)); vma->vm_mm = mm; vma->vm_ops = &dummy_vm_ops; - INIT_VMA(vma); + INIT_LIST_HEAD(&vma->anon_vma_chain); } static inline void vma_set_anonymous(struct vm_area_struct *vma) diff --git a/kernel/fork.c b/kernel/fork.c index fa92b2ab4081..b7a11a7a51e4 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -362,7 +362,7 @@ struct vm_area_struct *vm_area_dup(struct vm_area_struct *orig) if (new) { *new = *orig; - INIT_VMA(new); + INIT_LIST_HEAD(&new->anon_vma_chain); } return new; } From cf71c34ecb126e507f425fe3ae06e437c20ca43a Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:42:36 +0900 Subject: [PATCH 41/48] Revert "mm: make pte_unmap_same compatible with SPF" This reverts commit 7224f45453000022bb07762cbae3dbc46ff0b13c. Bug: 128240262 Change-Id: Ifdf06b5394bcfb111468fd7f220434a343fb48e8 Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- include/linux/mm_types.h | 1 - mm/memory.c | 39 +++++++++++---------------------------- 2 files changed, 11 insertions(+), 29 deletions(-) diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 41580bcd286a..e1f7840db114 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -705,7 +705,6 @@ enum vm_fault_reason { VM_FAULT_FALLBACK = (__force vm_fault_t)0x000800, VM_FAULT_DONE_COW = (__force vm_fault_t)0x001000, VM_FAULT_NEEDDSYNC = (__force vm_fault_t)0x002000, - VM_FAULT_PTNOTSAME = (__force vm_fault_t)0x004000, VM_FAULT_HINDEX_MASK = (__force vm_fault_t)0x0f0000, }; diff --git a/mm/memory.c b/mm/memory.c index 658c0580a973..1eb1ab086a06 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2229,29 +2229,21 @@ static inline bool pte_map_lock(struct vm_fault *vmf) * parts, do_swap_page must check under lock before unmapping the pte and * proceeding (but do_wp_page is only called after already making such a check; * and do_anonymous_page can safely check later on). - * - * pte_unmap_same() returns: - * 0 if the PTE are the same - * VM_FAULT_PTNOTSAME if the PTE are different - * VM_FAULT_RETRY if the VMA has changed in our back during - * a speculative page fault handling. */ -static inline int pte_unmap_same(struct vm_fault *vmf) +static inline int pte_unmap_same(struct mm_struct *mm, pmd_t *pmd, + pte_t *page_table, pte_t orig_pte) { - int ret = 0; - + int same = 1; #if defined(CONFIG_SMP) || defined(CONFIG_PREEMPT) if (sizeof(pte_t) > sizeof(unsigned long)) { - if (pte_spinlock(vmf)) { - if (!pte_same(*vmf->pte, vmf->orig_pte)) - ret = VM_FAULT_PTNOTSAME; - spin_unlock(vmf->ptl); - } else - ret = VM_FAULT_RETRY; + spinlock_t *ptl = pte_lockptr(mm, pmd); + spin_lock(ptl); + same = pte_same(*page_table, orig_pte); + spin_unlock(ptl); } #endif - pte_unmap(vmf->pte); - return ret; + pte_unmap(page_table); + return same; } static inline bool cow_user_page(struct page *dst, struct page *src, @@ -2967,19 +2959,10 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) pte_t pte; int locked; int exclusive = 0; - vm_fault_t ret; + vm_fault_t ret = 0; - ret = pte_unmap_same(vmf); - if (ret) { - /* - * If pte != orig_pte, this means another thread did the - * swap operation in our back. - * So nothing else to do. - */ - if (ret == VM_FAULT_PTNOTSAME) - ret = 0; + if (!pte_unmap_same(vma->vm_mm, vmf->pmd, vmf->pte, vmf->orig_pte)) goto out; - } entry = pte_to_swp_entry(vmf->orig_pte); if (unlikely(non_swap_entry(entry))) { From a1b67e1869163ff6c300d359cc1726416ea4efda Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:42:47 +0900 Subject: [PATCH 42/48] Revert "mm: introduce pte_spinlock for FAULT_FLAG_SPECULATIVE" This reverts commit 1d0d2c5e715af2dc30955778dda2dd2280631148. Bug: 128240262 Change-Id: Ie3d9a7f340ae93ef13475c41cbcb2b52f2ac720c Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- mm/memory.c | 15 ++++----------- 1 file changed, 4 insertions(+), 11 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 1eb1ab086a06..9a091ca4aa0f 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2208,13 +2208,6 @@ int apply_to_page_range(struct mm_struct *mm, unsigned long addr, } EXPORT_SYMBOL_GPL(apply_to_page_range); -static inline bool pte_spinlock(struct vm_fault *vmf) -{ - vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - spin_lock(vmf->ptl); - return true; -} - static inline bool pte_map_lock(struct vm_fault *vmf) { vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, @@ -3917,8 +3910,8 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) * validation through pte_unmap_same(). It's of NUMA type but * the pfn may be screwed if the read is non atomic. */ - if (!pte_spinlock(vmf)) - return VM_FAULT_RETRY; + vmf->ptl = pte_lockptr(vma->vm_mm, vmf->pmd); + spin_lock(vmf->ptl); if (unlikely(!pte_same(*vmf->pte, vmf->orig_pte))) { pte_unmap_unlock(vmf->pte, vmf->ptl); goto out; @@ -4111,8 +4104,8 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) if (pte_protnone(vmf->orig_pte) && vma_is_accessible(vmf->vma)) return do_numa_page(vmf); - if (!pte_spinlock(vmf)) - return VM_FAULT_RETRY; + vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); + spin_lock(vmf->ptl); entry = vmf->orig_pte; if (unlikely(!pte_same(*vmf->pte, entry))) goto unlock; From 98a98411f7276f2f87f5f42fd2048cdd62a95507 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:42:56 +0900 Subject: [PATCH 43/48] Revert "mm: prepare for FAULT_FLAG_SPECULATIVE" This reverts commit f447ae0acd3901bc5bec183c1b66f10b8bfeff91. Bug: 128240262 Change-Id: I362db97f606279ce66add0b6c3726438e80a597f Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- mm/memory.c | 81 ++++++++++++++++++----------------------------------- 1 file changed, 27 insertions(+), 54 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 9a091ca4aa0f..0b0c4489d9a1 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2208,13 +2208,6 @@ int apply_to_page_range(struct mm_struct *mm, unsigned long addr, } EXPORT_SYMBOL_GPL(apply_to_page_range); -static inline bool pte_map_lock(struct vm_fault *vmf) -{ - vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, - vmf->address, &vmf->ptl); - return true; -} - /* * handle_pte_fault chooses page fault handler according to an entry which was * read non-atomically. Before making any commitment, on those architectures @@ -2491,21 +2484,20 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) int page_copied = 0; struct mem_cgroup *memcg; struct mmu_notifier_range range; - int ret = VM_FAULT_OOM; if (unlikely(anon_vma_prepare(vma))) - goto out; + goto oom; if (is_zero_pfn(pte_pfn(vmf->orig_pte))) { new_page = alloc_zeroed_user_highpage_movable(vma, vmf->address); if (!new_page) - goto out; + goto oom; } else { new_page = alloc_page_vma(GFP_HIGHUSER_MOVABLE, vma, vmf->address); if (!new_page) - goto out; + goto oom; if (!cow_user_page(new_page, old_page, vmf)) { /* @@ -2522,7 +2514,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) } if (mem_cgroup_try_charge_delay(new_page, mm, GFP_KERNEL, &memcg, false)) - goto out_free_new; + goto oom_free_new; __SetPageUptodate(new_page); @@ -2534,10 +2526,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) /* * Re-check the pte - we dropped the lock */ - if (!pte_map_lock(vmf)) { - ret = VM_FAULT_RETRY; - goto out_uncharge; - } + vmf->pte = pte_offset_map_lock(mm, vmf->pmd, vmf->address, &vmf->ptl); if (likely(pte_same(*vmf->pte, vmf->orig_pte))) { if (old_page) { if (!PageAnon(old_page)) { @@ -2624,14 +2613,12 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) put_page(old_page); } return page_copied ? VM_FAULT_WRITE : 0; -out_uncharge: - mem_cgroup_cancel_charge(new_page, memcg, false); -out_free_new: +oom_free_new: put_page(new_page); -out: +oom: if (old_page) put_page(old_page); - return ret; + return VM_FAULT_OOM; } /** @@ -2653,8 +2640,8 @@ out: vm_fault_t finish_mkwrite_fault(struct vm_fault *vmf) { WARN_ON_ONCE(!(vmf->vma->vm_flags & VM_SHARED)); - if (!pte_map_lock(vmf)) - return VM_FAULT_RETRY; + vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); /* * We might have raced with another page fault while we released the * pte_offset_map_lock. @@ -3003,16 +2990,11 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) if (!page) { /* - * Back out if the VMA has changed in our back during - * a speculative page fault or if somebody else - * faulted in this pte while we released the pte lock. + * Back out if somebody else faulted in this pte + * while we released the pte lock. */ - if (!pte_map_lock(vmf)) { - delayacct_clear_flag(DELAYACCT_PF_SWAPIN); - ret = VM_FAULT_RETRY; - goto out; - } - + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, + vmf->address, &vmf->ptl); if (likely(pte_same(*vmf->pte, vmf->orig_pte))) ret = VM_FAULT_OOM; delayacct_clear_flag(DELAYACCT_PF_SWAPIN); @@ -3065,13 +3047,10 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) } /* - * Back out if the VMA has changed in our back during a speculative - * page fault or if somebody else already faulted in this pte. + * Back out if somebody else already faulted in this pte. */ - if (!pte_map_lock(vmf)) { - ret = VM_FAULT_RETRY; - goto out_cancel_cgroup; - } + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); if (unlikely(!pte_same(*vmf->pte, vmf->orig_pte))) goto out_nomap; @@ -3149,9 +3128,8 @@ unlock: out: return ret; out_nomap: - pte_unmap_unlock(vmf->pte, vmf->ptl); -out_cancel_cgroup: mem_cgroup_cancel_charge(page, memcg, false); + pte_unmap_unlock(vmf->pte, vmf->ptl); out_page: unlock_page(page); out_release: @@ -3202,8 +3180,8 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) !mm_forbids_zeropage(vma->vm_mm)) { entry = pte_mkspecial(pfn_pte(my_zero_pfn(vmf->address), vma->vm_page_prot)); - if (!pte_map_lock(vmf)) - return VM_FAULT_RETRY; + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, + vmf->address, &vmf->ptl); if (!pte_none(*vmf->pte)) goto unlock; ret = check_stable_address_space(vma->vm_mm); @@ -3239,16 +3217,14 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) if (vma->vm_flags & VM_WRITE) entry = pte_mkwrite(pte_mkdirty(entry)); - if (!pte_map_lock(vmf)) { - ret = VM_FAULT_RETRY; - goto release; - } + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); if (!pte_none(*vmf->pte)) - goto unlock_and_release; + goto release; ret = check_stable_address_space(vma->vm_mm); if (ret) - goto unlock_and_release; + goto release; /* Deliver the page fault to userland, check inside PT lock */ if (userfaultfd_missing(vma)) { @@ -3270,12 +3246,10 @@ setpte: unlock: pte_unmap_unlock(vmf->pte, vmf->ptl); return ret; -unlock_and_release: - pte_unmap_unlock(vmf->pte, vmf->ptl); release: mem_cgroup_cancel_charge(page, memcg, false); put_page(page); - return ret; + goto unlock; oom_free_page: put_page(page); oom: @@ -3399,9 +3373,8 @@ map_pte: * pte_none() under vmf->ptl protection when we return to * alloc_set_pte(). */ - if (!pte_map_lock(vmf)) - return VM_FAULT_RETRY; - + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); return 0; } From 8a8338f6aa79e90e12f6a19aef48c82e3fc55eeb Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Mon, 11 Mar 2019 11:56:57 +0900 Subject: [PATCH 44/48] Revert "mm: introduce CONFIG_SPECULATIVE_PAGE_FAULT" This reverts commit f783633cc8170788da1ce4e2dba6efa35e6147ca. Bug: 128240262 Change-Id: I2d5961521043932a767f828432f28e3418d7a77f Signed-off-by: Minchan Kim [dereference23: Forward port to msm-5.4] Signed-off-by: Alexander Winkowski --- mm/Kconfig | 23 ----------------------- 1 file changed, 23 deletions(-) diff --git a/mm/Kconfig b/mm/Kconfig index b1dc1e321d49..9f2aa4be3595 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -766,29 +766,6 @@ config HAVE_USERSPACE_LOW_MEMORY_KILLER when the OOM killer and userspace memory killer both have the potential to run). -config ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT - def_bool n - -config SPECULATIVE_PAGE_FAULT - bool "Speculative page faults" - default y - depends on ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT - depends on MMU && SMP - depends on QGKI - help - Try to handle user space page faults without holding the mmap_sem. - - This should allow better concurrency for massively threaded process - since the page fault handler will not wait for other threads memory - layout change to be done, assuming that this change is done in another - part of the process's memory space. This type of page fault is named - speculative page fault. - - If the speculative page fault fails because of a concurrency is - detected or because underlying PMD or PTE tables are not yet - allocating, it is failing its processing and a classic page fault - is then tried. - config GUP_BENCHMARK bool "Enable infrastructure for get_user_pages_fast() benchmarking" help From 8730ad1e9bf388a6dc071c477bc48f73cd08703c Mon Sep 17 00:00:00 2001 From: Alexander Winkowski Date: Fri, 12 Jan 2024 18:06:05 +0000 Subject: [PATCH 45/48] Revert "ANDROID: GKI: mm: add struct vm_fault fields for SPECULATIVE_PAGE_FAULTS" This reverts commit e89b2b7f5b06da2dbbd064ea17fbdb04a5796a32. Change-Id: Ia0b405e949e842ae90b69b9d2e169efd39237124 Signed-off-by: Alexander Winkowski --- include/linux/mm.h | 2 -- 1 file changed, 2 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 3290675ac1d6..5e852f2fbde1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -533,8 +533,6 @@ struct vm_fault { * page table to avoid allocation from * atomic context. */ - unsigned long vma_flags; /* Speculative Page Fault field */ - pgprot_t vma_page_prot; /* Speculative Page Fault field */ ANDROID_VENDOR_DATA(1); ANDROID_VENDOR_DATA(2); }; From f826e1aebd4d299a49c05f312a9c84ac201fe2d0 Mon Sep 17 00:00:00 2001 From: Sheenam Monga Date: Mon, 8 Aug 2022 12:55:41 +0530 Subject: [PATCH 46/48] qcacmn: Avoid incrementing usable channel count for 0 freq Currently if both filters are added to get usable channels i.e FILTER_CELLULAR_COEX and FILTER_WLAN_CONCURRENCY (3) then all channels for required band are added first and then response is updated based on cellular coex filter and invalid and avoided channel frequencies are removed based on provided mode but for FILTER_WLAN_CONCURRENCY pcl list is added, if frequency is not present in response channel list and count is incremented. As invalid frequencies are already removed from response before concurrency check , so some pcl frequencies will not be present in the list and count may be updated more than valid count. Fix is to add frequencies based on concurrency filter first and then remove invalid frequencies based on celluar coex filter to avoid any invalid increment of channel list count. Change-Id: I72b13c9c1f1bdfe3616d44fe893ce306634b022e CRs-Fixed: 3262059 --- .../umac/regulatory/core/src/reg_services_common.c | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c b/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c index 4678269ed294..7732e35b5ee4 100644 --- a/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c +++ b/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c @@ -3338,7 +3338,7 @@ reg_update_usable_chan_resp(struct wlan_objmgr_pdev *pdev, struct ch_params ch_params = {0}; int index = *count; - for (i = 0; i < len; i++) { + for (i = 0; i < len && index < NUM_CHANNELS; i++) { /* In case usable channels are required for multiple filter * mask, Some frequencies may present in res_msg . To avoid * frequency duplication, only mode mask is updated for @@ -3690,6 +3690,8 @@ reg_get_usable_channel_coex_filter(struct wlan_objmgr_pdev *pdev, chan_list[chan_enum].center_freq && freq_range.end_freq >= chan_list[chan_enum].center_freq) { + reg_debug("avoid freq %d", + chan_list[chan_enum].center_freq); reg_remove_freq(res_msg, chan_enum); } } @@ -3808,16 +3810,15 @@ wlan_reg_get_usable_channel(struct wlan_objmgr_pdev *pdev, } } - if (req_msg.filter_mask & 1 << FILTER_CELLULAR_COEX) - status = - reg_get_usable_channel_coex_filter(pdev, req_msg, res_msg, - chan_list, usable_channels); - if (req_msg.filter_mask & 1 << FILTER_WLAN_CONCURRENCY) status = reg_get_usable_channel_con_filter(pdev, req_msg, res_msg, usable_channels); + if (req_msg.filter_mask & 1 << FILTER_CELLULAR_COEX) + status = + reg_get_usable_channel_coex_filter(pdev, req_msg, res_msg, + chan_list, usable_channels); if (!(req_msg.filter_mask & 1 << FILTER_CELLULAR_COEX) && !(req_msg.filter_mask & 1 << FILTER_WLAN_CONCURRENCY)) status = From a12ebb70a58068fa72f2f75045d14552b4a2755f Mon Sep 17 00:00:00 2001 From: Aditya Kodukula Date: Wed, 8 Jan 2025 11:55:23 -0800 Subject: [PATCH 47/48] qcacld-3.0: Fix potential OOB memory access Currently in the wma_stats_ext_event_handler(), the buf_ptr is not pointing correctly to the event data received from FW. This is leading to an OOB memory access during qdf_mem_copy(). So, to avoid this issue correctly point the buf_ptr to the event data sent by the FW in the TLV. Change-Id: Iffa3e96a6a36eff5899a7a9a7febe0ebb9d7878f CRs-Fixed: 4011656 --- drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c b/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c index 3c8cbd8fb126..57d3d09ea1da 100644 --- a/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c +++ b/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c @@ -1,6 +1,6 @@ /* * Copyright (c) 2013-2021 The Linux Foundation. All rights reserved. - * Copyright (c) 2021-2024 Qualcomm Innovation Center, Inc. All rights reserved. + * Copyright (c) 2021-2025 Qualcomm Innovation Center, Inc. All rights reserved. * * Permission to use, copy, modify, and/or distribute this software for * any purpose with or without fee is hereby granted, provided that the @@ -708,7 +708,6 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, } stats_ext_info = param_buf->fixed_param; - buf_ptr = (uint8_t *)stats_ext_info; alloc_len = sizeof(tSirStatsExtEvent); alloc_len += stats_ext_info->data_len; @@ -725,7 +724,7 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, if (!stats_ext_event) return -ENOMEM; - buf_ptr += sizeof(wmi_stats_ext_event_fixed_param) + WMI_TLV_HDR_SIZE; + buf_ptr = (uint8_t *)param_buf->data; stats_ext_event->vdev_id = stats_ext_info->vdev_id; stats_ext_event->event_data_len = stats_ext_info->data_len; @@ -775,7 +774,6 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, } stats_ext_info = param_buf->fixed_param; - buf_ptr = (uint8_t *)stats_ext_info; alloc_len = sizeof(tSirStatsExtEvent); alloc_len += stats_ext_info->data_len; @@ -791,7 +789,7 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, if (!stats_ext_event) return -ENOMEM; - buf_ptr += sizeof(wmi_stats_ext_event_fixed_param) + WMI_TLV_HDR_SIZE; + buf_ptr = (uint8_t *)param_buf->data; stats_ext_event->vdev_id = stats_ext_info->vdev_id; stats_ext_event->event_data_len = stats_ext_info->data_len; From 6df72074a8cb8f2930417f3e6a7b06386fe8db5a Mon Sep 17 00:00:00 2001 From: Kai-Heng Feng Date: Wed, 13 Sep 2023 11:32:33 +0800 Subject: [PATCH 48/48] power: supply: core: Use blocking_notifier_call_chain to avoid RCU complaint AMD PMF driver can cause the following warning: [ 196.159546] ------------[ cut here ]------------ [ 196.159556] Voluntary context switch within RCU read-side critical section! [ 196.159571] WARNING: CPU: 0 PID: 9 at kernel/rcu/tree_plugin.h:320 rcu_note_context_switch+0x43d/0x560 [ 196.159604] Modules linked in: nvme_fabrics ccm rfcomm snd_hda_scodec_cs35l41_spi cmac algif_hash algif_skcipher af_alg bnep joydev btusb btrtl uvcvideo btintel btbcm videobuf2_vmalloc intel_rapl_msr btmtk videobuf2_memops uvc videobuf2_v4l2 intel_rapl_common binfmt_misc hid_sensor_als snd_sof_amd_vangogh hid_sensor_trigger bluetooth industrialio_triggered_buffer videodev snd_sof_amd_rembrandt hid_sensor_iio_common amdgpu ecdh_generic kfifo_buf videobuf2_common hp_wmi kvm_amd sparse_keymap snd_sof_amd_renoir wmi_bmof industrialio ecc mc nls_iso8859_1 kvm snd_sof_amd_acp irqbypass snd_sof_xtensa_dsp crct10dif_pclmul crc32_pclmul mt7921e snd_sof_pci snd_ctl_led polyval_clmulni mt7921_common polyval_generic snd_sof ghash_clmulni_intel mt792x_lib mt76_connac_lib sha512_ssse3 snd_sof_utils aesni_intel snd_hda_codec_realtek crypto_simd mt76 snd_hda_codec_generic cryptd snd_soc_core snd_hda_codec_hdmi rapl ledtrig_audio input_leds snd_compress i2c_algo_bit drm_ttm_helper mac80211 snd_pci_ps hid_multitouch ttm drm_exec [ 196.159970] drm_suballoc_helper snd_rpl_pci_acp6x amdxcp drm_buddy snd_hda_intel snd_acp_pci snd_hda_scodec_cs35l41_i2c serio_raw gpu_sched snd_hda_scodec_cs35l41 snd_acp_legacy_common snd_intel_dspcfg snd_hda_cs_dsp_ctls snd_hda_codec libarc4 drm_display_helper snd_pci_acp6x cs_dsp snd_hwdep snd_soc_cs35l41_lib video k10temp snd_pci_acp5x thunderbolt snd_hda_core drm_kms_helper cfg80211 snd_seq snd_rn_pci_acp3x snd_pcm snd_acp_config cec snd_soc_acpi snd_seq_device rc_core ccp snd_pci_acp3x snd_timer snd soundcore wmi amd_pmf platform_profile amd_pmc mac_hid serial_multi_instantiate wireless_hotkey hid_sensor_hub sch_fq_codel msr parport_pc ppdev lp parport efi_pstore ip_tables x_tables autofs4 btrfs blake2b_generic raid10 raid456 async_raid6_recov async_memcpy async_pq async_xor async_tx libcrc32c xor raid6_pq raid1 raid0 multipath linear dm_mirror dm_region_hash dm_log cdc_ether usbnet r8152 mii hid_generic nvme i2c_hid_acpi i2c_hid nvme_core i2c_piix4 xhci_pci amd_sfh drm xhci_pci_renesas nvme_common hid [ 196.160382] CPU: 0 PID: 9 Comm: kworker/0:1 Not tainted 6.6.0-rc1 #4 [ 196.160397] Hardware name: HP HP EliteBook 845 14 inch G10 Notebook PC/8B6E, BIOS V82 Ver. 01.02.00 08/24/2023 [ 196.160405] Workqueue: events power_supply_changed_work [ 196.160426] RIP: 0010:rcu_note_context_switch+0x43d/0x560 [ 196.160440] Code: 00 48 89 be 40 08 00 00 48 89 86 48 08 00 00 48 89 10 e9 63 fe ff ff 48 c7 c7 10 e7 b0 9e c6 05 e8 d8 20 02 01 e8 13 0f f3 ff <0f> 0b e9 27 fc ff ff a9 ff ff ff 7f 0f 84 cf fc ff ff 65 48 8b 3c [ 196.160450] RSP: 0018:ffffc900001878f0 EFLAGS: 00010046 [ 196.160462] RAX: 0000000000000000 RBX: ffff88885e834040 RCX: 0000000000000000 [ 196.160470] RDX: 0000000000000000 RSI: 0000000000000000 RDI: 0000000000000000 [ 196.160476] RBP: ffffc90000187910 R08: 0000000000000000 R09: 0000000000000000 [ 196.160482] R10: 0000000000000000 R11: 0000000000000000 R12: 0000000000000000 [ 196.160488] R13: 0000000000000000 R14: ffff888100990000 R15: ffff888100990000 [ 196.160495] FS: 0000000000000000(0000) GS:ffff88885e800000(0000) knlGS:0000000000000000 [ 196.160504] CS: 0010 DS: 0000 ES: 0000 CR0: 0000000080050033 [ 196.160512] CR2: 000055cb053c8246 CR3: 000000013443a000 CR4: 0000000000750ef0 [ 196.160520] PKRU: 55555554 [ 196.160526] Call Trace: [ 196.160532] [ 196.160548] ? show_regs+0x72/0x90 [ 196.160570] ? rcu_note_context_switch+0x43d/0x560 [ 196.160580] ? __warn+0x8d/0x160 [ 196.160600] ? rcu_note_context_switch+0x43d/0x560 [ 196.160613] ? report_bug+0x1bb/0x1d0 [ 196.160637] ? handle_bug+0x46/0x90 [ 196.160658] ? exc_invalid_op+0x19/0x80 [ 196.160675] ? asm_exc_invalid_op+0x1b/0x20 [ 196.160709] ? rcu_note_context_switch+0x43d/0x560 [ 196.160727] __schedule+0xb9/0x15f0 [ 196.160746] ? srso_alias_return_thunk+0x5/0x7f [ 196.160765] ? srso_alias_return_thunk+0x5/0x7f [ 196.160778] ? acpi_ns_search_one_scope+0xbe/0x270 [ 196.160806] schedule+0x68/0x110 [ 196.160820] schedule_timeout+0x151/0x160 [ 196.160829] ? srso_alias_return_thunk+0x5/0x7f [ 196.160842] ? srso_alias_return_thunk+0x5/0x7f [ 196.160855] ? acpi_ns_lookup+0x3c5/0xa90 [ 196.160878] __down_common+0xff/0x220 [ 196.160905] __down_timeout+0x16/0x30 [ 196.160920] down_timeout+0x64/0x70 [ 196.160938] acpi_os_wait_semaphore+0x85/0x200 [ 196.160959] acpi_ut_acquire_mutex+0x9e/0x280 [ 196.160979] acpi_ex_enter_interpreter+0x2d/0xb0 [ 196.160992] acpi_ns_evaluate+0x2f0/0x5f0 [ 196.161005] acpi_evaluate_object+0x172/0x490 [ 196.161018] ? acpi_os_signal_semaphore+0x8a/0xd0 [ 196.161038] acpi_evaluate_integer+0x52/0xe0 [ 196.161055] ? kfree+0x79/0x120 [ 196.161071] ? srso_alias_return_thunk+0x5/0x7f [ 196.161089] acpi_ac_get_state.part.0+0x27/0x80 [ 196.161110] get_ac_property+0x5c/0x70 [ 196.161127] ? __pfx___power_supply_is_system_supplied+0x10/0x10 [ 196.161146] __power_supply_is_system_supplied+0x44/0xb0 [ 196.161166] class_for_each_device+0x124/0x160 [ 196.161184] ? acpi_ac_get_state.part.0+0x27/0x80 [ 196.161203] ? srso_alias_return_thunk+0x5/0x7f [ 196.161223] power_supply_is_system_supplied+0x3c/0x70 [ 196.161243] amd_pmf_get_power_source+0xe/0x20 [amd_pmf] [ 196.161276] amd_pmf_power_slider_update_event+0x49/0x90 [amd_pmf] [ 196.161310] amd_pmf_pwr_src_notify_call+0xe7/0x100 [amd_pmf] [ 196.161340] notifier_call_chain+0x5f/0xe0 [ 196.161362] atomic_notifier_call_chain+0x33/0x60 [ 196.161378] power_supply_changed_work+0x84/0x110 [ 196.161394] process_one_work+0x178/0x360 [ 196.161412] ? __pfx_worker_thread+0x10/0x10 [ 196.161424] worker_thread+0x307/0x430 [ 196.161440] ? __pfx_worker_thread+0x10/0x10 [ 196.161451] kthread+0xf4/0x130 [ 196.161467] ? __pfx_kthread+0x10/0x10 [ 196.161486] ret_from_fork+0x43/0x70 [ 196.161502] ? __pfx_kthread+0x10/0x10 [ 196.161518] ret_from_fork_asm+0x1b/0x30 [ 196.161558] [ 196.161562] ---[ end trace 0000000000000000 ]--- Since there's no guarantee that all the callbacks can work in atomic context, switch to use blocking_notifier_call_chain to relax the constraint. Signed-off-by: Kai-Heng Feng Reported-by: Allen Zhong Fixes: 4c71ae414474 ("platform/x86/amd/pmf: Add support SPS PMF feature") Closes: https://bugzilla.kernel.org/show_bug.cgi?id=217571 Reviewed-by: Mario Limonciello Link: https://lore.kernel.org/r/20230913033233.602986-1-kai.heng.feng@canonical.com Signed-off-by: Sebastian Reichel Change-Id: I2311e9b1eb61bd88f4ed3a48104bb79a3a0800ce --- drivers/power/supply/power_supply_core.c | 8 ++++---- include/linux/power_supply.h | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/drivers/power/supply/power_supply_core.c b/drivers/power/supply/power_supply_core.c index 191b134ba0b2..e084f97314b3 100644 --- a/drivers/power/supply/power_supply_core.c +++ b/drivers/power/supply/power_supply_core.c @@ -27,7 +27,7 @@ struct class *power_supply_class; EXPORT_SYMBOL_GPL(power_supply_class); -ATOMIC_NOTIFIER_HEAD(power_supply_notifier); +BLOCKING_NOTIFIER_HEAD(power_supply_notifier); EXPORT_SYMBOL_GPL(power_supply_notifier); static struct device_type power_supply_dev_type; @@ -95,7 +95,7 @@ static void power_supply_changed_work(struct work_struct *work) class_for_each_device(power_supply_class, NULL, psy, __power_supply_changed_work); power_supply_update_leds(psy); - atomic_notifier_call_chain(&power_supply_notifier, + blocking_notifier_call_chain(&power_supply_notifier, PSY_EVENT_PROP_CHANGED, psy); kobject_uevent(&psy->dev.kobj, KOBJ_CHANGE); spin_lock_irqsave(&psy->changed_lock, flags); @@ -913,13 +913,13 @@ static void power_supply_dev_release(struct device *dev) int power_supply_reg_notifier(struct notifier_block *nb) { - return atomic_notifier_chain_register(&power_supply_notifier, nb); + return blocking_notifier_chain_register(&power_supply_notifier, nb); } EXPORT_SYMBOL_GPL(power_supply_reg_notifier); void power_supply_unreg_notifier(struct notifier_block *nb) { - atomic_notifier_chain_unregister(&power_supply_notifier, nb); + blocking_notifier_chain_unregister(&power_supply_notifier, nb); } EXPORT_SYMBOL_GPL(power_supply_unreg_notifier); diff --git a/include/linux/power_supply.h b/include/linux/power_supply.h index d6357e3a393e..d5cb1b803663 100644 --- a/include/linux/power_supply.h +++ b/include/linux/power_supply.h @@ -402,7 +402,7 @@ struct power_supply_battery_info { int resist_table_size; }; -extern struct atomic_notifier_head power_supply_notifier; +extern struct blocking_notifier_head power_supply_notifier; extern int power_supply_reg_notifier(struct notifier_block *nb); extern void power_supply_unreg_notifier(struct notifier_block *nb); extern struct power_supply *power_supply_get_by_name(const char *name);