diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index 72c71370dcec..090fff24a26b 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -191,7 +191,6 @@ config ARM64 select SYSCTL_EXCEPTION_TRACE select THREAD_INFO_IN_TASK select HAVE_ARCH_USERFAULTFD_MINOR if USERFAULTFD - select ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT help ARM 64-bit (AArch64) Linux support. diff --git a/arch/arm64/mm/fault.c b/arch/arm64/mm/fault.c index 78fe15dd857d..8fcd1dbe1fb7 100644 --- a/arch/arm64/mm/fault.c +++ b/arch/arm64/mm/fault.c @@ -410,9 +410,10 @@ static void do_bad_area(unsigned long addr, unsigned int esr, struct pt_regs *re #define VM_FAULT_BADMAP ((__force vm_fault_t)0x010000) #define VM_FAULT_BADACCESS ((__force vm_fault_t)0x020000) -static int __do_page_fault(struct vm_area_struct *vma, unsigned long addr, +static vm_fault_t __do_page_fault(struct mm_struct *mm, unsigned long addr, unsigned int mm_flags, unsigned long vm_flags) { + struct vm_area_struct *vma = find_vma(mm, addr); if (unlikely(!vma)) return VM_FAULT_BADMAP; @@ -459,7 +460,6 @@ static int __kprobes do_page_fault(unsigned long addr, unsigned int esr, vm_fault_t fault, major = 0; unsigned long vm_flags = VM_READ | VM_WRITE | VM_EXEC; unsigned int mm_flags = FAULT_FLAG_DEFAULT; - struct vm_area_struct *vma = NULL; if (kprobe_page_fault(regs, esr)) return 0; @@ -499,14 +499,6 @@ static int __kprobes do_page_fault(unsigned long addr, unsigned int esr, perf_sw_event(PERF_COUNT_SW_PAGE_FAULTS, 1, regs, addr); - /* - * let's try a speculative page fault without grabbing the - * mmap_sem. - */ - fault = handle_speculative_fault(mm, addr, mm_flags, &vma); - if (fault != VM_FAULT_RETRY) - goto done; - /* * As per x86, we may deadlock here. However, since the kernel only * validly references user space from well defined areas of the code, @@ -531,10 +523,7 @@ retry: #endif } - if (!vma || !can_reuse_spf_vma(vma, addr)) - vma = find_vma(mm, addr); - - fault = __do_page_fault(vma, addr, mm_flags, vm_flags); + fault = __do_page_fault(mm, addr, mm_flags, vm_flags); major |= fault & VM_FAULT_MAJOR; /* Quick path to respond to signals */ @@ -547,20 +536,11 @@ retry: if (fault & VM_FAULT_RETRY) { if (mm_flags & FAULT_FLAG_ALLOW_RETRY) { mm_flags |= FAULT_FLAG_TRIED; - - /* - * Do not try to reuse this vma and fetch it - * again since we will release the mmap_sem. - */ - vma = NULL; - goto retry; } } up_read(&mm->mmap_sem); -done: - /* * Handle the "normal" (no error) case first. */ diff --git a/drivers/power/supply/power_supply_core.c b/drivers/power/supply/power_supply_core.c index 191b134ba0b2..e084f97314b3 100644 --- a/drivers/power/supply/power_supply_core.c +++ b/drivers/power/supply/power_supply_core.c @@ -27,7 +27,7 @@ struct class *power_supply_class; EXPORT_SYMBOL_GPL(power_supply_class); -ATOMIC_NOTIFIER_HEAD(power_supply_notifier); +BLOCKING_NOTIFIER_HEAD(power_supply_notifier); EXPORT_SYMBOL_GPL(power_supply_notifier); static struct device_type power_supply_dev_type; @@ -95,7 +95,7 @@ static void power_supply_changed_work(struct work_struct *work) class_for_each_device(power_supply_class, NULL, psy, __power_supply_changed_work); power_supply_update_leds(psy); - atomic_notifier_call_chain(&power_supply_notifier, + blocking_notifier_call_chain(&power_supply_notifier, PSY_EVENT_PROP_CHANGED, psy); kobject_uevent(&psy->dev.kobj, KOBJ_CHANGE); spin_lock_irqsave(&psy->changed_lock, flags); @@ -913,13 +913,13 @@ static void power_supply_dev_release(struct device *dev) int power_supply_reg_notifier(struct notifier_block *nb) { - return atomic_notifier_chain_register(&power_supply_notifier, nb); + return blocking_notifier_chain_register(&power_supply_notifier, nb); } EXPORT_SYMBOL_GPL(power_supply_reg_notifier); void power_supply_unreg_notifier(struct notifier_block *nb) { - atomic_notifier_chain_unregister(&power_supply_notifier, nb); + blocking_notifier_chain_unregister(&power_supply_notifier, nb); } EXPORT_SYMBOL_GPL(power_supply_unreg_notifier); diff --git a/drivers/power/supply/qti_battery_charger.c b/drivers/power/supply/qti_battery_charger.c index 3f2993698fcc..3ba574ab29d9 100644 --- a/drivers/power/supply/qti_battery_charger.c +++ b/drivers/power/supply/qti_battery_charger.c @@ -1801,13 +1801,13 @@ static ssize_t charging_enabled_store(struct class *c, if (val) { /* * Enable charging, i.e. set the restricted current back to - * its default value and unset the restriction boolean flag. + * the thermal limit and unset the restriction boolean flag. */ rc = __battery_psy_set_charge_current(bcdev, - DEFAULT_RESTRICT_FCC_UA); + bcdev->thermal_fcc_ua); if (rc < 0) return rc; - bcdev->restrict_fcc_ua = DEFAULT_RESTRICT_FCC_UA; + bcdev->restrict_fcc_ua = bcdev->thermal_fcc_ua; bcdev->restrict_chg_en = 0; } else { /* diff --git a/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c b/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c index 4678269ed294..7732e35b5ee4 100644 --- a/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c +++ b/drivers/staging/qca-wifi-host-cmn/umac/regulatory/core/src/reg_services_common.c @@ -3338,7 +3338,7 @@ reg_update_usable_chan_resp(struct wlan_objmgr_pdev *pdev, struct ch_params ch_params = {0}; int index = *count; - for (i = 0; i < len; i++) { + for (i = 0; i < len && index < NUM_CHANNELS; i++) { /* In case usable channels are required for multiple filter * mask, Some frequencies may present in res_msg . To avoid * frequency duplication, only mode mask is updated for @@ -3690,6 +3690,8 @@ reg_get_usable_channel_coex_filter(struct wlan_objmgr_pdev *pdev, chan_list[chan_enum].center_freq && freq_range.end_freq >= chan_list[chan_enum].center_freq) { + reg_debug("avoid freq %d", + chan_list[chan_enum].center_freq); reg_remove_freq(res_msg, chan_enum); } } @@ -3808,16 +3810,15 @@ wlan_reg_get_usable_channel(struct wlan_objmgr_pdev *pdev, } } - if (req_msg.filter_mask & 1 << FILTER_CELLULAR_COEX) - status = - reg_get_usable_channel_coex_filter(pdev, req_msg, res_msg, - chan_list, usable_channels); - if (req_msg.filter_mask & 1 << FILTER_WLAN_CONCURRENCY) status = reg_get_usable_channel_con_filter(pdev, req_msg, res_msg, usable_channels); + if (req_msg.filter_mask & 1 << FILTER_CELLULAR_COEX) + status = + reg_get_usable_channel_coex_filter(pdev, req_msg, res_msg, + chan_list, usable_channels); if (!(req_msg.filter_mask & 1 << FILTER_CELLULAR_COEX) && !(req_msg.filter_mask & 1 << FILTER_WLAN_CONCURRENCY)) status = diff --git a/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c b/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c index 3c8cbd8fb126..57d3d09ea1da 100644 --- a/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c +++ b/drivers/staging/qcacld-3.0/core/wma/src/wma_utils.c @@ -1,6 +1,6 @@ /* * Copyright (c) 2013-2021 The Linux Foundation. All rights reserved. - * Copyright (c) 2021-2024 Qualcomm Innovation Center, Inc. All rights reserved. + * Copyright (c) 2021-2025 Qualcomm Innovation Center, Inc. All rights reserved. * * Permission to use, copy, modify, and/or distribute this software for * any purpose with or without fee is hereby granted, provided that the @@ -708,7 +708,6 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, } stats_ext_info = param_buf->fixed_param; - buf_ptr = (uint8_t *)stats_ext_info; alloc_len = sizeof(tSirStatsExtEvent); alloc_len += stats_ext_info->data_len; @@ -725,7 +724,7 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, if (!stats_ext_event) return -ENOMEM; - buf_ptr += sizeof(wmi_stats_ext_event_fixed_param) + WMI_TLV_HDR_SIZE; + buf_ptr = (uint8_t *)param_buf->data; stats_ext_event->vdev_id = stats_ext_info->vdev_id; stats_ext_event->event_data_len = stats_ext_info->data_len; @@ -775,7 +774,6 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, } stats_ext_info = param_buf->fixed_param; - buf_ptr = (uint8_t *)stats_ext_info; alloc_len = sizeof(tSirStatsExtEvent); alloc_len += stats_ext_info->data_len; @@ -791,7 +789,7 @@ int wma_stats_ext_event_handler(void *handle, uint8_t *event_buf, if (!stats_ext_event) return -ENOMEM; - buf_ptr += sizeof(wmi_stats_ext_event_fixed_param) + WMI_TLV_HDR_SIZE; + buf_ptr = (uint8_t *)param_buf->data; stats_ext_event->vdev_id = stats_ext_info->vdev_id; stats_ext_event->event_data_len = stats_ext_info->data_len; diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 96062d525784..ef9a2d4d9200 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1296,11 +1296,8 @@ static ssize_t clear_refs_write(struct file *file, const char __user *buf, goto out_mm; } for (vma = mm->mmap; vma; vma = vma->vm_next) { - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, - vma->vm_flags & ~VM_SOFTDIRTY); + vma->vm_flags &= ~VM_SOFTDIRTY; vma_set_page_prot(vma); - vm_write_end(vma); } downgrade_write(&mm->mmap_sem); break; diff --git a/fs/userfaultfd.c b/fs/userfaultfd.c index 4a6719214dc4..de1203b48bb9 100644 --- a/fs/userfaultfd.c +++ b/fs/userfaultfd.c @@ -678,11 +678,8 @@ int dup_userfaultfd(struct vm_area_struct *vma, struct list_head *fcs) octx = vma->vm_userfaultfd_ctx.ctx; if (!octx || !(octx->features & UFFD_FEATURE_EVENT_FORK)) { - vm_write_begin(vma); vma->vm_userfaultfd_ctx = NULL_VM_UFFD_CTX; - WRITE_ONCE(vma->vm_flags, - vma->vm_flags & ~__VM_UFFD_FLAGS); - vm_write_end(vma); + vma->vm_flags &= ~__VM_UFFD_FLAGS; return 0; } @@ -924,10 +921,8 @@ static int userfaultfd_release(struct inode *inode, struct file *file) else prev = vma; } - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, new_flags); + vma->vm_flags = new_flags; vma->vm_userfaultfd_ctx = NULL_VM_UFFD_CTX; - vm_write_end(vma); } up_write(&mm->mmap_sem); mmput(mm); @@ -1499,10 +1494,8 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, * the next vma was merged into the current one and * the current one has not been updated yet. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, new_flags)); + vma->vm_flags = vma_pad_fixup_flags(vma, new_flags); vma->vm_userfaultfd_ctx.ctx = ctx; - vm_write_end(vma); if (is_vm_hugetlb_page(vma) && uffd_disable_huge_pmd_share(vma)) hugetlb_unshare_all_pmds(vma); @@ -1674,10 +1667,8 @@ static int userfaultfd_unregister(struct userfaultfd_ctx *ctx, * the next vma was merged into the current one and * the current one has not been updated yet. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, new_flags)); + vma->vm_flags = vma_pad_fixup_flags(vma, new_flags); vma->vm_userfaultfd_ctx = NULL_VM_UFFD_CTX; - vm_write_end(vma); skip: prev = vma; diff --git a/include/linux/hugetlb_inline.h b/include/linux/hugetlb_inline.h index 9e25283d6fc9..0660a03d37d9 100644 --- a/include/linux/hugetlb_inline.h +++ b/include/linux/hugetlb_inline.h @@ -8,7 +8,7 @@ static inline bool is_vm_hugetlb_page(struct vm_area_struct *vma) { - return !!(READ_ONCE(vma->vm_flags) & VM_HUGETLB); + return !!(vma->vm_flags & VM_HUGETLB); } #else diff --git a/include/linux/migrate.h b/include/linux/migrate.h index dff5b45f0e0e..400e635e4272 100644 --- a/include/linux/migrate.h +++ b/include/linux/migrate.h @@ -131,14 +131,14 @@ static inline void __ClearPageMovable(struct page *page) #ifdef CONFIG_NUMA_BALANCING extern bool pmd_trans_migrating(pmd_t pmd); extern int migrate_misplaced_page(struct page *page, - struct vm_fault *vmf, int node); + struct vm_area_struct *vma, int node); #else static inline bool pmd_trans_migrating(pmd_t pmd) { return false; } static inline int migrate_misplaced_page(struct page *page, - struct vm_fault *vmf, int node) + struct vm_area_struct *vma, int node) { return -EAGAIN; /* can't migrate now */ } diff --git a/include/linux/mm.h b/include/linux/mm.h index 1a67be54dc47..5e852f2fbde1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -417,7 +417,6 @@ extern pgprot_t protection_map[16]; * @FAULT_FLAG_INSTRUCTION: The fault was during an instruction fetch. * @FAULT_FLAG_INTERRUPTIBLE: The fault can be interrupted by non-fatal signals. * @FAULT_FLAG_PREFAULT_OLD: Make faultaround ptes old. - * @FAULT_FLAG_SPECULATIVE: Speculative fault, not holding mmap_sem. * * About @FAULT_FLAG_ALLOW_RETRY and @FAULT_FLAG_TRIED: we can specify * whether we would allow page faults to retry by specifying these two @@ -449,7 +448,6 @@ extern pgprot_t protection_map[16]; #define FAULT_FLAG_INSTRUCTION 0x100 #define FAULT_FLAG_INTERRUPTIBLE 0x200 #define FAULT_FLAG_PREFAULT_OLD 0x400 -#define FAULT_FLAG_SPECULATIVE 0x800 /* * The default fault flags that should be used by most of the @@ -505,10 +503,6 @@ struct vm_fault { gfp_t gfp_mask; /* gfp mask to be used for allocations */ pgoff_t pgoff; /* Logical page offset based on vma */ unsigned long address; /* Faulting virtual address */ -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - unsigned int sequence; - pmd_t orig_pmd; /* value of PMD at the time of fault */ -#endif pmd_t *pmd; /* Pointer to pmd entry matching * the 'address' */ pud_t *pud; /* Pointer to pud entry matching @@ -539,8 +533,6 @@ struct vm_fault { * page table to avoid allocation from * atomic context. */ - unsigned long vma_flags; /* Speculative Page Fault field */ - pgprot_t vma_page_prot; /* Speculative Page Fault field */ ANDROID_VENDOR_DATA(1); ANDROID_VENDOR_DATA(2); }; @@ -624,15 +616,6 @@ struct vm_operations_struct { ANDROID_KABI_RESERVE(4); }; -static inline void INIT_VMA(struct vm_area_struct *vma) -{ - INIT_LIST_HEAD(&vma->anon_vma_chain); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - seqcount_init(&vma->vm_sequence); - atomic_set(&vma->vm_ref_count, 1); -#endif -} - static inline void vma_init(struct vm_area_struct *vma, struct mm_struct *mm) { static const struct vm_operations_struct dummy_vm_ops = {}; @@ -640,7 +623,7 @@ static inline void vma_init(struct vm_area_struct *vma, struct mm_struct *mm) memset(vma, 0, sizeof(*vma)); vma->vm_mm = mm; vma->vm_ops = &dummy_vm_ops; - INIT_VMA(vma); + INIT_LIST_HEAD(&vma->anon_vma_chain); } static inline void vma_set_anonymous(struct vm_area_struct *vma) @@ -951,9 +934,9 @@ void free_compound_page(struct page *page); * pte_mkwrite. But get_user_pages can cause write faults for mappings * that do not have writing enabled, when used by access_process_vm. */ -static inline pte_t maybe_mkwrite(pte_t pte, unsigned long vma_flags) +static inline pte_t maybe_mkwrite(pte_t pte, struct vm_area_struct *vma) { - if (likely(vma_flags & VM_WRITE)) + if (likely(vma->vm_flags & VM_WRITE)) pte = pte_mkwrite(pte); return pte; } @@ -1574,14 +1557,8 @@ struct zap_details { struct page *single_page; /* Locked page to be unmapped */ }; -struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, - pte_t pte, unsigned long vma_flags); -static inline struct page *vm_normal_page(struct vm_area_struct *vma, - unsigned long addr, pte_t pte) -{ - return _vm_normal_page(vma, addr, pte, vma->vm_flags); -} - +struct page *vm_normal_page(struct vm_area_struct *vma, unsigned long addr, + pte_t pte); struct page *vm_normal_page_pmd(struct vm_area_struct *vma, unsigned long addr, pmd_t pmd); @@ -1610,34 +1587,6 @@ int follow_phys(struct vm_area_struct *vma, unsigned long address, int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, void *buf, int len, int write); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -static inline void vm_write_begin(struct vm_area_struct *vma) -{ - /* - * Isolated vma might be freed without exclusive mmap_lock but - * speculative page fault handler still needs to know it was changed. - */ - if (!RB_EMPTY_NODE(&vma->vm_rb)) - WARN_ON_ONCE(!rwsem_is_locked(&(vma->vm_mm)->mmap_sem)); - /* - * The reads never spins and preemption - * disablement is not required. - */ - raw_write_seqcount_begin(&vma->vm_sequence); -} -static inline void vm_write_end(struct vm_area_struct *vma) -{ - raw_write_seqcount_end(&vma->vm_sequence); -} -#else -static inline void vm_write_begin(struct vm_area_struct *vma) -{ -} -static inline void vm_write_end(struct vm_area_struct *vma) -{ -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - extern void truncate_pagecache(struct inode *inode, loff_t new); extern void truncate_setsize(struct inode *inode, loff_t newsize); void pagecache_isize_extended(struct inode *inode, loff_t from, loff_t to); @@ -1649,43 +1598,6 @@ int invalidate_inode_page(struct page *page); #ifdef CONFIG_MMU extern vm_fault_t handle_mm_fault(struct vm_area_struct *vma, unsigned long address, unsigned int flags); - -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -extern int __handle_speculative_fault(struct mm_struct *mm, - unsigned long address, - unsigned int flags, - struct vm_area_struct **vma); -static inline int handle_speculative_fault(struct mm_struct *mm, - unsigned long address, - unsigned int flags, - struct vm_area_struct **vma) -{ - /* - * Try speculative page fault for multithreaded user space task only. - */ - if (!(flags & FAULT_FLAG_USER) || atomic_read(&mm->mm_users) == 1) { - *vma = NULL; - return VM_FAULT_RETRY; - } - return __handle_speculative_fault(mm, address, flags, vma); -} -extern bool can_reuse_spf_vma(struct vm_area_struct *vma, - unsigned long address); -#else -static inline int handle_speculative_fault(struct mm_struct *mm, - unsigned long address, - unsigned int flags, - struct vm_area_struct **vma) -{ - return VM_FAULT_RETRY; -} -static inline bool can_reuse_spf_vma(struct vm_area_struct *vma, - unsigned long address) -{ - return false; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - extern int fixup_user_fault(struct task_struct *tsk, struct mm_struct *mm, unsigned long address, unsigned int fault_flags, bool *unlocked); @@ -2488,29 +2400,16 @@ void anon_vma_interval_tree_verify(struct anon_vma_chain *node); extern int __vm_enough_memory(struct mm_struct *mm, long pages, int cap_sys_admin); extern int __vma_adjust(struct vm_area_struct *vma, unsigned long start, unsigned long end, pgoff_t pgoff, struct vm_area_struct *insert, - struct vm_area_struct *expand, bool keep_locked); + struct vm_area_struct *expand); static inline int vma_adjust(struct vm_area_struct *vma, unsigned long start, unsigned long end, pgoff_t pgoff, struct vm_area_struct *insert) { - return __vma_adjust(vma, start, end, pgoff, insert, NULL, false); + return __vma_adjust(vma, start, end, pgoff, insert, NULL); } - -extern struct vm_area_struct *__vma_merge(struct mm_struct *mm, +extern struct vm_area_struct *vma_merge(struct mm_struct *, struct vm_area_struct *prev, unsigned long addr, unsigned long end, - unsigned long vm_flags, struct anon_vma *anon, struct file *file, - pgoff_t pgoff, struct mempolicy *mpol, struct vm_userfaultfd_ctx uff, - const char __user *user, bool keep_locked); - -static inline struct vm_area_struct *vma_merge(struct mm_struct *mm, - struct vm_area_struct *prev, unsigned long addr, unsigned long end, - unsigned long vm_flags, struct anon_vma *anon, struct file *file, - pgoff_t off, struct mempolicy *pol, struct vm_userfaultfd_ctx uff, - const char __user *user) -{ - return __vma_merge(mm, prev, addr, end, vm_flags, anon, file, off, - pol, uff, user, false); -} - + unsigned long vm_flags, struct anon_vma *, struct file *, pgoff_t, + struct mempolicy *, struct vm_userfaultfd_ctx, const char __user *); extern struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *); extern int __split_vma(struct mm_struct *, struct vm_area_struct *, unsigned long addr, int new_below); diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index cc6185a0e05f..e1f7840db114 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -362,10 +362,7 @@ struct vm_area_struct { struct mempolicy *vm_policy; /* NUMA policy for the VMA */ #endif struct vm_userfaultfd_ctx vm_userfaultfd_ctx; -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - seqcount_t vm_sequence; - atomic_t vm_ref_count; /* see vma_get(), vma_put() */ -#endif + ANDROID_KABI_RESERVE(1); ANDROID_KABI_RESERVE(2); ANDROID_KABI_RESERVE(3); @@ -390,9 +387,6 @@ struct mm_struct { struct vm_area_struct *mmap; /* list of VMAs */ struct rb_root mm_rb; u64 vmacache_seqnum; /* per-thread vmacache */ -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - rwlock_t mm_rb_lock; -#endif #ifdef CONFIG_MMU unsigned long (*get_unmapped_area) (struct file *filp, unsigned long addr, unsigned long len, @@ -711,7 +705,6 @@ enum vm_fault_reason { VM_FAULT_FALLBACK = (__force vm_fault_t)0x000800, VM_FAULT_DONE_COW = (__force vm_fault_t)0x001000, VM_FAULT_NEEDDSYNC = (__force vm_fault_t)0x002000, - VM_FAULT_PTNOTSAME = (__force vm_fault_t)0x004000, VM_FAULT_HINDEX_MASK = (__force vm_fault_t)0x0f0000, }; diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 5ccb6baf050f..207432ae6e19 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -494,8 +494,8 @@ static inline pgoff_t linear_page_index(struct vm_area_struct *vma, pgoff_t pgoff; if (unlikely(is_vm_hugetlb_page(vma))) return linear_hugepage_index(vma, address); - pgoff = (address - READ_ONCE(vma->vm_start)) >> PAGE_SHIFT; - pgoff += READ_ONCE(vma->vm_pgoff); + pgoff = (address - vma->vm_start) >> PAGE_SHIFT; + pgoff += vma->vm_pgoff; return pgoff; } diff --git a/include/linux/power_supply.h b/include/linux/power_supply.h index d6357e3a393e..d5cb1b803663 100644 --- a/include/linux/power_supply.h +++ b/include/linux/power_supply.h @@ -402,7 +402,7 @@ struct power_supply_battery_info { int resist_table_size; }; -extern struct atomic_notifier_head power_supply_notifier; +extern struct blocking_notifier_head power_supply_notifier; extern int power_supply_reg_notifier(struct notifier_block *nb); extern void power_supply_unreg_notifier(struct notifier_block *nb); extern struct power_supply *power_supply_get_by_name(const char *name); diff --git a/include/linux/rmap.h b/include/linux/rmap.h index 9821f6b6fd55..c5d593edb8fa 100644 --- a/include/linux/rmap.h +++ b/include/linux/rmap.h @@ -198,16 +198,8 @@ void page_add_anon_rmap(struct page *, struct vm_area_struct *, unsigned long, bool); void do_page_add_anon_rmap(struct page *, struct vm_area_struct *, unsigned long, int); -void __page_add_new_anon_rmap(struct page *page, struct vm_area_struct *vma, - unsigned long address, bool compound); -static inline void page_add_new_anon_rmap(struct page *page, - struct vm_area_struct *vma, - unsigned long address, bool compound) -{ - VM_BUG_ON_VMA(address < vma->vm_start || address >= vma->vm_end, vma); - __page_add_new_anon_rmap(page, vma, address, compound); -} - +void page_add_new_anon_rmap(struct page *, struct vm_area_struct *, + unsigned long, bool); void page_add_file_rmap(struct page *, bool); void page_remove_rmap(struct page *, bool); diff --git a/include/linux/swap.h b/include/linux/swap.h index bb50a6d0eee4..d995f024877c 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -344,14 +344,8 @@ extern void deactivate_page(struct page *page); extern void mark_page_lazyfree(struct page *page); extern void swap_setup(void); -extern void __lru_cache_add_active_or_unevictable(struct page *page, - unsigned long vma_flags); - -static inline void lru_cache_add_active_or_unevictable(struct page *page, - struct vm_area_struct *vma) -{ - return __lru_cache_add_active_or_unevictable(page, vma->vm_flags); -} +extern void lru_cache_add_active_or_unevictable(struct page *page, + struct vm_area_struct *vma); /* linux/mm/vmscan.c */ extern unsigned long zone_reclaimable_pages(struct zone *zone); diff --git a/include/linux/vm_event_item.h b/include/linux/vm_event_item.h index 7f221e8a87cc..5a43dc19c317 100644 --- a/include/linux/vm_event_item.h +++ b/include/linux/vm_event_item.h @@ -113,10 +113,6 @@ enum vm_event_item { PGPGIN, PGPGOUT, #ifdef CONFIG_SWAP SWAP_RA, SWAP_RA_HIT, -#endif -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - SPECULATIVE_PGFAULT_ANON, - SPECULATIVE_PGFAULT_FILE, #endif NR_VM_EVENT_ITEMS }; diff --git a/include/trace/events/pagefault.h b/include/trace/events/pagefault.h deleted file mode 100644 index a9643b3759f2..000000000000 --- a/include/trace/events/pagefault.h +++ /dev/null @@ -1,88 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#undef TRACE_SYSTEM -#define TRACE_SYSTEM pagefault - -#if !defined(_TRACE_PAGEFAULT_H) || defined(TRACE_HEADER_MULTI_READ) -#define _TRACE_PAGEFAULT_H - -#include -#include - -DECLARE_EVENT_CLASS(spf, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address), - - TP_STRUCT__entry( - __field(unsigned long, caller) - __field(unsigned long, vm_start) - __field(unsigned long, vm_end) - __field(unsigned long, address) - ), - - TP_fast_assign( - __entry->caller = caller; - __entry->vm_start = vma->vm_start; - __entry->vm_end = vma->vm_end; - __entry->address = address; - ), - - TP_printk("ip:%lx vma:%lx-%lx address:%lx", - __entry->caller, __entry->vm_start, __entry->vm_end, - __entry->address) -); - -DEFINE_EVENT(spf, spf_pte_lock, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_changed, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_noanon, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_notsup, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_vma_access, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -DEFINE_EVENT(spf, spf_pmd_changed, - - TP_PROTO(unsigned long caller, - struct vm_area_struct *vma, unsigned long address), - - TP_ARGS(caller, vma, address) -); - -#endif /* _TRACE_PAGEFAULT_H */ - -/* This part must be outside protection */ -#include diff --git a/kernel/fork.c b/kernel/fork.c index 74f6f5e26ee4..b7a11a7a51e4 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -362,7 +362,7 @@ struct vm_area_struct *vm_area_dup(struct vm_area_struct *orig) if (new) { *new = *orig; - INIT_VMA(new); + INIT_LIST_HEAD(&new->anon_vma_chain); } return new; } @@ -486,7 +486,7 @@ EXPORT_SYMBOL(free_task); static __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm) { - struct vm_area_struct *mpnt, *tmp, *prev, **pprev, *last = NULL; + struct vm_area_struct *mpnt, *tmp, *prev, **pprev; struct rb_node **rb_link, *rb_parent; int retval; unsigned long charge; @@ -605,18 +605,8 @@ static __latent_entropy int dup_mmap(struct mm_struct *mm, rb_parent = &tmp->vm_rb; mm->map_count++; - if (!(tmp->vm_flags & VM_WIPEONFORK)) { - if (IS_ENABLED(CONFIG_SPECULATIVE_PAGE_FAULT)) { - /* - * Mark this VMA as changing to prevent the - * speculative page fault hanlder to process - * it until the TLB are flushed below. - */ - last = mpnt; - vm_write_begin(mpnt); - } + if (!(tmp->vm_flags & VM_WIPEONFORK)) retval = copy_page_range(mm, oldmm, mpnt); - } if (tmp->vm_ops && tmp->vm_ops->open) tmp->vm_ops->open(tmp); @@ -629,22 +619,6 @@ static __latent_entropy int dup_mmap(struct mm_struct *mm, out: up_write(&mm->mmap_sem); flush_tlb_mm(oldmm); - - if (IS_ENABLED(CONFIG_SPECULATIVE_PAGE_FAULT)) { - /* - * Since the TLB has been flush, we can safely unmark the - * copied VMAs and allows the speculative page fault handler to - * process them again. - * Walk back the VMA list from the last marked VMA. - */ - for (; last; last = last->vm_prev) { - if (last->vm_flags & VM_DONTCOPY) - continue; - if (!(last->vm_flags & VM_WIPEONFORK)) - vm_write_end(last); - } - } - up_write(&oldmm->mmap_sem); dup_userfaultfd_complete(&uf); fail_uprobe_end: @@ -1062,9 +1036,6 @@ static struct mm_struct *mm_init(struct mm_struct *mm, struct task_struct *p, mm->mmap = NULL; mm->mm_rb = RB_ROOT; mm->vmacache_seqnum = 0; -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - rwlock_init(&mm->mm_rb_lock); -#endif atomic_set(&mm->mm_users, 1); atomic_set(&mm->mm_count, 1); init_rwsem(&mm->mmap_sem); diff --git a/mm/Kconfig b/mm/Kconfig index cf70c44a4288..987534d9867c 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -780,29 +780,6 @@ config HAVE_USERSPACE_LOW_MEMORY_KILLER when the OOM killer and userspace memory killer both have the potential to run). -config ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT - def_bool n - -config SPECULATIVE_PAGE_FAULT - bool "Speculative page faults" - default y - depends on ARCH_SUPPORTS_SPECULATIVE_PAGE_FAULT - depends on MMU && SMP - depends on QGKI - help - Try to handle user space page faults without holding the mmap_sem. - - This should allow better concurrency for massively threaded process - since the page fault handler will not wait for other threads memory - layout change to be done, assuming that this change is done in another - part of the process's memory space. This type of page fault is named - speculative page fault. - - If the speculative page fault fails because of a concurrency is - detected or because underlying PMD or PTE tables are not yet - allocating, it is failing its processing and a classic page fault - is then tried. - config GUP_BENCHMARK bool "Enable infrastructure for get_user_pages_fast() benchmarking" help diff --git a/mm/filemap.c b/mm/filemap.c index 5cf65acf9a95..cbd98d4624ec 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -2567,12 +2567,12 @@ static struct file *do_sync_mmap_readahead(struct vm_fault *vmf) #endif /* If we don't want any read-ahead, don't bother */ - if (vmf->vma_flags & VM_RAND_READ) + if (vmf->vma->vm_flags & VM_RAND_READ) return fpin; if (!ra->ra_pages) return fpin; - if (vmf->vma_flags & VM_SEQ_READ) { + if (vmf->vma->vm_flags & VM_SEQ_READ) { fpin = maybe_unlock_mmap_for_io(vmf, fpin); page_cache_sync_readahead(mapping, ra, file, offset, ra->ra_pages); @@ -2624,7 +2624,7 @@ static struct file *do_async_mmap_readahead(struct vm_fault *vmf, pgoff_t offset = vmf->pgoff; /* If we don't want any read-ahead, don't bother */ - if (vmf->vma_flags & VM_RAND_READ || !ra->ra_pages) + if (vmf->vma->vm_flags & VM_RAND_READ || !ra->ra_pages) return fpin; if (ra->mmap_miss > 0) ra->mmap_miss--; @@ -2647,9 +2647,7 @@ static struct file *do_async_mmap_readahead(struct vm_fault *vmf, * it in the page cache, and handles the special cases reasonably without * having a lot of duplicated code. * - * If FAULT_FLAG_SPECULATIVE is set, this function runs with elevated vma - * refcount and with mmap lock not held. - * Otherwise, vma->vm_mm->mmap_sem must be held on entry. + * vma->vm_mm->mmap_sem must be held on entry. * * If our return value has VM_FAULT_RETRY set, it's because the mmap_sem * may be dropped before doing I/O or by lock_page_maybe_drop_mmap(). @@ -2674,52 +2672,6 @@ vm_fault_t filemap_fault(struct vm_fault *vmf) struct page *page; vm_fault_t ret = 0; - if (vmf->flags & FAULT_FLAG_SPECULATIVE) { - page = find_get_page(mapping, offset); - if (unlikely(!page)) - return VM_FAULT_RETRY; - - if (unlikely(PageReadahead(page))) - goto page_put; - - if (!trylock_page(page)) - goto page_put; - - if (unlikely(compound_head(page)->mapping != mapping)) - goto page_unlock; - VM_BUG_ON_PAGE(page_to_pgoff(page) != offset, page); - if (unlikely(!PageUptodate(page))) - goto page_unlock; - - max_off = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); - if (unlikely(offset >= max_off)) - goto page_unlock; - - /* - * Update readahead mmap_miss statistic. - * - * Note that we are not sure if finish_fault() will - * manage to complete the transaction. If it fails, - * we'll come back to filemap_fault() non-speculative - * case which will update mmap_miss a second time. - * This is not ideal, we would prefer to guarantee the - * update will happen exactly once. - */ - if (!(vmf->vma->vm_flags & VM_RAND_READ) && ra->ra_pages) { - unsigned int mmap_miss = READ_ONCE(ra->mmap_miss); - if (mmap_miss) - WRITE_ONCE(ra->mmap_miss, --mmap_miss); - } - - vmf->page = page; - return VM_FAULT_LOCKED; -page_unlock: - unlock_page(page); -page_put: - put_page(page); - return VM_FAULT_RETRY; - } - max_off = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); if (unlikely(offset >= max_off)) return VM_FAULT_SIGBUS; diff --git a/mm/huge_memory.c b/mm/huge_memory.c index e86ae9ed9685..3cfe90082085 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -1270,8 +1270,8 @@ static vm_fault_t do_huge_pmd_wp_page_fallback(struct vm_fault *vmf, for (i = 0; i < HPAGE_PMD_NR; i++, haddr += PAGE_SIZE) { pte_t entry; - entry = mk_pte(pages[i], vmf->vma_page_prot); - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = mk_pte(pages[i], vma->vm_page_prot); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); memcg = (void *)page_private(pages[i]); set_page_private(pages[i], 0); page_add_new_anon_rmap(pages[i], vmf->vma, haddr, false); @@ -2263,7 +2263,7 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd, entry = pte_swp_mksoft_dirty(entry); } else { entry = mk_pte(page + i, READ_ONCE(vma->vm_page_prot)); - entry = maybe_mkwrite(entry, vma->vm_flags); + entry = maybe_mkwrite(entry, vma); if (!write) entry = pte_wrprotect(entry); if (!young) diff --git a/mm/init-mm.c b/mm/init-mm.c index 0a4abdcf8886..19603302a77f 100644 --- a/mm/init-mm.c +++ b/mm/init-mm.c @@ -28,9 +28,6 @@ */ struct mm_struct init_mm = { .mm_rb = RB_ROOT, -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - .mm_rb_lock = __RW_LOCK_UNLOCKED(init_mm.mm_rb_lock), -#endif .pgd = swapper_pg_dir, .mm_users = ATOMIC_INIT(2), .mm_count = ATOMIC_INIT(1), diff --git a/mm/internal.h b/mm/internal.h index a3e960c45322..698713321327 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -36,26 +36,6 @@ void page_writeback_init(void); vm_fault_t do_swap_page(struct vm_fault *vmf); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -extern struct vm_area_struct *get_vma(struct mm_struct *mm, - unsigned long addr); -extern void put_vma(struct vm_area_struct *vma); - -static inline bool vma_has_changed(struct vm_fault *vmf) -{ - int ret = RB_EMPTY_NODE(&vmf->vma->vm_rb); - unsigned int seq = READ_ONCE(vmf->vma->vm_sequence.sequence); - - /* - * Matches both the wmb in write_seqlock_{begin,end}() and - * the wmb in vma_rb_erase(). - */ - smp_rmb(); - - return ret || seq != vmf->sequence; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *start_vma, unsigned long floor, unsigned long ceiling); diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 2bde07dd5258..f1f98305433e 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -918,8 +918,6 @@ static bool __collapse_huge_page_swapin(struct mm_struct *mm, .flags = FAULT_FLAG_ALLOW_RETRY, .pmd = pmd, .pgoff = linear_page_index(vma, address), - .vma_flags = vma->vm_flags, - .vma_page_prot = vma->vm_page_prot, }; /* we only decide to swapin, if there is enough young ptes */ @@ -1043,7 +1041,6 @@ static void collapse_huge_page(struct mm_struct *mm, if (mm_find_pmd(mm, address) != pmd) goto out; - vm_write_begin(vma); anon_vma_lock_write(vma->anon_vma); mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, NULL, mm, @@ -1081,7 +1078,6 @@ static void collapse_huge_page(struct mm_struct *mm, pmd_populate(mm, pmd, pmd_pgtable(_pmd)); spin_unlock(pmd_ptl); anon_vma_unlock_write(vma->anon_vma); - vm_write_end(vma); result = SCAN_FAIL; goto out; } @@ -1117,7 +1113,6 @@ static void collapse_huge_page(struct mm_struct *mm, set_pmd_at(mm, address, pmd, _pmd); update_mmu_cache_pmd(vma, address, pmd); spin_unlock(pmd_ptl); - vm_write_end(vma); *hpage = NULL; @@ -1345,8 +1340,6 @@ void collapse_pte_mapped_thp(struct mm_struct *mm, unsigned long addr) if (!pmd) goto drop_hpage; - vm_write_begin(vma); - /* * We need to lock the mapping so that from here on, only GUP-fast and * hardware page walks can access the parts of the page tables that @@ -1414,7 +1407,6 @@ void collapse_pte_mapped_thp(struct mm_struct *mm, unsigned long addr) haddr + HPAGE_PMD_SIZE); mmu_notifier_invalidate_range_start(&range); _pmd = pmdp_collapse_flush(vma, haddr, pmd); - vm_write_end(vma); mm_dec_nr_ptes(mm); tlb_remove_table_sync_one(); mmu_notifier_invalidate_range_end(&range); @@ -1431,7 +1423,6 @@ drop_hpage: abort: pte_unmap_unlock(start_pte, ptl); - vm_write_end(vma); i_mmap_unlock_write(vma->vm_file->f_mapping); goto drop_hpage; } @@ -1512,10 +1503,8 @@ static void retract_page_tables(struct address_space *mapping, pgoff_t pgoff) NULL, mm, addr, addr + HPAGE_PMD_SIZE); mmu_notifier_invalidate_range_start(&range); - vm_write_begin(vma); /* assume page table is clear */ _pmd = pmdp_collapse_flush(vma, addr, pmd); - vm_write_end(vma); mm_dec_nr_ptes(mm); tlb_remove_table_sync_one(); pte_free(mm, pmd_pgtable(_pmd)); diff --git a/mm/madvise.c b/mm/madvise.c index c1f85188efb2..62fb98954d71 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -172,9 +172,7 @@ success: /* * vm_flags is protected by the mmap_sem held in write mode. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, new_flags)); - vm_write_end(vma); + vma->vm_flags = vma_pad_fixup_flags(vma, new_flags); out_convert_errno: /* diff --git a/mm/memory.c b/mm/memory.c index 7c2b886d2537..41572f4ccf48 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -410,9 +410,7 @@ void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *vma, * Hide vma from rmap and truncate_pagecache before freeing * pgtables */ - vm_write_begin(vma); unlink_anon_vmas(vma); - vm_write_end(vma); unlink_file_vma(vma); if (is_vm_hugetlb_page(vma)) { @@ -426,9 +424,7 @@ void free_pgtables(struct mmu_gather *tlb, struct vm_area_struct *vma, && !is_vm_hugetlb_page(next)) { vma = next; next = vma->vm_next; - vm_write_begin(vma); unlink_anon_vmas(vma); - vm_write_end(vma); unlink_file_vma(vma); } free_pgd_range(tlb, addr, vma->vm_end, @@ -555,7 +551,7 @@ static void print_bad_pte(struct vm_area_struct *vma, unsigned long addr, if (page) dump_page(page, "bad pte"); pr_alert("addr:%px vm_flags:%08lx anon_vma:%px mapping:%px index:%lx\n", - (void *)addr, READ_ONCE(vma->vm_flags), vma->anon_vma, mapping, index); + (void *)addr, vma->vm_flags, vma->anon_vma, mapping, index); pr_alert("file:%pD fault:%ps mmap:%ps readpage:%ps\n", vma->vm_file, vma->vm_ops ? vma->vm_ops->fault : NULL, @@ -566,8 +562,7 @@ static void print_bad_pte(struct vm_area_struct *vma, unsigned long addr, } /* - * __vm_normal_page -- This function gets the "struct page" associated with - * a pte. + * vm_normal_page -- This function gets the "struct page" associated with a pte. * * "Special" mappings do not wish to be associated with a "struct page" (either * it doesn't exist, or it exists but they don't want to touch it). In this @@ -608,8 +603,8 @@ static void print_bad_pte(struct vm_area_struct *vma, unsigned long addr, * PFNMAP mappings in order to support COWable mappings. * */ -struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, - pte_t pte, unsigned long vma_flags) +struct page *vm_normal_page(struct vm_area_struct *vma, unsigned long addr, + pte_t pte) { unsigned long pfn = pte_pfn(pte); @@ -618,7 +613,7 @@ struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, goto check_pfn; if (vma->vm_ops && vma->vm_ops->find_special_page) return vma->vm_ops->find_special_page(vma, addr); - if (vma_flags & (VM_PFNMAP | VM_MIXEDMAP)) + if (vma->vm_flags & (VM_PFNMAP | VM_MIXEDMAP)) return NULL; if (is_zero_pfn(pfn)) return NULL; @@ -630,13 +625,9 @@ struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, } /* !CONFIG_ARCH_HAS_PTE_SPECIAL case follows: */ - /* - * This part should never get called when CONFIG_SPECULATIVE_PAGE_FAULT - * is set. This is mainly because we can't rely on vm_start. - */ - if (unlikely(vma_flags & (VM_PFNMAP|VM_MIXEDMAP))) { - if (vma_flags & VM_MIXEDMAP) { + if (unlikely(vma->vm_flags & (VM_PFNMAP|VM_MIXEDMAP))) { + if (vma->vm_flags & VM_MIXEDMAP) { if (!pfn_valid(pfn)) return NULL; goto out; @@ -645,7 +636,7 @@ struct page *_vm_normal_page(struct vm_area_struct *vma, unsigned long addr, off = (addr - vma->vm_start) >> PAGE_SHIFT; if (pfn == vma->vm_pgoff + off) return NULL; - if (!is_cow_mapping(vma_flags)) + if (!is_cow_mapping(vma->vm_flags)) return NULL; } } @@ -1670,8 +1661,7 @@ static vm_fault_t insert_pfn(struct vm_area_struct *vma, unsigned long addr, goto out_unlock; } entry = pte_mkyoung(*pte); - entry = maybe_mkwrite(pte_mkdirty(entry), - vma->vm_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); if (ptep_set_access_flags(vma, addr, pte, entry, 1)) update_mmu_cache(vma, addr, pte); } @@ -1686,7 +1676,7 @@ static vm_fault_t insert_pfn(struct vm_area_struct *vma, unsigned long addr, if (mkwrite) { entry = pte_mkyoung(entry); - entry = maybe_mkwrite(pte_mkdirty(entry), vma->vm_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); } set_pte_at(mm, addr, pte, entry); @@ -2218,128 +2208,6 @@ int apply_to_page_range(struct mm_struct *mm, unsigned long addr, } EXPORT_SYMBOL_GPL(apply_to_page_range); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -static bool pte_spinlock(struct vm_fault *vmf) -{ - bool ret = false; -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - pmd_t pmdval; -#endif - - /* Check if vma is still valid */ - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE)) { - vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - spin_lock(vmf->ptl); - return true; - } - - local_irq_disable(); - if (vma_has_changed(vmf)) - goto out; - -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - /* - * We check if the pmd value is still the same to ensure that there - * is not a huge collapse operation in progress in our back. - */ - pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) - goto out; -#endif - - vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - if (unlikely(!spin_trylock(vmf->ptl))) - goto out; - - if (vma_has_changed(vmf)) { - spin_unlock(vmf->ptl); - goto out; - } - - ret = true; -out: - local_irq_enable(); - return ret; -} - -static bool pte_map_lock(struct vm_fault *vmf) -{ - bool ret = false; - pte_t *pte; - spinlock_t *ptl; -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - pmd_t pmdval; -#endif - - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE)) { - vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, - vmf->address, &vmf->ptl); - return true; - } - - /* - * The first vma_has_changed() guarantees the page-tables are still - * valid, having IRQs disabled ensures they stay around, hence the - * second vma_has_changed() to make sure they are still valid once - * we've got the lock. After that a concurrent zap_pte_range() will - * block on the PTL and thus we're safe. - */ - local_irq_disable(); - if (vma_has_changed(vmf)) - goto out; - -#ifdef CONFIG_TRANSPARENT_HUGEPAGE - /* - * We check if the pmd value is still the same to ensure that there - * is not a huge collapse operation in progress in our back. - */ - pmdval = READ_ONCE(*vmf->pmd); - if (!pmd_same(pmdval, vmf->orig_pmd)) - goto out; -#endif - - /* - * Same as pte_offset_map_lock() except that we call - * spin_trylock() in place of spin_lock() to avoid race with - * unmap path which may have the lock and wait for this CPU - * to invalidate TLB but this CPU has irq disabled. - * Since we are in a speculative patch, accept it could fail - */ - ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - pte = pte_offset_map(vmf->pmd, vmf->address); - if (unlikely(!spin_trylock(ptl))) { - pte_unmap(pte); - goto out; - } - - if (vma_has_changed(vmf)) { - pte_unmap_unlock(pte, ptl); - goto out; - } - - vmf->pte = pte; - vmf->ptl = ptl; - ret = true; -out: - local_irq_enable(); - return ret; -} -#else -static inline bool pte_spinlock(struct vm_fault *vmf) -{ - vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); - spin_lock(vmf->ptl); - return true; -} - -static inline bool pte_map_lock(struct vm_fault *vmf) -{ - vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, - vmf->address, &vmf->ptl); - return true; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - /* * handle_pte_fault chooses page fault handler according to an entry which was * read non-atomically. Before making any commitment, on those architectures @@ -2347,29 +2215,21 @@ static inline bool pte_map_lock(struct vm_fault *vmf) * parts, do_swap_page must check under lock before unmapping the pte and * proceeding (but do_wp_page is only called after already making such a check; * and do_anonymous_page can safely check later on). - * - * pte_unmap_same() returns: - * 0 if the PTE are the same - * VM_FAULT_PTNOTSAME if the PTE are different - * VM_FAULT_RETRY if the VMA has changed in our back during - * a speculative page fault handling. */ -static inline int pte_unmap_same(struct vm_fault *vmf) +static inline int pte_unmap_same(struct mm_struct *mm, pmd_t *pmd, + pte_t *page_table, pte_t orig_pte) { - int ret = 0; - + int same = 1; #if defined(CONFIG_SMP) || defined(CONFIG_PREEMPT) if (sizeof(pte_t) > sizeof(unsigned long)) { - if (pte_spinlock(vmf)) { - if (!pte_same(*vmf->pte, vmf->orig_pte)) - ret = VM_FAULT_PTNOTSAME; - spin_unlock(vmf->ptl); - } else - ret = VM_FAULT_RETRY; + spinlock_t *ptl = pte_lockptr(mm, pmd); + spin_lock(ptl); + same = pte_same(*page_table, orig_pte); + spin_unlock(ptl); } #endif - pte_unmap(vmf->pte); - return ret; + pte_unmap(page_table); + return same; } static inline bool cow_user_page(struct page *dst, struct page *src, @@ -2592,7 +2452,7 @@ static inline void wp_page_reuse(struct vm_fault *vmf) flush_cache_page(vma, vmf->address, pte_pfn(vmf->orig_pte)); entry = pte_mkyoung(vmf->orig_pte); - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); if (ptep_set_access_flags(vma, vmf->address, vmf->pte, entry, 1)) update_mmu_cache(vma, vmf->address, vmf->pte); pte_unmap_unlock(vmf->pte, vmf->ptl); @@ -2624,21 +2484,20 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) int page_copied = 0; struct mem_cgroup *memcg; struct mmu_notifier_range range; - int ret = VM_FAULT_OOM; if (unlikely(anon_vma_prepare(vma))) - goto out; + goto oom; if (is_zero_pfn(pte_pfn(vmf->orig_pte))) { new_page = alloc_zeroed_user_highpage_movable(vma, vmf->address); if (!new_page) - goto out; + goto oom; } else { new_page = alloc_page_vma(GFP_HIGHUSER_MOVABLE, vma, vmf->address); if (!new_page) - goto out; + goto oom; if (!cow_user_page(new_page, old_page, vmf)) { /* @@ -2655,7 +2514,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) } if (mem_cgroup_try_charge_delay(new_page, mm, GFP_KERNEL, &memcg, false)) - goto out_free_new; + goto oom_free_new; __SetPageUptodate(new_page); @@ -2667,10 +2526,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) /* * Re-check the pte - we dropped the lock */ - if (!pte_map_lock(vmf)) { - ret = VM_FAULT_RETRY; - goto out_uncharge; - } + vmf->pte = pte_offset_map_lock(mm, vmf->pmd, vmf->address, &vmf->ptl); if (likely(pte_same(*vmf->pte, vmf->orig_pte))) { if (old_page) { if (!PageAnon(old_page)) { @@ -2682,8 +2538,8 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) inc_mm_counter_fast(mm, MM_ANONPAGES); } flush_cache_page(vma, vmf->address, pte_pfn(vmf->orig_pte)); - entry = mk_pte(new_page, vmf->vma_page_prot); - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = mk_pte(new_page, vma->vm_page_prot); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); /* * Clear the pte entry and flush it first, before updating the * pte with the new entry. This will avoid a race condition @@ -2691,9 +2547,9 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) * thread doing COW. */ ptep_clear_flush_notify(vma, vmf->address, vmf->pte); - __page_add_new_anon_rmap(new_page, vma, vmf->address, false); + page_add_new_anon_rmap(new_page, vma, vmf->address, false); mem_cgroup_commit_charge(new_page, memcg, false, false); - __lru_cache_add_active_or_unevictable(new_page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(new_page, vma); /* * We call the notify macro here because, when using secondary * mmu page tables (such as kvm shadow page tables), we want the @@ -2748,7 +2604,7 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) * Don't let another task, with possibly unlocked vma, * keep the mlocked page. */ - if (page_copied && (vmf->vma_flags & VM_LOCKED)) { + if (page_copied && (vma->vm_flags & VM_LOCKED)) { lock_page(old_page); /* LRU manipulation */ if (PageMlocked(old_page)) munlock_vma_page(old_page); @@ -2757,14 +2613,12 @@ static vm_fault_t wp_page_copy(struct vm_fault *vmf) put_page(old_page); } return page_copied ? VM_FAULT_WRITE : 0; -out_uncharge: - mem_cgroup_cancel_charge(new_page, memcg, false); -out_free_new: +oom_free_new: put_page(new_page); -out: +oom: if (old_page) put_page(old_page); - return ret; + return VM_FAULT_OOM; } /** @@ -2785,9 +2639,9 @@ out: */ vm_fault_t finish_mkwrite_fault(struct vm_fault *vmf) { - WARN_ON_ONCE(!(vmf->vma_flags & VM_SHARED)); - if (!pte_map_lock(vmf)) - return VM_FAULT_RETRY; + WARN_ON_ONCE(!(vmf->vma->vm_flags & VM_SHARED)); + vmf->pte = pte_offset_map_lock(vmf->vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); /* * We might have raced with another page fault while we released the * pte_offset_map_lock. @@ -2887,8 +2741,7 @@ static vm_fault_t do_wp_page(struct vm_fault *vmf) mm_tlb_flush_pending(vmf->vma->vm_mm))) flush_tlb_page(vmf->vma, vmf->address); - vmf->page = _vm_normal_page(vma, vmf->address, vmf->orig_pte, - vmf->vma_flags); + vmf->page = vm_normal_page(vma, vmf->address, vmf->orig_pte); if (!vmf->page) { /* * VM_MIXEDMAP !pfn_valid() case, or VM_SOFTDIRTY clear on a @@ -2897,7 +2750,7 @@ static vm_fault_t do_wp_page(struct vm_fault *vmf) * We should not cow pages in a shared writeable mapping. * Just mark the pages writable and/or call ops->pfn_mkwrite. */ - if ((vmf->vma_flags & (VM_WRITE|VM_SHARED)) == + if ((vma->vm_flags & (VM_WRITE|VM_SHARED)) == (VM_WRITE|VM_SHARED)) return wp_pfn_shared(vmf); @@ -2929,7 +2782,7 @@ static vm_fault_t do_wp_page(struct vm_fault *vmf) unlock_page(page); wp_page_reuse(vmf); return VM_FAULT_WRITE; - } else if (unlikely((vmf->vma_flags & (VM_WRITE|VM_SHARED)) == + } else if (unlikely((vma->vm_flags & (VM_WRITE|VM_SHARED)) == (VM_WRITE|VM_SHARED))) { return wp_page_shared(vmf); } @@ -3086,24 +2939,10 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) pte_t pte; int locked; int exclusive = 0; - vm_fault_t ret; + vm_fault_t ret = 0; - if (vmf->flags & FAULT_FLAG_SPECULATIVE) { - pte_unmap(vmf->pte); - return VM_FAULT_RETRY; - } - - ret = pte_unmap_same(vmf); - if (ret) { - /* - * If pte != orig_pte, this means another thread did the - * swap operation in our back. - * So nothing else to do. - */ - if (ret == VM_FAULT_PTNOTSAME) - ret = 0; + if (!pte_unmap_same(vma->vm_mm, vmf->pmd, vmf->pte, vmf->orig_pte)) goto out; - } entry = pte_to_swp_entry(vmf->orig_pte); if (unlikely(non_swap_entry(entry))) { @@ -3152,17 +2991,6 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) lru_cache_add_anon(page); swap_readpage(page, true); } - } else if (vmf->flags & FAULT_FLAG_SPECULATIVE) { - /* - * Don't try readahead during a speculative page fault - * as the VMA's boundaries may change in our back. - * If the page is not in the swap cache and synchronous - * read is disabled, fall back to the regular page fault - * mechanism. - */ - delayacct_clear_flag(DELAYACCT_PF_SWAPIN); - ret = VM_FAULT_RETRY; - goto out; } else { page = swapin_readahead(entry, GFP_HIGHUSER_MOVABLE | __GFP_CMA | __GFP_OFFLINABLE, vmf); @@ -3171,16 +2999,11 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) if (!page) { /* - * Back out if the VMA has changed in our back during - * a speculative page fault or if somebody else - * faulted in this pte while we released the pte lock. + * Back out if somebody else faulted in this pte + * while we released the pte lock. */ - if (!pte_map_lock(vmf)) { - delayacct_clear_flag(DELAYACCT_PF_SWAPIN); - ret = VM_FAULT_RETRY; - goto out; - } - + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, + vmf->address, &vmf->ptl); if (likely(pte_same(*vmf->pte, vmf->orig_pte))) ret = VM_FAULT_OOM; delayacct_clear_flag(DELAYACCT_PF_SWAPIN); @@ -3233,13 +3056,10 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) } /* - * Back out if the VMA has changed in our back during a speculative - * page fault or if somebody else already faulted in this pte. + * Back out if somebody else already faulted in this pte. */ - if (!pte_map_lock(vmf)) { - ret = VM_FAULT_RETRY; - goto out_cancel_cgroup; - } + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); if (unlikely(!pte_same(*vmf->pte, vmf->orig_pte))) goto out_nomap; @@ -3260,9 +3080,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); dec_mm_counter_fast(vma->vm_mm, MM_SWAPENTS); - pte = mk_pte(page, vmf->vma_page_prot); + pte = mk_pte(page, vma->vm_page_prot); if ((vmf->flags & FAULT_FLAG_WRITE) && reuse_swap_page(page, NULL)) { - pte = maybe_mkwrite(pte_mkdirty(pte), vmf->vma_flags); + pte = maybe_mkwrite(pte_mkdirty(pte), vma); vmf->flags &= ~FAULT_FLAG_WRITE; ret |= VM_FAULT_WRITE; exclusive = RMAP_EXCLUSIVE; @@ -3276,9 +3096,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) /* ksm created a completely new copy */ if (unlikely(page != swapcache && swapcache)) { - __page_add_new_anon_rmap(page, vma, vmf->address, false); + page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); - __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(page, vma); } else { do_page_add_anon_rmap(page, vma, vmf->address, exclusive); mem_cgroup_commit_charge(page, memcg, true, false); @@ -3287,7 +3107,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) swap_free(entry); if (mem_cgroup_swap_full(page) || - (vmf->vma_flags & VM_LOCKED) || PageMlocked(page)) + (vma->vm_flags & VM_LOCKED) || PageMlocked(page)) try_to_free_swap(page); unlock_page(page); if (page != swapcache && swapcache) { @@ -3317,9 +3137,8 @@ unlock: out: return ret; out_nomap: - pte_unmap_unlock(vmf->pte, vmf->ptl); -out_cancel_cgroup: mem_cgroup_cancel_charge(page, memcg, false); + pte_unmap_unlock(vmf->pte, vmf->ptl); out_page: unlock_page(page); out_release: @@ -3345,13 +3164,9 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) pte_t entry; /* File mapping without ->vm_ops ? */ - if (vmf->vma_flags & VM_SHARED) + if (vma->vm_flags & VM_SHARED) return VM_FAULT_SIGBUS; - /* Do not check unstable pmd, if it's changed will retry later */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto skip_pmd_checks; - /* * Use pte_alloc() instead of pte_alloc_map(). We can't run * pte_offset_map() on pmds where a huge pmd might be created @@ -3369,27 +3184,18 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf) if (unlikely(pmd_trans_unstable(vmf->pmd))) return 0; -skip_pmd_checks: /* Use the zero-page for reads */ if (!(vmf->flags & FAULT_FLAG_WRITE) && !mm_forbids_zeropage(vma->vm_mm)) { entry = pte_mkspecial(pfn_pte(my_zero_pfn(vmf->address), - vmf->vma_page_prot)); - if (!pte_map_lock(vmf)) - return VM_FAULT_RETRY; + vma->vm_page_prot)); + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, + vmf->address, &vmf->ptl); if (!pte_none(*vmf->pte)) goto unlock; ret = check_stable_address_space(vma->vm_mm); if (ret) goto unlock; - /* - * Don't call the userfaultfd during the speculative path. - * We already checked for the VMA to not be managed through - * userfaultfd, but it may be set in our back once we have lock - * the pte. In such a case we can ignore it this time. - */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto setpte; /* Deliver the page fault to userland, check inside PT lock */ if (userfaultfd_missing(vma)) { pte_unmap_unlock(vmf->pte, vmf->ptl); @@ -3416,24 +3222,21 @@ skip_pmd_checks: */ __SetPageUptodate(page); - entry = mk_pte(page, vmf->vma_page_prot); - if (vmf->vma_flags & VM_WRITE) + entry = mk_pte(page, vma->vm_page_prot); + if (vma->vm_flags & VM_WRITE) entry = pte_mkwrite(pte_mkdirty(entry)); - if (!pte_map_lock(vmf)) { - ret = VM_FAULT_RETRY; - goto release; - } + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); if (!pte_none(*vmf->pte)) - goto unlock_and_release; + goto release; ret = check_stable_address_space(vma->vm_mm); if (ret) - goto unlock_and_release; + goto release; /* Deliver the page fault to userland, check inside PT lock */ - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE) && - userfaultfd_missing(vma)) { + if (userfaultfd_missing(vma)) { pte_unmap_unlock(vmf->pte, vmf->ptl); mem_cgroup_cancel_charge(page, memcg, false); put_page(page); @@ -3441,9 +3244,9 @@ skip_pmd_checks: } inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); - __page_add_new_anon_rmap(page, vma, vmf->address, false); + page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); - __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(page, vma); setpte: set_pte_at(vma->vm_mm, vmf->address, vmf->pte, entry); @@ -3452,12 +3255,10 @@ setpte: unlock: pte_unmap_unlock(vmf->pte, vmf->ptl); return ret; -unlock_and_release: - pte_unmap_unlock(vmf->pte, vmf->ptl); release: mem_cgroup_cancel_charge(page, memcg, false); put_page(page); - return ret; + goto unlock; oom_free_page: put_page(page); oom: @@ -3474,10 +3275,6 @@ static vm_fault_t __do_fault(struct vm_fault *vmf) struct vm_area_struct *vma = vmf->vma; vm_fault_t ret; - /* Do not check unstable pmd, if it's changed will retry later */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto skip_pmd_checks; - /* * Preallocate pte before we take page_lock because this might lead to * deadlocks for memcg reclaim which waits for pages under writeback: @@ -3500,7 +3297,6 @@ static vm_fault_t __do_fault(struct vm_fault *vmf) smp_wmb(); /* See comment in __pte_alloc() */ } -skip_pmd_checks: ret = vma->vm_ops->fault(vmf); if (unlikely(ret & (VM_FAULT_ERROR | VM_FAULT_NOPAGE | VM_FAULT_RETRY | VM_FAULT_DONE_COW))) @@ -3546,7 +3342,7 @@ static vm_fault_t pte_alloc_one_map(struct vm_fault *vmf) { struct vm_area_struct *vma = vmf->vma; - if (!pmd_none(*vmf->pmd) || (vmf->flags & FAULT_FLAG_SPECULATIVE)) + if (!pmd_none(*vmf->pmd)) goto map_pte; if (vmf->prealloc_pte) { vmf->ptl = pmd_lock(vma->vm_mm, vmf->pmd); @@ -3586,9 +3382,8 @@ map_pte: * pte_none() under vmf->ptl protection when we return to * alloc_set_pte(). */ - if (!pte_map_lock(vmf)) - return VM_FAULT_RETRY; - + vmf->pte = pte_offset_map_lock(vma->vm_mm, vmf->pmd, vmf->address, + &vmf->ptl); return 0; } @@ -3639,7 +3434,7 @@ static vm_fault_t do_set_pmd(struct vm_fault *vmf, struct page *page) for (i = 0; i < HPAGE_PMD_NR; i++) flush_icache_page(vma, page + i); - entry = mk_huge_pmd(page, vmf->vma_page_prot); + entry = mk_huge_pmd(page, vma->vm_page_prot); if (write) entry = maybe_pmd_mkwrite(pmd_mkdirty(entry), vma); @@ -3715,19 +3510,19 @@ vm_fault_t alloc_set_pte(struct vm_fault *vmf, struct mem_cgroup *memcg, return VM_FAULT_NOPAGE; flush_icache_page(vma, page); - entry = mk_pte(page, vmf->vma_page_prot); + entry = mk_pte(page, vma->vm_page_prot); if (write) - entry = maybe_mkwrite(pte_mkdirty(entry), vmf->vma_flags); + entry = maybe_mkwrite(pte_mkdirty(entry), vma); if (vmf->flags & FAULT_FLAG_PREFAULT_OLD) entry = pte_mkold(entry); /* copy-on-write page */ - if (write && !(vmf->vma_flags & VM_SHARED)) { + if (write && !(vma->vm_flags & VM_SHARED)) { inc_mm_counter_fast(vma->vm_mm, MM_ANONPAGES); - __page_add_new_anon_rmap(page, vma, vmf->address, false); + page_add_new_anon_rmap(page, vma, vmf->address, false); mem_cgroup_commit_charge(page, memcg, false, false); - __lru_cache_add_active_or_unevictable(page, vmf->vma_flags); + lru_cache_add_active_or_unevictable(page, vma); } else { inc_mm_counter_fast(vma->vm_mm, mm_counter_file(page)); page_add_file_rmap(page, false); @@ -3763,7 +3558,7 @@ vm_fault_t finish_fault(struct vm_fault *vmf) /* Did we COW the page? */ if ((vmf->flags & FAULT_FLAG_WRITE) && - !(vmf->vma_flags & VM_SHARED)) + !(vmf->vma->vm_flags & VM_SHARED)) page = vmf->cow_page; else page = vmf->page; @@ -3874,8 +3669,7 @@ static vm_fault_t do_fault_around(struct vm_fault *vmf) end_pgoff = min3(end_pgoff, vma_data_pages(vmf->vma) + vmf->vma->vm_pgoff - 1, start_pgoff + nr_pages - 1); - if (!(vmf->flags & FAULT_FLAG_SPECULATIVE) && - pmd_none(*vmf->pmd)) { + if (pmd_none(*vmf->pmd)) { vmf->prealloc_pte = pte_alloc_one(vmf->vma->vm_mm); if (!vmf->prealloc_pte) goto out; @@ -4053,7 +3847,7 @@ static vm_fault_t do_fault(struct vm_fault *vmf) } } else if (!(vmf->flags & FAULT_FLAG_WRITE)) ret = do_read_fault(vmf); - else if (!(vmf->vma_flags & VM_SHARED)) + else if (!(vma->vm_flags & VM_SHARED)) ret = do_cow_fault(vmf); else ret = do_shared_fault(vmf); @@ -4098,8 +3892,8 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) * validation through pte_unmap_same(). It's of NUMA type but * the pfn may be screwed if the read is non atomic. */ - if (!pte_spinlock(vmf)) - return VM_FAULT_RETRY; + vmf->ptl = pte_lockptr(vma->vm_mm, vmf->pmd); + spin_lock(vmf->ptl); if (unlikely(!pte_same(*vmf->pte, vmf->orig_pte))) { pte_unmap_unlock(vmf->pte, vmf->ptl); goto out; @@ -4110,14 +3904,14 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) * accessible ptes, some can allow access by kernel mode. */ old_pte = ptep_modify_prot_start(vma, vmf->address, vmf->pte); - pte = pte_modify(old_pte, vmf->vma_page_prot); + pte = pte_modify(old_pte, vma->vm_page_prot); pte = pte_mkyoung(pte); if (was_writable) pte = pte_mkwrite(pte); ptep_modify_prot_commit(vma, vmf->address, vmf->pte, old_pte, pte); update_mmu_cache(vma, vmf->address, vmf->pte); - page = _vm_normal_page(vma, vmf->address, pte, vmf->vma_flags); + page = vm_normal_page(vma, vmf->address, pte); if (!page) { pte_unmap_unlock(vmf->pte, vmf->ptl); return 0; @@ -4144,7 +3938,7 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) * Flag if the page is shared between multiple address spaces. This * is later used when determining whether to group tasks together */ - if (page_mapcount(page) > 1 && (vmf->vma_flags & VM_SHARED)) + if (page_mapcount(page) > 1 && (vma->vm_flags & VM_SHARED)) flags |= TNF_SHARED; last_cpupid = page_cpupid_last(page); @@ -4158,7 +3952,7 @@ static vm_fault_t do_numa_page(struct vm_fault *vmf) } /* Migrate to the requested node */ - migrated = migrate_misplaced_page(page, vmf, target_nid); + migrated = migrate_misplaced_page(page, vma, target_nid); if (migrated) { page_nid = target_nid; flags |= TNF_MIGRATED; @@ -4189,7 +3983,7 @@ static inline vm_fault_t wp_huge_pmd(struct vm_fault *vmf, pmd_t orig_pmd) return vmf->vma->vm_ops->huge_fault(vmf, PE_SIZE_PMD); /* COW handled on pte level: split pmd */ - VM_BUG_ON_VMA(vmf->vma_flags & VM_SHARED, vmf->vma); + VM_BUG_ON_VMA(vmf->vma->vm_flags & VM_SHARED, vmf->vma); __split_huge_pmd(vmf->vma, vmf->pmd, vmf->address, false, NULL); return VM_FAULT_FALLBACK; @@ -4242,11 +4036,6 @@ static vm_fault_t wp_huge_pud(struct vm_fault *vmf, pud_t orig_pud) static vm_fault_t handle_pte_fault(struct vm_fault *vmf) { pte_t entry; - vm_fault_t ret = 0; - - /* Do not check unstable pmd, if it's changed will retry later */ - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - goto skip_pmd_checks; if (unlikely(pmd_none(*vmf->pmd))) { /* @@ -4257,6 +4046,7 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) */ vmf->pte = NULL; } else { + /* See comment in pte_alloc_one_map() */ if (pmd_devmap_trans_unstable(vmf->pmd)) return 0; /* @@ -4264,9 +4054,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) * pmd from under us anymore at this point because we hold the * mmap_sem read mode and khugepaged takes it in write mode. * So now it's safe to run pte_offset_map(). - * This is not applicable to the speculative page fault handler - * but in that case, the pte is fetched earlier in - * handle_speculative_fault(). */ vmf->pte = pte_offset_map(vmf->pmd, vmf->address); vmf->orig_pte = *vmf->pte; @@ -4286,7 +4073,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf) } } -skip_pmd_checks: if (!vmf->pte) { if (vma_is_anonymous(vmf->vma)) return do_anonymous_page(vmf); @@ -4300,8 +4086,8 @@ skip_pmd_checks: if (pte_protnone(vmf->orig_pte) && vma_is_accessible(vmf->vma)) return do_numa_page(vmf); - if (!pte_spinlock(vmf)) - return VM_FAULT_RETRY; + vmf->ptl = pte_lockptr(vmf->vma->vm_mm, vmf->pmd); + spin_lock(vmf->ptl); entry = vmf->orig_pte; if (unlikely(!pte_same(*vmf->pte, entry))) goto unlock; @@ -4323,12 +4109,10 @@ skip_pmd_checks: */ if (vmf->flags & FAULT_FLAG_WRITE) flush_tlb_fix_spurious_fault(vmf->vma, vmf->address); - if (vmf->flags & FAULT_FLAG_SPECULATIVE) - ret = VM_FAULT_RETRY; } unlock: pte_unmap_unlock(vmf->pte, vmf->ptl); - return ret; + return 0; } /* @@ -4346,8 +4130,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, .flags = flags, .pgoff = linear_page_index(vma, address), .gfp_mask = __get_fault_gfp_mask(vma), - .vma_flags = vma->vm_flags, - .vma_page_prot = vma->vm_page_prot, }; unsigned int dirty = flags & FAULT_FLAG_WRITE; struct mm_struct *mm = vma->vm_mm; @@ -4389,10 +4171,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, vmf.pmd = pmd_alloc(mm, vmf.pud, address); if (!vmf.pmd) return VM_FAULT_OOM; - -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - vmf.sequence = raw_read_seqcount(&vma->vm_sequence); -#endif if (pmd_none(*vmf.pmd) && __transparent_hugepage_enabled(vma)) { ret = create_huge_pmd(&vmf); if (!(ret & VM_FAULT_FALLBACK)) @@ -4426,252 +4204,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma, return handle_pte_fault(&vmf); } -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - -#ifndef CONFIG_ARCH_HAS_PTE_SPECIAL -/* This is required by vm_normal_page() */ -#error "Speculative page fault handler requires CONFIG_ARCH_HAS_PTE_SPECIAL" -#endif -/* - * vm_normal_page() adds some processing which should be done while - * hodling the mmap_sem. - */ - -/* - * Tries to handle the page fault in a speculative way, without grabbing the - * mmap_sem. - * When VM_FAULT_RETRY is returned, the vma pointer is valid and this vma must - * be checked later when the mmap_sem has been grabbed by calling - * can_reuse_spf_vma(). - * This is needed as the returned vma is kept in memory until the call to - * can_reuse_spf_vma() is made. - */ -int __handle_speculative_fault(struct mm_struct *mm, unsigned long address, - unsigned int flags, struct vm_area_struct **vma) -{ - struct vm_fault vmf = { - .address = address, - }; - pgd_t *pgd, pgdval; - p4d_t *p4d, p4dval; - pud_t pudval; - int seq, ret; - - /* Clear flags that may lead to release the mmap_sem to retry */ - flags &= ~(FAULT_FLAG_ALLOW_RETRY|FAULT_FLAG_KILLABLE); - flags |= FAULT_FLAG_SPECULATIVE; - - *vma = get_vma(mm, address); - if (!*vma) - return VM_FAULT_RETRY; - vmf.vma = *vma; - - /* rmb <-> seqlock,vma_rb_erase() */ - seq = raw_read_seqcount(&vmf.vma->vm_sequence); - if (seq & 1) - return VM_FAULT_RETRY; - - /* - * __anon_vma_prepare() requires the mmap_sem to be held - * because vm_next and vm_prev must be safe. This can't be guaranteed - * in the speculative path. - */ - if (unlikely((vma_is_anonymous(vmf.vma) && !vmf.vma->anon_vma) || - (!vma_is_anonymous(vmf.vma) && - !(vmf.vma->vm_flags & VM_SHARED) && - (flags & FAULT_FLAG_WRITE) && - !vmf.vma->anon_vma))) - return VM_FAULT_RETRY; - - vmf.vma_flags = READ_ONCE(vmf.vma->vm_flags); - vmf.vma_page_prot = READ_ONCE(vmf.vma->vm_page_prot); - - /* Can't call userland page fault handler in the speculative path */ - if (unlikely(vmf.vma_flags & VM_UFFD_MISSING)) - return VM_FAULT_RETRY; - - if (vmf.vma_flags & VM_GROWSDOWN || vmf.vma_flags & VM_GROWSUP) - /* - * This could be detected by the check address against VMA's - * boundaries but we want to trace it as not supported instead - * of changed. - */ - return VM_FAULT_RETRY; - - if (address < READ_ONCE(vmf.vma->vm_start) - || READ_ONCE(vmf.vma->vm_end) <= address) - return VM_FAULT_RETRY; - - if (!arch_vma_access_permitted(vmf.vma, flags & FAULT_FLAG_WRITE, - flags & FAULT_FLAG_INSTRUCTION, - flags & FAULT_FLAG_REMOTE)) - goto out_segv; - - /* This is one is required to check that the VMA has write access set */ - if (flags & FAULT_FLAG_WRITE) { - if (unlikely(!(vmf.vma_flags & VM_WRITE))) - goto out_segv; - } else if (unlikely(!(vmf.vma_flags & (VM_READ|VM_EXEC|VM_WRITE)))) - goto out_segv; - -#ifdef CONFIG_NUMA - struct mempolicy *pol; - - /* - * MPOL_INTERLEAVE implies additional checks in - * mpol_misplaced() which are not compatible with the - *speculative page fault processing. - */ - pol = __get_vma_policy(vmf.vma, address); - if (!pol) - pol = get_task_policy(current); - if (pol && pol->mode == MPOL_INTERLEAVE) - return VM_FAULT_RETRY; -#endif - - /* - * Do a speculative lookup of the PTE entry. - */ - local_irq_disable(); - pgd = pgd_offset(mm, address); - pgdval = READ_ONCE(*pgd); - if (pgd_none(pgdval) || unlikely(pgd_bad(pgdval))) - goto out_walk; - - p4d = p4d_offset(pgd, address); - if (pgd_val(READ_ONCE(*pgd)) != pgd_val(pgdval)) - goto out_walk; - p4dval = READ_ONCE(*p4d); - if (p4d_none(p4dval) || unlikely(p4d_bad(p4dval))) - goto out_walk; - - vmf.pud = pud_offset(p4d, address); - if (p4d_val(READ_ONCE(*p4d)) != p4d_val(p4dval)) - goto out_walk; - pudval = READ_ONCE(*vmf.pud); - if (pud_none(pudval) || unlikely(pud_bad(pudval))) - goto out_walk; - - /* Huge pages at PUD level are not supported. */ - if (unlikely(pud_trans_huge(pudval))) - goto out_walk; - - vmf.pmd = pmd_offset(vmf.pud, address); - if (pud_val(READ_ONCE(*vmf.pud)) != pud_val(pudval)) - goto out_walk; - vmf.orig_pmd = READ_ONCE(*vmf.pmd); - /* - * pmd_none could mean that a hugepage collapse is in progress - * in our back as collapse_huge_page() mark it before - * invalidating the pte (which is done once the IPI is catched - * by all CPU and we have interrupt disabled). - * For this reason we cannot handle THP in a speculative way since we - * can't safely indentify an in progress collapse operation done in our - * back on that PMD. - * Regarding the order of the following checks, see comment in - * pmd_devmap_trans_unstable() - */ - if (unlikely(pmd_devmap(vmf.orig_pmd) || - pmd_none(vmf.orig_pmd) || pmd_trans_huge(vmf.orig_pmd) || - is_swap_pmd(vmf.orig_pmd))) - goto out_walk; - - /* - * The above does not allocate/instantiate page-tables because doing so - * would lead to the possibility of instantiating page-tables after - * free_pgtables() -- and consequently leaking them. - * - * The result is that we take at least one !speculative fault per PMD - * in order to instantiate it. - */ - - vmf.pte = pte_offset_map(vmf.pmd, address); - if (pmd_val(READ_ONCE(*vmf.pmd)) != pmd_val(vmf.orig_pmd)) { - pte_unmap(vmf.pte); - vmf.pte = NULL; - goto out_walk; - } - vmf.orig_pte = READ_ONCE(*vmf.pte); - barrier(); /* See comment in handle_pte_fault() */ - if (pte_none(vmf.orig_pte)) { - pte_unmap(vmf.pte); - vmf.pte = NULL; - } - - vmf.pgoff = linear_page_index(vmf.vma, address); - vmf.gfp_mask = __get_fault_gfp_mask(vmf.vma); - vmf.sequence = seq; - vmf.flags = flags; - - local_irq_enable(); - - /* - * We need to re-validate the VMA after checking the bounds, otherwise - * we might have a false positive on the bounds. - */ - if (read_seqcount_retry(&vmf.vma->vm_sequence, seq)) - return VM_FAULT_RETRY; - - mem_cgroup_enter_user_fault(); - ret = handle_pte_fault(&vmf); - mem_cgroup_exit_user_fault(); - - /* - * If there is no need to retry, don't return the vma to the caller. - */ - if (ret != VM_FAULT_RETRY) { - check_sync_rss_stat(current); - if (vma_is_anonymous(vmf.vma)) - count_vm_event(SPECULATIVE_PGFAULT_ANON); - else - count_vm_event(SPECULATIVE_PGFAULT_FILE); - put_vma(vmf.vma); - *vma = NULL; - } - - /* - * The task may have entered a memcg OOM situation but - * if the allocation error was handled gracefully (no - * VM_FAULT_OOM), there is no need to kill anything. - * Just clean up the OOM state peacefully. - */ - if (task_in_memcg_oom(current) && !(ret & VM_FAULT_OOM)) - mem_cgroup_oom_synchronize(false); - return ret; - -out_walk: - local_irq_enable(); - return VM_FAULT_RETRY; - -out_segv: - /* - * We don't return VM_FAULT_RETRY so the caller is not expected to - * retrieve the fetched VMA. - */ - put_vma(vmf.vma); - *vma = NULL; - return VM_FAULT_SIGSEGV; -} - -/* - * This is used to know if the vma fetch in the speculative page fault handler - * is still valid when trying the regular fault path while holding the - * mmap_sem. - * The call to put_vma(vma) must be made after checking the vma's fields, as - * the vma may be freed by put_vma(). In such a case it is expected that false - * is returned. - */ -bool can_reuse_spf_vma(struct vm_area_struct *vma, unsigned long address) -{ - bool ret; - - ret = !RB_EMPTY_NODE(&vma->vm_rb) && - vma->vm_start <= address && address < vma->vm_end; - put_vma(vma); - return ret; -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - /* * By the time we get here, we already hold the mm semaphore * diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 05c02b191995..c3170460aa52 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -380,11 +380,8 @@ void mpol_rebind_mm(struct mm_struct *mm, nodemask_t *new) struct vm_area_struct *vma; down_write(&mm->mmap_sem); - for (vma = mm->mmap; vma; vma = vma->vm_next) { - vm_write_begin(vma); + for (vma = mm->mmap; vma; vma = vma->vm_next) mpol_rebind_policy(vma->vm_policy, new); - vm_write_end(vma); - } up_write(&mm->mmap_sem); } @@ -715,7 +712,6 @@ static int vma_replace_policy(struct vm_area_struct *vma, if (IS_ERR(new)) return PTR_ERR(new); - vm_write_begin(vma); if (vma->vm_ops && vma->vm_ops->set_policy) { err = vma->vm_ops->set_policy(vma, new); if (err) @@ -723,17 +719,11 @@ static int vma_replace_policy(struct vm_area_struct *vma, } old = vma->vm_policy; - /* - * The speculative page fault handler accesses this field without - * hodling the mmap_sem. - */ - WRITE_ONCE(vma->vm_policy, new); - vm_write_end(vma); + vma->vm_policy = new; /* protected by mmap_sem */ mpol_put(old); return 0; err_out: - vm_write_end(vma); mpol_put(new); return err; } @@ -1708,28 +1698,23 @@ COMPAT_SYSCALL_DEFINE4(migrate_pages, compat_pid_t, pid, struct mempolicy *__get_vma_policy(struct vm_area_struct *vma, unsigned long addr) { - struct mempolicy *pol; + struct mempolicy *pol = NULL; - if (!vma) - return NULL; + if (vma) { + if (vma->vm_ops && vma->vm_ops->get_policy) { + pol = vma->vm_ops->get_policy(vma, addr); + } else if (vma->vm_policy) { + pol = vma->vm_policy; - if (vma->vm_ops && vma->vm_ops->get_policy) - return vma->vm_ops->get_policy(vma, addr); - - /* - * This could be called without holding the mmap_sem in the - * speculative page fault handler's path. - */ - pol = READ_ONCE(vma->vm_policy); - if (pol) { - /* - * shmem_alloc_page() passes MPOL_F_SHARED policy with - * a pseudo vma whose vma->vm_ops=NULL. Take a reference - * count on these policies which will be dropped by - * mpol_cond_put() later - */ - if (mpol_needs_cond_ref(pol)) - mpol_get(pol); + /* + * shmem_alloc_page() passes MPOL_F_SHARED policy with + * a pseudo vma whose vma->vm_ops=NULL. Take a reference + * count on these policies which will be dropped by + * mpol_cond_put() later + */ + if (mpol_needs_cond_ref(pol)) + mpol_get(pol); + } } return pol; diff --git a/mm/migrate.c b/mm/migrate.c index 9a6bdbceec00..1c58ab614c50 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -241,7 +241,7 @@ static bool remove_migration_pte(struct page *page, struct vm_area_struct *vma, */ entry = pte_to_swp_entry(*pvmw.pte); if (is_write_migration_entry(entry)) - pte = maybe_mkwrite(pte, vma->vm_flags); + pte = maybe_mkwrite(pte, vma); if (unlikely(is_zone_device_page(new))) { if (is_device_private_page(new)) { @@ -1976,7 +1976,7 @@ bool pmd_trans_migrating(pmd_t pmd) * node. Caller is expected to have an elevated reference count on * the page that will be dropped by this function before returning. */ -int migrate_misplaced_page(struct page *page, struct vm_fault *vmf, +int migrate_misplaced_page(struct page *page, struct vm_area_struct *vma, int node) { pg_data_t *pgdat = NODE_DATA(node); @@ -1989,7 +1989,7 @@ int migrate_misplaced_page(struct page *page, struct vm_fault *vmf, * with execute permissions as they are probably shared libraries. */ if (page_mapcount(page) != 1 && page_is_file_cache(page) && - (vmf->vma_flags & VM_EXEC)) + (vma->vm_flags & VM_EXEC)) goto out; /* diff --git a/mm/mlock.c b/mm/mlock.c index 5805a284512d..ff9529310b3b 100644 --- a/mm/mlock.c +++ b/mm/mlock.c @@ -446,9 +446,7 @@ static unsigned long __munlock_pagevec_fill(struct pagevec *pvec, void munlock_vma_pages_range(struct vm_area_struct *vma, unsigned long start, unsigned long end) { - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma->vm_flags & VM_LOCKED_CLEAR_MASK); - vm_write_end(vma); + vma->vm_flags &= VM_LOCKED_CLEAR_MASK; while (start < end) { struct page *page; @@ -572,11 +570,11 @@ success: * It's okay if try_to_unmap_one unmaps a page just after we * set VM_LOCKED, populate_vma_page_range will bring it back. */ - if (lock) { - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, newflags)); - vm_write_end(vma); - } else + + if (lock) + vma->vm_flags = vma_pad_fixup_flags(vma, newflags); + + else munlock_vma_pages_range(vma, start, end); out: diff --git a/mm/mmap.c b/mm/mmap.c index fcc6f69fd4be..bcc12a4f0662 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -166,27 +166,6 @@ void unlink_file_vma(struct vm_area_struct *vma) } } -static void __free_vma(struct vm_area_struct *vma) -{ - if (vma->vm_file) - fput(vma->vm_file); - mpol_put(vma_policy(vma)); - vm_area_free(vma); -} - -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -void put_vma(struct vm_area_struct *vma) -{ - if (atomic_dec_and_test(&vma->vm_ref_count)) - __free_vma(vma); -} -#else -static inline void put_vma(struct vm_area_struct *vma) -{ - __free_vma(vma); -} -#endif - /* * Close a vm structure and free it, returning the next. */ @@ -197,7 +176,10 @@ static struct vm_area_struct *remove_vma(struct vm_area_struct *vma) might_sleep(); if (vma->vm_ops && vma->vm_ops->close) vma->vm_ops->close(vma); - put_vma(vma); + if (vma->vm_file) + fput(vma->vm_file); + mpol_put(vma_policy(vma)); + vm_area_free(vma); return next; } @@ -450,13 +432,6 @@ static void validate_mm(struct mm_struct *mm) RB_DECLARE_CALLBACKS_MAX(static, vma_gap_callbacks, struct vm_area_struct, vm_rb, unsigned long, rb_subtree_gap, vma_compute_gap) -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -#define mm_rb_write_lock(mm) write_lock(&(mm)->mm_rb_lock) -#define mm_rb_write_unlock(mm) write_unlock(&(mm)->mm_rb_lock) -#else -#define mm_rb_write_lock(mm) do { } while (0) -#define mm_rb_write_unlock(mm) do { } while (0) -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ /* * Update augmented rbtree rb_subtree_gap values after vma->vm_start or @@ -473,37 +448,26 @@ static void vma_gap_update(struct vm_area_struct *vma) } static inline void vma_rb_insert(struct vm_area_struct *vma, - struct mm_struct *mm) + struct rb_root *root) { - struct rb_root *root = &mm->mm_rb; - /* All rb_subtree_gap values must be consistent prior to insertion */ validate_mm_rb(root, NULL); rb_insert_augmented(&vma->vm_rb, root, &vma_gap_callbacks); } -static void __vma_rb_erase(struct vm_area_struct *vma, struct mm_struct *mm) +static void __vma_rb_erase(struct vm_area_struct *vma, struct rb_root *root) { - struct rb_root *root = &mm->mm_rb; /* * Note rb_erase_augmented is a fairly large inline function, * so make sure we instantiate it only once with our desired * augmented rbtree callbacks. */ - mm_rb_write_lock(mm); rb_erase_augmented(&vma->vm_rb, root, &vma_gap_callbacks); - mm_rb_write_unlock(mm); /* wmb */ - - /* - * Ensure the removal is complete before clearing the node. - * Matched by vma_has_changed()/handle_speculative_fault(). - */ - RB_CLEAR_NODE(&vma->vm_rb); } static __always_inline void vma_rb_erase_ignore(struct vm_area_struct *vma, - struct mm_struct *mm, + struct rb_root *root, struct vm_area_struct *ignore) { /* @@ -511,21 +475,21 @@ static __always_inline void vma_rb_erase_ignore(struct vm_area_struct *vma, * with the possible exception of the "next" vma being erased if * next->vm_start was reduced. */ - validate_mm_rb(&mm->mm_rb, ignore); + validate_mm_rb(root, ignore); - __vma_rb_erase(vma, mm); + __vma_rb_erase(vma, root); } static __always_inline void vma_rb_erase(struct vm_area_struct *vma, - struct mm_struct *mm) + struct rb_root *root) { /* * All rb_subtree_gap values must be consistent prior to erase, * with the possible exception of the vma being erased. */ - validate_mm_rb(&mm->mm_rb, vma); + validate_mm_rb(root, vma); - __vma_rb_erase(vma, mm); + __vma_rb_erase(vma, root); } /* @@ -640,12 +604,10 @@ void __vma_link_rb(struct mm_struct *mm, struct vm_area_struct *vma, * immediately update the gap to the correct value. Finally we * rebalance the rbtree after all augmented values have been set. */ - mm_rb_write_lock(mm); rb_link_node(&vma->vm_rb, rb_parent, rb_link); vma->rb_subtree_gap = 0; vma_gap_update(vma); - vma_rb_insert(vma, mm); - mm_rb_write_unlock(mm); + vma_rb_insert(vma, &mm->mm_rb); } static void __vma_link_file(struct vm_area_struct *vma) @@ -721,7 +683,7 @@ static __always_inline void __vma_unlink_common(struct mm_struct *mm, { struct vm_area_struct *next; - vma_rb_erase_ignore(vma, mm, ignore); + vma_rb_erase_ignore(vma, &mm->mm_rb, ignore); next = vma->vm_next; if (has_prev) prev->vm_next = next; @@ -755,7 +717,7 @@ static inline void __vma_unlink_prev(struct mm_struct *mm, */ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, unsigned long end, pgoff_t pgoff, struct vm_area_struct *insert, - struct vm_area_struct *expand, bool keep_locked) + struct vm_area_struct *expand) { struct mm_struct *mm = vma->vm_mm; struct vm_area_struct *next = vma->vm_next, *orig_vma = vma; @@ -767,10 +729,6 @@ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, long adjust_next = 0; int remove_next = 0; - vm_write_begin(vma); - if (next) - vm_write_begin(next); - if (next && !insert) { struct vm_area_struct *exporter = NULL, *importer = NULL; @@ -851,12 +809,8 @@ int __vma_adjust(struct vm_area_struct *vma, unsigned long start, importer->anon_vma = exporter->anon_vma; error = anon_vma_clone(importer, exporter); - if (error) { - if (next && next != vma) - vm_write_end(next); - vm_write_end(vma); + if (error) return error; - } } } again: @@ -902,18 +856,17 @@ again: } if (start != vma->vm_start) { - WRITE_ONCE(vma->vm_start, start); + vma->vm_start = start; start_changed = true; } if (end != vma->vm_end) { - WRITE_ONCE(vma->vm_end, end); + vma->vm_end = end; end_changed = true; } - WRITE_ONCE(vma->vm_pgoff, pgoff); + vma->vm_pgoff = pgoff; if (adjust_next) { - WRITE_ONCE(next->vm_start, - next->vm_start + (adjust_next << PAGE_SHIFT)); - WRITE_ONCE(next->vm_pgoff, next->vm_pgoff + adjust_next); + next->vm_start += adjust_next << PAGE_SHIFT; + next->vm_pgoff += adjust_next; } if (root) { @@ -978,13 +931,15 @@ again: } if (remove_next) { - if (file) + if (file) { uprobe_munmap(next, next->vm_start, next->vm_end); + fput(file); + } if (next->anon_vma) anon_vma_merge(vma, next); mm->map_count--; - vm_write_end(next); - put_vma(next); + mpol_put(vma_policy(next)); + vm_area_free(next); /* * In mprotect's case 6 (see comments on vma_merge), * we must remove another next too. It would clutter @@ -998,8 +953,6 @@ again: * "vma->vm_next" gap must be updated. */ next = vma->vm_next; - if (next) - vm_write_begin(next); } else { /* * For the scope of the comment "next" and @@ -1046,11 +999,6 @@ again: if (insert && file) uprobe_mmap(insert); - if (next && next != vma) - vm_write_end(next); - if (!keep_locked) - vm_write_end(vma); - validate_mm(mm); return 0; @@ -1192,13 +1140,13 @@ can_vma_merge_after(struct vm_area_struct *vma, unsigned long vm_flags, * parameter) may establish ptes with the wrong permissions of NNNN * instead of the right permissions of XXXX. */ -struct vm_area_struct *__vma_merge(struct mm_struct *mm, +struct vm_area_struct *vma_merge(struct mm_struct *mm, struct vm_area_struct *prev, unsigned long addr, unsigned long end, unsigned long vm_flags, struct anon_vma *anon_vma, struct file *file, pgoff_t pgoff, struct mempolicy *policy, struct vm_userfaultfd_ctx vm_userfaultfd_ctx, - const char __user *anon_name, bool keep_locked) + const char __user *anon_name) { pgoff_t pglen = (end - addr) >> PAGE_SHIFT; struct vm_area_struct *area, *next; @@ -1248,11 +1196,10 @@ struct vm_area_struct *__vma_merge(struct mm_struct *mm, /* cases 1, 6 */ err = __vma_adjust(prev, prev->vm_start, next->vm_end, prev->vm_pgoff, NULL, - prev, keep_locked); + prev); } else /* cases 2, 5, 7 */ err = __vma_adjust(prev, prev->vm_start, - end, prev->vm_pgoff, NULL, prev, - keep_locked); + end, prev->vm_pgoff, NULL, prev); if (err) return NULL; khugepaged_enter_vma_merge(prev, vm_flags); @@ -1270,12 +1217,10 @@ struct vm_area_struct *__vma_merge(struct mm_struct *mm, anon_name)) { if (prev && addr < prev->vm_end) /* case 4 */ err = __vma_adjust(prev, prev->vm_start, - addr, prev->vm_pgoff, NULL, next, - keep_locked); + addr, prev->vm_pgoff, NULL, next); else { /* cases 3, 8 */ err = __vma_adjust(area, addr, next->vm_end, - next->vm_pgoff - pglen, NULL, next, - keep_locked); + next->vm_pgoff - pglen, NULL, next); /* * In case 3 area is already equal to next and * this is a noop, but in case 8 "area" has @@ -1899,14 +1844,12 @@ unsigned long mmap_region(struct file *file, unsigned long addr, out: perf_event_mmap(vma); - vm_write_begin(vma); vm_stat_account(mm, vm_flags, len >> PAGE_SHIFT); if (vm_flags & VM_LOCKED) { if ((vm_flags & VM_SPECIAL) || vma_is_dax(vma) || is_vm_hugetlb_page(vma) || vma == get_gate_vma(current->mm)) - WRITE_ONCE(vma->vm_flags, - vma->vm_flags & VM_LOCKED_CLEAR_MASK); + vma->vm_flags &= VM_LOCKED_CLEAR_MASK; else mm->locked_vm += (len >> PAGE_SHIFT); } @@ -1921,10 +1864,9 @@ out: * then new mapped in-place (which must be aimed as * a completely new data area). */ - WRITE_ONCE(vma->vm_flags, vma->vm_flags | VM_SOFTDIRTY); + vma->vm_flags |= VM_SOFTDIRTY; vma_set_page_prot(vma); - vm_write_end(vma); return addr; @@ -2299,11 +2241,15 @@ get_unmapped_area(struct file *file, unsigned long addr, unsigned long len, EXPORT_SYMBOL(get_unmapped_area); /* Look up the first VMA which satisfies addr < vm_end, NULL if none. */ -static struct vm_area_struct *__find_vma(struct mm_struct *mm, - unsigned long addr) +struct vm_area_struct *find_vma(struct mm_struct *mm, unsigned long addr) { struct rb_node *rb_node; - struct vm_area_struct *vma = NULL; + struct vm_area_struct *vma; + + /* Check the cache first. */ + vma = vmacache_find(mm, addr); + if (likely(vma)) + return vma; rb_node = mm->mm_rb.rb_node; @@ -2321,54 +2267,13 @@ static struct vm_area_struct *__find_vma(struct mm_struct *mm, rb_node = rb_node->rb_right; } - return vma; -} - -struct vm_area_struct *find_vma(struct mm_struct *mm, unsigned long addr) -{ - struct vm_area_struct *vma; - - /* Check the cache first. */ - vma = vmacache_find(mm, addr); - if (likely(vma)) - return vma; - - vma = __find_vma(mm, addr); if (vma) vmacache_update(addr, vma); return vma; } + EXPORT_SYMBOL(find_vma); -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -struct vm_area_struct *get_vma(struct mm_struct *mm, unsigned long addr) -{ - struct vm_area_struct *vma = NULL; - - read_lock(&mm->mm_rb_lock); - vma = __find_vma(mm, addr); - - /* - * If there is a concurrent fast mremap, bail out since the entire - * PMD/PUD subtree may have been remapped. - * - * This is usually safe for conventional mremap since it takes the - * PTE locks as does SPF. However fast mremap only takes the lock - * at the PMD/PUD level which is ok as it is done with the mmap - * write lock held. But since SPF, as the term implies forgoes, - * taking the mmap read lock and also cannot take PTL lock at the - * larger PMD/PUD granualrity, since it would introduce huge - * contention in the page fault path; fall back to regular fault - * handling. - */ - if (vma && !atomic_inc_unless_negative(&vma->vm_ref_count)) - vma = NULL; - read_unlock(&mm->mm_rb_lock); - - return vma; -} -#endif - /* * Same as find_vma, but also return a pointer to the previous VMA in *pprev. */ @@ -2589,8 +2494,8 @@ int expand_downwards(struct vm_area_struct *vma, mm->locked_vm += grow; vm_stat_account(mm, vma->vm_flags, grow); anon_vma_interval_tree_pre_update_vma(vma); - WRITE_ONCE(vma->vm_start, address); - WRITE_ONCE(vma->vm_pgoff, vma->vm_pgoff - grow); + vma->vm_start = address; + vma->vm_pgoff -= grow; anon_vma_interval_tree_post_update_vma(vma); vma_gap_update(vma); spin_unlock(&mm->page_table_lock); @@ -2753,7 +2658,7 @@ detach_vmas_to_be_unmapped(struct mm_struct *mm, struct vm_area_struct *vma, insertion_point = (prev ? &prev->vm_next : &mm->mmap); vma->vm_prev = NULL; do { - vma_rb_erase(vma, mm); + vma_rb_erase(vma, &mm->mm_rb); mm->map_count--; tail_vma = vma; vma = vma->vm_next; @@ -3259,9 +3164,10 @@ void exit_mmap(struct mm_struct *mm) (void)__oom_reap_task_mm(mm); set_bit(MMF_OOM_SKIP, &mm->flags); + down_write(&mm->mmap_sem); + up_write(&mm->mmap_sem); } - down_write(&mm->mmap_sem); if (mm->locked_vm) { vma = mm->mmap; while (vma) { @@ -3274,11 +3180,8 @@ void exit_mmap(struct mm_struct *mm) arch_exit_mmap(mm); vma = mm->mmap; - if (!vma) { - /* Can happen if dup_mmap() received an OOM */ - up_write(&mm->mmap_sem);; + if (!vma) /* Can happen if dup_mmap() received an OOM */ return; - } lru_add_drain(); flush_cache_mm(mm); @@ -3289,14 +3192,16 @@ void exit_mmap(struct mm_struct *mm) free_pgtables(&tlb, vma, FIRST_USER_ADDRESS, USER_PGTABLES_CEILING); tlb_finish_mmu(&tlb, 0, -1); - /* Walk the list again, actually closing and freeing it. */ + /* + * Walk the list again, actually closing and freeing it, + * with preemption enabled, without holding any MM locks. + */ while (vma) { if (vma->vm_flags & VM_ACCOUNT) nr_accounted += vma_pages(vma); vma = remove_vma(vma); cond_resched(); } - up_write(&mm->mmap_sem); vm_unacct_memory(nr_accounted); } @@ -3363,21 +3268,9 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, if (find_vma_links(mm, addr, addr + len, &prev, &rb_link, &rb_parent)) return NULL; /* should never get here */ - - /* There is 3 cases to manage here in - * AAAA AAAA AAAA AAAA - * PPPP.... PPPP......NNNN PPPP....NNNN PP........NN - * PPPPPPPP(A) PPPP..NNNNNNNN(B) PPPPPPPPPPPP(1) NULL - * PPPPPPPPNNNN(2) - * PPPPNNNNNNNN(3) - * - * new_vma == prev in case A,1,2 - * new_vma == next in case B,3 - */ - new_vma = __vma_merge(mm, prev, addr, addr + len, vma->vm_flags, - vma->anon_vma, vma->vm_file, pgoff, - vma_policy(vma), vma->vm_userfaultfd_ctx, - vma_get_anon_name(vma), true); + new_vma = vma_merge(mm, prev, addr, addr + len, vma->vm_flags, + vma->anon_vma, vma->vm_file, pgoff, vma_policy(vma), + vma->vm_userfaultfd_ctx, vma_get_anon_name(vma)); if (new_vma) { /* * Source vma may have been merged into new_vma @@ -3415,15 +3308,6 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, get_file(new_vma->vm_file); if (new_vma->vm_ops && new_vma->vm_ops->open) new_vma->vm_ops->open(new_vma); - /* - * As the VMA is linked right now, it may be hit by the - * speculative page fault handler. But we don't want it to - * to start mapping page in this area until the caller has - * potentially move the pte from the moved VMA. To prevent - * that we protect it right now, and let the caller unprotect - * it once the move is done. - */ - vm_write_begin(new_vma); vma_link(mm, new_vma, prev, rb_link, rb_parent); *need_rmap_locks = false; } diff --git a/mm/mprotect.c b/mm/mprotect.c index c1e4a8da68b0..99cf8a97b2e4 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -455,14 +455,12 @@ success: * vm_flags and vm_page_prot are protected by the mmap_sem * held in write mode. */ - vm_write_begin(vma); - WRITE_ONCE(vma->vm_flags, vma_pad_fixup_flags(vma, newflags)); + vma->vm_flags = vma_pad_fixup_flags(vma, newflags); dirty_accountable = vma_wants_writenotify(vma, vma->vm_page_prot); vma_set_page_prot(vma); change_protection(vma, start, end, vma->vm_page_prot, dirty_accountable, 0); - vm_write_end(vma); /* * Private VM_LOCKED VMA becoming writable: trigger COW to avoid major diff --git a/mm/mremap.c b/mm/mremap.c index 4588c5d90884..8b1a5a6f7c50 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -210,38 +210,6 @@ static void move_ptes(struct vm_area_struct *vma, pmd_t *old_pmd, drop_rmap_locks(vma); } -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT -static inline bool trylock_vma_ref_count(struct vm_area_struct *vma) -{ - /* - * If we have the only reference, swap the refcount to -1. This - * will prevent other concurrent references by get_vma() for SPFs. - */ - return atomic_cmpxchg(&vma->vm_ref_count, 1, -1) == 1; -} - -/* - * Restore the VMA reference count to 1 after a fast mremap. - */ -static inline void unlock_vma_ref_count(struct vm_area_struct *vma) -{ - /* - * This should only be called after a corresponding, - * successful trylock_vma_ref_count(). - */ - VM_BUG_ON_VMA(atomic_cmpxchg(&vma->vm_ref_count, -1, 1) != -1, - vma); -} -#else /* !CONFIG_SPECULATIVE_PAGE_FAULT */ -static inline bool trylock_vma_ref_count(struct vm_area_struct *vma) -{ - return true; -} -static inline void unlock_vma_ref_count(struct vm_area_struct *vma) -{ -} -#endif /* CONFIG_SPECULATIVE_PAGE_FAULT */ - #ifdef CONFIG_HAVE_MOVE_PMD static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, unsigned long new_addr, unsigned long old_end, @@ -262,14 +230,6 @@ static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, if (WARN_ON(!pmd_none(*new_pmd))) return false; - /* - * We hold both exclusive mmap_lock and rmap_lock at this point and - * cannot block. If we cannot immediately take exclusive ownership - * of the VMA fallback to the move_ptes(). - */ - if (!trylock_vma_ref_count(vma)) - return false; - /* * We don't have to worry about the ordering of src and dst * ptlocks because exclusive mmap_sem prevents deadlock. @@ -292,7 +252,6 @@ static bool move_normal_pmd(struct vm_area_struct *vma, unsigned long old_addr, spin_unlock(new_ptl); spin_unlock(old_ptl); - unlock_vma_ref_count(vma); return true; } #else @@ -568,14 +527,6 @@ static unsigned long move_vma(struct vm_area_struct *vma, return -ENOMEM; } - /* new_vma is returned protected by copy_vma, to prevent speculative - * page fault to be done in the destination area before we move the pte. - * Now, we must also protect the source VMA since we don't want pages - * to be mapped in our back while we are copying the PTEs. - */ - if (vma != new_vma) - vm_write_begin(vma); - moved_len = move_page_tables(vma, old_addr, new_vma, new_addr, old_len, need_rmap_locks); if (moved_len < old_len) { @@ -592,8 +543,6 @@ static unsigned long move_vma(struct vm_area_struct *vma, */ move_page_tables(new_vma, new_addr, vma, old_addr, moved_len, true); - if (vma != new_vma) - vm_write_end(vma); vma = new_vma; old_len = new_len; old_addr = new_addr; @@ -602,10 +551,7 @@ static unsigned long move_vma(struct vm_area_struct *vma, mremap_userfaultfd_prep(new_vma, uf); arch_remap(mm, old_addr, old_addr + old_len, new_addr, new_addr + new_len); - if (vma != new_vma) - vm_write_end(vma); } - vm_write_end(new_vma); /* Conceal VM_ACCOUNT so old reservation is not undone */ if (vm_flags & VM_ACCOUNT && !(flags & MREMAP_DONTUNMAP)) { diff --git a/mm/rmap.c b/mm/rmap.c index 2b5886600b38..ba7e3fe2ae9b 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -1150,7 +1150,7 @@ void do_page_add_anon_rmap(struct page *page, } /** - * __page_add_new_anon_rmap - add pte mapping to a new anonymous page + * page_add_new_anon_rmap - add pte mapping to a new anonymous page * @page: the page to add the mapping to * @vma: the vm area in which the mapping is added * @address: the user virtual address mapped @@ -1160,11 +1160,12 @@ void do_page_add_anon_rmap(struct page *page, * This means the inc-and-test can be bypassed. * Page does not have to be locked. */ -void __page_add_new_anon_rmap(struct page *page, +void page_add_new_anon_rmap(struct page *page, struct vm_area_struct *vma, unsigned long address, bool compound) { int nr = compound ? hpage_nr_pages(page) : 1; + VM_BUG_ON_VMA(address < vma->vm_start || address >= vma->vm_end, vma); __SetPageSwapBacked(page); if (compound) { VM_BUG_ON_PAGE(!PageTransHuge(page), page); diff --git a/mm/shmem.c b/mm/shmem.c index 9ae6831e7b07..e2f72b220988 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2072,10 +2072,10 @@ static vm_fault_t shmem_fault(struct vm_fault *vmf) sgp = SGP_CACHE; - if ((vmf->vma_flags & VM_NOHUGEPAGE) || + if ((vma->vm_flags & VM_NOHUGEPAGE) || test_bit(MMF_DISABLE_THP, &vma->vm_mm->flags)) sgp = SGP_NOHUGE; - else if (vmf->vma_flags & VM_HUGEPAGE) + else if (vma->vm_flags & VM_HUGEPAGE) sgp = SGP_HUGE; err = shmem_getpage_gfp(inode, vmf->pgoff, &vmf->page, sgp, diff --git a/mm/swap.c b/mm/swap.c index 159174565497..1c5021b57e2c 100644 --- a/mm/swap.c +++ b/mm/swap.c @@ -452,12 +452,12 @@ void lru_cache_add(struct page *page) * directly back onto it's zone's unevictable list, it does NOT use a * per cpu pagevec. */ -void __lru_cache_add_active_or_unevictable(struct page *page, - unsigned long vma_flags) +void lru_cache_add_active_or_unevictable(struct page *page, + struct vm_area_struct *vma) { VM_BUG_ON_PAGE(PageLRU(page), page); - if (likely((vma_flags & (VM_LOCKED | VM_SPECIAL)) != VM_LOCKED)) + if (likely((vma->vm_flags & (VM_LOCKED | VM_SPECIAL)) != VM_LOCKED)) SetPageActive(page); else if (!TestSetPageMlocked(page)) { /* diff --git a/mm/swap_state.c b/mm/swap_state.c index 16709312e949..55b457c382b6 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -537,11 +537,7 @@ static unsigned long swapin_nr_pages(unsigned long offset) * This has been extended to use the NUMA policies from the mm triggering * the readahead. * - * Caller must hold down_read on the vma->vm_mm if vmf->vma is not NULL. - * This is needed to ensure the VMA will not be freed in our back. In the case - * of the speculative page fault handler, this cannot happen, even if we don't - * hold the mmap_sem. Callees are assumed to take care of reading VMA's fields - * using READ_ONCE() to read consistent values. + * Caller must hold read mmap_sem if vmf->vma is not NULL. */ struct page *swap_cluster_readahead(swp_entry_t entry, gfp_t gfp_mask, struct vm_fault *vmf) @@ -638,9 +634,9 @@ static inline void swap_ra_clamp_pfn(struct vm_area_struct *vma, unsigned long *start, unsigned long *end) { - *start = max3(lpfn, PFN_DOWN(READ_ONCE(vma->vm_start)), + *start = max3(lpfn, PFN_DOWN(vma->vm_start), PFN_DOWN(faddr & PMD_MASK)); - *end = min3(rpfn, PFN_DOWN(READ_ONCE(vma->vm_end)), + *end = min3(rpfn, PFN_DOWN(vma->vm_end), PFN_DOWN((faddr & PMD_MASK) + PMD_SIZE)); } diff --git a/mm/vmstat.c b/mm/vmstat.c index af290e42d497..10d281bfd71b 100644 --- a/mm/vmstat.c +++ b/mm/vmstat.c @@ -1300,11 +1300,7 @@ const char * const vmstat_text[] = { "swap_ra", "swap_ra_hit", #endif -#ifdef CONFIG_SPECULATIVE_PAGE_FAULT - "speculative_pgfault_anon", - "speculative_pgfault_file", -#endif -#endif /* CONFIG_VM_EVENT_COUNTERS */ +#endif /* CONFIG_VM_EVENTS_COUNTERS */ }; #endif /* CONFIG_PROC_FS || CONFIG_SYSFS || CONFIG_NUMA */