From fde15c0345f73f9c1e31aa6bfee1c5d316bc8906 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Thu, 9 May 2013 16:21:24 +0900 Subject: [PATCH 1/3] mm: Per process reclaim These day, there are many platforms available in the embedded market and they are smarter than kernel which has very limited information about working set so they want to involve memory management more heavily like android's lowmemory killer and ashmem or recent many lowmemory notifier. One of the simple imagine scenario about userspace's intelligence is that platform can manage tasks as forground and background so it would be better to reclaim background's task pages for end-user's *responsibility* although it has frequent referenced pages. This patch adds new knob "reclaim under proc//" so task manager can reclaim any target process anytime, anywhere. It could give another method to platform for using memory efficiently. It can avoid process killing for getting free memory, which was really terrible experience because I lost my best score of game I had ever after I switch the phone call while I enjoyed the game. Reclaim file-backed pages only. echo file > /proc/PID/reclaim Reclaim anonymous pages only. echo anon > /proc/PID/reclaim Reclaim all pages echo all > /proc/PID/reclaim Change-Id: Iabdb7bc2ef3dc4d94e3ea005fbe18f4cd06739ab Signed-off-by: Minchan Kim Patch-mainline: linux-mm @ 9 May 2013 16:21:24 [vinmenon@codeaurora.org: merge conflict fixes] Signed-off-by: Vinayak Menon --- fs/proc/base.c | 3 ++ fs/proc/internal.h | 1 + fs/proc/task_mmu.c | 119 +++++++++++++++++++++++++++++++++++++++++++ include/linux/rmap.h | 4 ++ mm/Kconfig | 14 +++++ mm/vmscan.c | 56 ++++++++++++++++++++ 6 files changed, 197 insertions(+) diff --git a/fs/proc/base.c b/fs/proc/base.c index a3e72c567844..fdc8e819015e 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -3237,6 +3237,9 @@ static const struct pid_entry tgid_base_stuff[] = { REG("mounts", S_IRUGO, proc_mounts_operations), REG("mountinfo", S_IRUGO, proc_mountinfo_operations), REG("mountstats", S_IRUSR, proc_mountstats_operations), +#ifdef CONFIG_PROCESS_RECLAIM + REG("reclaim", 0200, proc_reclaim_operations), +#endif #ifdef CONFIG_PROC_PAGE_MONITOR REG("clear_refs", S_IWUSR, proc_clear_refs_operations), REG("smaps", S_IRUGO, proc_pid_smaps_operations), diff --git a/fs/proc/internal.h b/fs/proc/internal.h index 95846552ef24..91ee4eb4a275 100644 --- a/fs/proc/internal.h +++ b/fs/proc/internal.h @@ -205,6 +205,7 @@ struct pde_opener { extern const struct inode_operations proc_link_inode_operations; extern const struct inode_operations proc_pid_link_inode_operations; extern const struct super_operations proc_sops; +extern const struct file_operations proc_reclaim_operations; void proc_init_kmemcache(void); void set_proc_pid_nlink(void); diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 1ce1cb08a5d3..5ab6450ba19c 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -19,6 +19,7 @@ #include #include #include +#include #include #include @@ -1708,6 +1709,124 @@ const struct file_operations proc_pagemap_operations = { }; #endif /* CONFIG_PROC_PAGE_MONITOR */ +#ifdef CONFIG_PROCESS_RECLAIM +static int reclaim_pte_range(pmd_t *pmd, unsigned long addr, + unsigned long end, struct mm_walk *walk) +{ + struct vm_area_struct *vma = walk->private; + pte_t *pte, ptent; + spinlock_t *ptl; + struct page *page; + LIST_HEAD(page_list); + int isolated; + + split_huge_pmd(vma, addr, pmd); + if (pmd_trans_unstable(pmd)) + return 0; +cont: + isolated = 0; + pte = pte_offset_map_lock(vma->vm_mm, pmd, addr, &ptl); + for (; addr != end; pte++, addr += PAGE_SIZE) { + ptent = *pte; + if (!pte_present(ptent)) + continue; + + page = vm_normal_page(vma, addr, ptent); + if (!page) + continue; + + if (isolate_lru_page(page)) + continue; + + list_add(&page->lru, &page_list); + inc_node_page_state(page, NR_ISOLATED_ANON + + page_is_file_cache(page)); + isolated++; + if (isolated >= SWAP_CLUSTER_MAX) + break; + } + pte_unmap_unlock(pte - 1, ptl); + reclaim_pages_from_list(&page_list); + if (addr != end) + goto cont; + + cond_resched(); + return 0; +} + +enum reclaim_type { + RECLAIM_FILE, + RECLAIM_ANON, + RECLAIM_ALL, + RECLAIM_RANGE, +}; + +static ssize_t reclaim_write(struct file *file, const char __user *buf, + size_t count, loff_t *ppos) +{ + struct task_struct *task; + char buffer[PROC_NUMBUF]; + struct mm_struct *mm; + struct vm_area_struct *vma; + enum reclaim_type type; + char *type_buf; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + + if (copy_from_user(buffer, buf, count)) + return -EFAULT; + + type_buf = strstrip(buffer); + if (!strcmp(type_buf, "file")) + type = RECLAIM_FILE; + else if (!strcmp(type_buf, "anon")) + type = RECLAIM_ANON; + else if (!strcmp(type_buf, "all")) + type = RECLAIM_ALL; + else + return -EINVAL; + + task = get_proc_task(file->f_path.dentry->d_inode); + if (!task) + return -ESRCH; + + mm = get_task_mm(task); + if (mm) { + const struct mm_walk_ops reclaim_walk_ops = { + .pmd_entry = reclaim_pte_range, + }; + + down_read(&mm->mmap_sem); + for (vma = mm->mmap; vma; vma = vma->vm_next) { + + if (is_vm_hugetlb_page(vma)) + continue; + + if (type == RECLAIM_ANON && vma->vm_file) + continue; + if (type == RECLAIM_FILE && !vma->vm_file) + continue; + + walk_page_range(mm, vma->vm_start, vma->vm_end, + &reclaim_walk_ops, vma); + } + flush_tlb_mm(mm); + up_read(&mm->mmap_sem); + mmput(mm); + } + put_task_struct(task); + + return count; +} + +const struct file_operations proc_reclaim_operations = { + .write = reclaim_write, + .llseek = noop_llseek, +}; +#endif + #ifdef CONFIG_NUMA struct numa_maps { diff --git a/include/linux/rmap.h b/include/linux/rmap.h index 988d176472df..caae95805da5 100644 --- a/include/linux/rmap.h +++ b/include/linux/rmap.h @@ -12,6 +12,10 @@ #include #include +extern int isolate_lru_page(struct page *page); +extern void putback_lru_page(struct page *page); +extern unsigned long reclaim_pages_from_list(struct list_head *page_list); + /* * The anon_vma heads a list of private "related" vmas, to scan if * an anonymous page pointing to this anon_vma needs to be unmapped: diff --git a/mm/Kconfig b/mm/Kconfig index 9fe6287ae035..35ecdc808844 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -832,3 +832,17 @@ config OOM_TASK_PRIORITY_ADJ_LIMIT before considering tasks with a lower oom_score_adj value. endmenu + +config PROCESS_RECLAIM + bool "Enable process reclaim" + depends on PROC_FS + depends on QGKI + default y + help + It allows to reclaim pages of the process by /proc/pid/reclaim. + + (echo file > /proc/PID/reclaim) reclaims file-backed pages only. + (echo anon > /proc/PID/reclaim) reclaims anonymous pages only. + (echo all > /proc/PID/reclaim) reclaims all pages. + + Any other value is ignored. diff --git a/mm/vmscan.c b/mm/vmscan.c index e974f288f569..6b5a5fa76f35 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1565,6 +1565,62 @@ unsigned long reclaim_clean_pages_from_list(struct zone *zone, return ret; } +#ifdef CONFIG_PROCESS_RECLAIM +static unsigned long shrink_page(struct page *page, + struct zone *zone, + struct scan_control *sc, + enum ttu_flags ttu_flags, + bool force_reclaim, + struct list_head *ret_pages) +{ + int reclaimed; + LIST_HEAD(page_list); + + list_add(&page->lru, &page_list); + + reclaimed = shrink_page_list(&page_list, zone->zone_pgdat, sc, + ttu_flags, NULL, force_reclaim); + if (!reclaimed) + list_splice(&page_list, ret_pages); + + return reclaimed; +} + +unsigned long reclaim_pages_from_list(struct list_head *page_list) +{ + struct scan_control sc = { + .gfp_mask = GFP_KERNEL, + .priority = DEF_PRIORITY, + .may_writepage = 1, + .may_unmap = 1, + .may_swap = 1, + }; + + LIST_HEAD(ret_pages); + struct page *page; + unsigned long nr_reclaimed = 0; + + while (!list_empty(page_list)) { + page = lru_to_page(page_list); + list_del(&page->lru); + + ClearPageActive(page); + nr_reclaimed += shrink_page(page, page_zone(page), &sc, + TTU_IGNORE_ACCESS, true, &ret_pages); + } + + while (!list_empty(&ret_pages)) { + page = lru_to_page(&ret_pages); + list_del(&page->lru); + dec_node_page_state(page, NR_ISOLATED_ANON + + page_is_file_cache(page)); + putback_lru_page(page); + } + + return nr_reclaimed; +} +#endif + /* * Attempt to remove the specified page from its LRU. Only take this page * if it is of the appropriate PageActive status. Pages which are being From f99a8d7a20cc3cdcae03c42b2a749b402b0bf7a6 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Thu, 9 May 2013 16:21:25 +0900 Subject: [PATCH 2/3] mm: make shrink_page_list with pages work from multiple zones Shrink_page_list expects all pages come from a same zone but it's too limited to use. This patch removes the dependency so next patch can use shrink_page_list with pages from multiple zones. Change-Id: I34469b7f0a79f2b79e30e40033ba8b3e1dd5f2d0 Signed-off-by: Minchan Kim Patch-mainline: linux-mm @ 9 May 2013 16:21:25 [vinmenon@codeaurora.org: changes for node based lrus] Signed-off-by: Vinayak Menon --- mm/vmscan.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 6b5a5fa76f35..098513613142 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1152,6 +1152,8 @@ static unsigned long shrink_page_list(struct list_head *page_list, goto keep; VM_BUG_ON_PAGE(PageActive(page), page); + if (pgdat) + VM_BUG_ON_PAGE(page_pgdat(page) != pgdat, page); nr_pages = compound_nr(page); @@ -1238,7 +1240,8 @@ static unsigned long shrink_page_list(struct list_head *page_list, /* Case 1 above */ if (current_is_kswapd() && PageReclaim(page) && - test_bit(PGDAT_WRITEBACK, &pgdat->flags)) { + (pgdat && + test_bit(PGDAT_WRITEBACK, &pgdat->flags))) { stat->nr_immediate++; goto activate_locked; @@ -1372,7 +1375,8 @@ static unsigned long shrink_page_list(struct list_head *page_list, */ if (page_is_file_cache(page) && (!current_is_kswapd() || !PageReclaim(page) || - !test_bit(PGDAT_DIRTY, &pgdat->flags))) { + (pgdat && + !test_bit(PGDAT_DIRTY, &pgdat->flags)))) { /* * Immediately reclaim when written back. * Similar in principal to deactivate_page() From d0456e9e94e1843dece57b2369c1dfa1ee57ac68 Mon Sep 17 00:00:00 2001 From: Minchan Kim Date: Thu, 9 May 2013 16:21:26 +0900 Subject: [PATCH 3/3] mm: Remove shrink_page By previous patch, shrink_page_list can handle pages from multiple zone so let's remove shrink_page. Change-Id: I3526377aa6ee6142b8f3ec63396e7ada1e442505 Signed-off-by: Minchan Kim Patch-mainline: linux-mm @ 22 Apr 2013 17:45:03 [vinmenon@codeaurora.org: trivial merge conflict fixes] Signed-off-by: Vinayak Menon --- mm/vmscan.c | 46 +++++++++++++++------------------------------- 1 file changed, 15 insertions(+), 31 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 098513613142..258f345c03c5 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1499,6 +1499,13 @@ free_it: (*get_compound_page_dtor(page))(page); else list_add(&page->lru, &free_pages); + /* + * If pagelist are from multiple nodes, we should decrease + * NR_ISOLATED_ANON + x on freed pages in here. + */ + if (!pgdat) + dec_node_page_state(page, NR_ISOLATED_ANON + + page_is_file_cache(page)); continue; activate_locked_split: @@ -1570,26 +1577,6 @@ unsigned long reclaim_clean_pages_from_list(struct zone *zone, } #ifdef CONFIG_PROCESS_RECLAIM -static unsigned long shrink_page(struct page *page, - struct zone *zone, - struct scan_control *sc, - enum ttu_flags ttu_flags, - bool force_reclaim, - struct list_head *ret_pages) -{ - int reclaimed; - LIST_HEAD(page_list); - - list_add(&page->lru, &page_list); - - reclaimed = shrink_page_list(&page_list, zone->zone_pgdat, sc, - ttu_flags, NULL, force_reclaim); - if (!reclaimed) - list_splice(&page_list, ret_pages); - - return reclaimed; -} - unsigned long reclaim_pages_from_list(struct list_head *page_list) { struct scan_control sc = { @@ -1600,22 +1587,19 @@ unsigned long reclaim_pages_from_list(struct list_head *page_list) .may_swap = 1, }; - LIST_HEAD(ret_pages); + unsigned long nr_reclaimed; + struct reclaim_stat stat; struct page *page; - unsigned long nr_reclaimed = 0; + + list_for_each_entry(page, page_list, lru) + ClearPageActive(page); + + nr_reclaimed = shrink_page_list(page_list, NULL, &sc, + TTU_IGNORE_ACCESS, &stat, true); while (!list_empty(page_list)) { page = lru_to_page(page_list); list_del(&page->lru); - - ClearPageActive(page); - nr_reclaimed += shrink_page(page, page_zone(page), &sc, - TTU_IGNORE_ACCESS, true, &ret_pages); - } - - while (!list_empty(&ret_pages)) { - page = lru_to_page(&ret_pages); - list_del(&page->lru); dec_node_page_state(page, NR_ISOLATED_ANON + page_is_file_cache(page)); putback_lru_page(page);