diff --git a/drivers/moto_mm/Android.mk b/drivers/moto_mm/Android.mk new file mode 100644 index 000000000000..d142c23bd9c9 --- /dev/null +++ b/drivers/moto_mm/Android.mk @@ -0,0 +1,12 @@ +DLKM_DIR := motorola/kernel/modules +LOCAL_PATH := $(call my-dir) + + +include $(CLEAR_VARS) +LOCAL_MODULE := moto_mm.ko +LOCAL_MODULE_TAGS := optional +LOCAL_MODULE_PATH := $(KERNEL_MODULES_OUT) +KBUILD_OPTIONS_GKI += GKI_OBJ_MODULE_DIR=gki + +include $(DLKM_DIR)/AndroidKernelModule.mk + diff --git a/drivers/moto_mm/Kbuild b/drivers/moto_mm/Kbuild new file mode 100644 index 000000000000..7aeed6cf3254 --- /dev/null +++ b/drivers/moto_mm/Kbuild @@ -0,0 +1,6 @@ +# add -Wall to try to catch everything we can. +EXTRA_CFLAGS += -Wall +EXTRA_CFLAGS += -I$(ANDROID_BUILD_TOP)/motorola/kernel/modules/include + +obj-m += moto_mm.o + diff --git a/drivers/moto_mm/Kconfig b/drivers/moto_mm/Kconfig new file mode 100644 index 000000000000..f66554cd5c45 --- /dev/null +++ b/drivers/moto_mm/Kconfig @@ -0,0 +1 @@ +# SPDX-License-Identifier: GPL-2.0 diff --git a/drivers/moto_mm/Makefile b/drivers/moto_mm/Makefile new file mode 100755 index 000000000000..bfbdc14740ef --- /dev/null +++ b/drivers/moto_mm/Makefile @@ -0,0 +1,14 @@ +all: modules + +modules: + $(MAKE) -C $(KERNEL_SRC) M=$(M) modules $(KBUILD_OPTIONS) + +modules_install: + $(MAKE) INSTALL_MOD_STRIP=1 -C $(KERNEL_SRC) M=$(M) modules_install + +%: + $(MAKE) -C $(KERNEL_SRC) M=$(M) $@ $(KBUILD_OPTIONS) + +clean: + rm -f *.o *.ko *.mod.c *.mod.o *~ .*.cmd Module.symvers + rm -rf .tmp_versions diff --git a/drivers/moto_mm/moto_mm.c b/drivers/moto_mm/moto_mm.c new file mode 100755 index 000000000000..7a3b4065f28c --- /dev/null +++ b/drivers/moto_mm/moto_mm.c @@ -0,0 +1,89 @@ +/* + * Copyright (C) 2023 Motorola Mobility LLC + * + * This software is licensed under the terms of the GNU General Public + * License version 2, as published by the Free Software Foundation, and + * may be copied, distributed, and modified under those terms. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + */ + +#define pr_fmt(fmt) "moto_mm: " fmt + +#include +#include +#include +#include + + +static void tune_inactive_ratio_hook(void *data, unsigned long *inactive_ratio, int file) +{ + if (file) + *inactive_ratio = min(2UL, *inactive_ratio); + else + *inactive_ratio = 1; + + return; +} + + +#define REGISTER_HOOK(name) do {\ + rc = register_trace_android_vh_##name(name##_hook, NULL);\ + if (rc) {\ + pr_err("register hook %s failed", #name);\ + goto err_out_##name;\ + }\ +} while (0) + +#define UNREGISTER_HOOK(name) do {\ + unregister_trace_android_vh_##name(name##_hook, NULL);\ +} while (0) + +#define ERROR_OUT(name) err_out_##name + +static int register_all_hooks(void) +{ + int rc; + + /* tune_inactive_ratio_hook */ + REGISTER_HOOK(tune_inactive_ratio); + return 0; + + UNREGISTER_HOOK(tune_inactive_ratio); +ERROR_OUT(tune_inactive_ratio): + return rc; +} + +static void unregister_all_hook(void) +{ + UNREGISTER_HOOK(tune_inactive_ratio); +} + +static int __init moto_mm_init(void) +{ + int ret = 0; + + ret = register_all_hooks(); + if (ret != 0) { + return ret; + } + + pr_info("moto_mm_init succeed!\n"); + return 0; +} + +static void __exit moto_mm_exit(void) +{ + unregister_all_hook(); + + pr_info("moto_mm_exit succeed!\n"); + return; +} + +module_init(moto_mm_init); +module_exit(moto_mm_exit); +MODULE_DESCRIPTION("Motorola mm optimizations driver"); +MODULE_LICENSE("GPL v2"); diff --git a/drivers/moto_swap/Android.mk b/drivers/moto_swap/Android.mk new file mode 100644 index 000000000000..9ca784a20e67 --- /dev/null +++ b/drivers/moto_swap/Android.mk @@ -0,0 +1,17 @@ +DLKM_DIR := motorola/kernel/modules +LOCAL_PATH := $(call my-dir) + +KERNEL_CFLAGS += CONFIG_HYBRIDSWAP_ZRAM=y +#KERNEL_CFLAGS += CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK=y +#KERNEL_CFLAGS += CONFIG_HYBRIDSWAP_ASYNC_COMPRESS=y +KERNEL_CFLAGS += CONFIG_HYBRIDSWAP=y +KERNEL_CFLAGS += CONFIG_HYBRIDSWAP_SWAPD=y +KERNEL_CFLAGS += CONFIG_HYBRIDSWAP_CORE=y + +include $(CLEAR_VARS) +LOCAL_MODULE := moto_swap.ko +LOCAL_MODULE_TAGS := optional +LOCAL_MODULE_PATH := $(KERNEL_MODULES_OUT) +KBUILD_OPTIONS_GKI += GKI_OBJ_MODULE_DIR=gki +include $(DLKM_DIR)/AndroidKernelModule.mk + diff --git a/drivers/moto_swap/Kbuild b/drivers/moto_swap/Kbuild new file mode 100644 index 000000000000..e2812fe5b538 --- /dev/null +++ b/drivers/moto_swap/Kbuild @@ -0,0 +1,45 @@ +HAVE_KERNEL_5_4 = $(shell test -d $(ANDROID_BUILD_TOP)/kernel/msm-5.4 && echo 1) + +ifeq ($(HAVE_KERNEL_5_4),1) +ZRAM_SRC = zram-5.4 +EXTRA_CFLAGS += -DCONFIG_ZRAM_5_4 +else +ZRAM_SRC = zram-5.10 +endif + +# add -Wall to try to catch everything we can. +EXTRA_CFLAGS += -Wall +EXTRA_CFLAGS += -I$(ANDROID_BUILD_TOP)/motorola/kernel/modules/include + +ifneq ($(filter m y,$(CONFIG_HYBRIDSWAP_ZRAM)),) +EXTRA_CFLAGS += -DCONFIG_HYBRIDSWAP_ZRAM +endif + +ifneq ($(filter m y,$(CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK)),) +EXTRA_CFLAGS += -DCONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +endif + +ifneq ($(filter m y,$(CONFIG_HYBRIDSWAP)),) +EXTRA_CFLAGS += -DCONFIG_HYBRIDSWAP +endif + +ifneq ($(filter m y,$(CONFIG_HYBRIDSWAP_SWAPD)),) +EXTRA_CFLAGS += -DCONFIG_HYBRIDSWAP_SWAPD +endif + +ifneq ($(filter m y,$(CONFIG_HYBRIDSWAP_CORE)),) +EXTRA_CFLAGS += -DCONFIG_HYBRIDSWAP_CORE +endif + +moto_swap-objs += $(ZRAM_SRC)/zram_drv.o +moto_swap-objs += hybridswap/hybridswap_main.o +moto_swap-objs += hybridswap/hybridswap_eswap.o + +ifneq ($(filter m y,$(CONFIG_HYBRIDSWAP_ASYNC_COMPRESS)),) +moto_swap-objs += hybridswap/hybridswap_akcompress.o +endif + +moto_swap-objs += hybridswap/hybridswap_swapd.o +moto_swap-objs += $(ZRAM_SRC)/zcomp.o + +obj-m += moto_swap.o diff --git a/drivers/moto_swap/Kconfig b/drivers/moto_swap/Kconfig new file mode 100644 index 000000000000..7b58f3946dd8 --- /dev/null +++ b/drivers/moto_swap/Kconfig @@ -0,0 +1,78 @@ +# SPDX-License-Identifier: GPL-2.0 +config HYBRIDSWAP_ZRAM + tristate "Compressed RAM block device support" + depends on BLOCK && SYSFS && ZSMALLOC && CRYPTO && !ZRAM + select CRYPTO_LZO + help + Creates virtual block devices called /dev/zramX (X = 0, 1, ...). + Pages written to these disks are compressed and stored in memory + itself. These disks allow very fast I/O and compression provides + good amounts of memory savings. + + It has several use cases, for example: /tmp storage, use as swap + disks and maybe many more. + + See Documentation/admin-guide/blockdev/zram.rst for more information. + +config HYBRIDSWAP_ZRAM_WRITEBACK + bool "Write back incompressible or idle page to backing device" + depends on HYBRIDSWAP_ZRAM + help + With incompressible page, there is no memory saving to keep it + in memory. Instead, write it out to backing device. + For this feature, admin should set up backing device via + /sys/block/zramX/backing_dev. + + With /sys/block/zramX/{idle,writeback}, application could ask + idle page's writeback to the backing device to save in memory. + + See Documentation/admin-guide/blockdev/zram.rst for more information. + +config HYBRIDSWAP_ZRAM_MEMORY_TRACKING + bool "Track zRam block status" + depends on HYBRIDSWAP_ZRAM && DEBUG_FS + help + With this feature, admin can track the state of allocated blocks + of zRAM. Admin could see the information via + /sys/kernel/debug/zram/zramX/block_state. + + See Documentation/admin-guide/blockdev/zram.rst for more information. + +config HYBRIDSWAP + bool "Enable Hybridswap" + depends on MEMCG && HYBRIDSWAP_ZRAM && !HYBRIDSWAP_ZRAM_WRITEBACK + default y + help + Hybridswap is a intelligent memory management solution. + +config HYBRIDSWAP_SWAPD + bool "Enable hybridswap swapd thread to reclaim anon pages in background" + default y + depends on HYBRIDSWAP + help + swapd is a kernel thread that reclaim anonymous pages in the + background. When the use of swap pages reaches the watermark + and the refault of anonymous pages is high, the content of + zram will exchanged to eswap by a certain percentage. + +# Selected when system need hybridswap container +config HYBRIDSWAP_CORE + bool "Hybridswap container device support" + depends on HYBRIDSWAP_ZRAM && HYBRIDSWAP + default y + help + Say Y here if you want to use the hybridswap + as the backend device in ZRAM. + If unsure, say N here. + This module can't be compiled as a module, + the module is as one part of the ZRAM driver. + +config HYBRIDSWAP_ASYNC_COMPRESS + bool "hypbridswap support asynchronous compress anon pages" + depends on HYBRIDSWAP_ZRAM && HYBRIDSWAP + default n + help + Say Y here if you want to create asynchronous thread + for compress anon pages. + If unsure, say N here. + This feature will reduce the kswapd cpu load. diff --git a/drivers/moto_swap/Makefile b/drivers/moto_swap/Makefile new file mode 100644 index 000000000000..bfbdc14740ef --- /dev/null +++ b/drivers/moto_swap/Makefile @@ -0,0 +1,14 @@ +all: modules + +modules: + $(MAKE) -C $(KERNEL_SRC) M=$(M) modules $(KBUILD_OPTIONS) + +modules_install: + $(MAKE) INSTALL_MOD_STRIP=1 -C $(KERNEL_SRC) M=$(M) modules_install + +%: + $(MAKE) -C $(KERNEL_SRC) M=$(M) $@ $(KBUILD_OPTIONS) + +clean: + rm -f *.o *.ko *.mod.c *.mod.o *~ .*.cmd Module.symvers + rm -rf .tmp_versions diff --git a/drivers/moto_swap/hybridswap/hybridswap.h b/drivers/moto_swap/hybridswap/hybridswap.h new file mode 100644 index 000000000000..2fa4e13d7ad5 --- /dev/null +++ b/drivers/moto_swap/hybridswap/hybridswap.h @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (C) 2020-2022 Oplus. All rights reserved. + */ + +#ifndef HYBRIDSWAP_H +#define HYBRIDSWAP_H +extern int __init hybridswap_pre_init(void); +extern ssize_t hybridswap_vmstat_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_loglevel_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_loglevel_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_enable_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_enable_show(struct device *dev, + struct device_attribute *attr, char *buf); +#ifdef CONFIG_HYBRIDSWAP_CORE +extern void hybridswap_record(struct zram *zram, u32 index, struct mem_cgroup *memcg); +extern void hybridswap_untrack(struct zram *zram, u32 index); +extern int hybridswap_page_fault(struct zram *zram, u32 index); +extern bool hybridswap_delete(struct zram *zram, u32 index); + +extern ssize_t hybridswap_report_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_stat_snap_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_meminfo_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_core_enable_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_core_enable_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_loop_device_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_loop_device_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_dev_life_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_dev_life_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_quota_day_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_quota_day_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern ssize_t hybridswap_zram_increase_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_zram_increase_show(struct device *dev, + struct device_attribute *attr, char *buf); +#endif + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +/* 63---48,47--32,31-0 : cgroup id, thread_index, index*/ +#define ZRAM_INDEX_SHIFT 32 +#define CACHE_INDEX_SHIFT 32 +#define CACHE_INDEX_MASK ((1llu << CACHE_INDEX_SHIFT) - 1) +#define ZRAM_INDEX_MASK ((1llu << ZRAM_INDEX_SHIFT) - 1) + +#define cache_index_val(index) (((unsigned long)index & CACHE_INDEX_MASK) << ZRAM_INDEX_SHIFT) +#define zram_index_val(id) ((unsigned long)id & ZRAM_INDEX_MASK) +#define mk_page_val(cache_index, index) (cache_index_val(cache_index) | zram_index_val(index)) + +#define fetch_cache_id(page) ((page->private >> 32) & CACHE_INDEX_MASK) +#define fetch_zram_index(page) (page->private & ZRAM_INDEX_MASK) + +#define zram_set_page(zram, index, page) (zram->table[index].page = page) +#define zram_fetch_page(zram, index) (zram->table[index].page) + +extern void del_page_from_cache(struct page *page); +extern int add_anon_page2cache(struct zram * zram, u32 index, + struct page *page); +extern ssize_t hybridswap_akcompress_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_akcompress_show(struct device *dev, + struct device_attribute *attr, char *buf); +extern void put_free_page(struct page *page); +extern void put_anon_pages(struct page *page); +extern int akcompress_cache_page_fault(struct zram *zram, + struct page *page, u32 index); +extern void destroy_akcompressd_task(struct zram *zram); +#endif + +#ifdef CONFIG_HYBRIDSWAP_SWAPD +extern ssize_t hybridswap_swapd_pause_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len); +extern ssize_t hybridswap_swapd_pause_show(struct device *dev, + struct device_attribute *attr, char *buf); +#endif +static inline bool current_is_mswapd(void) +{ +#ifdef CONFIG_HYBRIDSWAP_SWAPD + return (strncmp(current->comm, "mswapd:", sizeof("mswapd:") - 1) == 0); +#else + return false; +#endif +} +#endif /* HYBRIDSWAP_H */ diff --git a/drivers/moto_swap/hybridswap/hybridswap_akcompress.c b/drivers/moto_swap/hybridswap/hybridswap_akcompress.c new file mode 100644 index 000000000000..265924e91a45 --- /dev/null +++ b/drivers/moto_swap/hybridswap/hybridswap_akcompress.c @@ -0,0 +1,580 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (C) 2020-2022 Oplus. All rights reserved. + */ + +#define pr_fmt(fmt) "moto_swap: " fmt + +#include +#include +#include +#include +#include + +#ifdef CONFIG_ZRAM_5_4 +#include "../zram-5.4/zram_drv.h" +#include "../zram-5.4/zram_drv_internal.h" +#else +#include "../zram-5.10/zram_drv.h" +#include "../zram-5.10/zram_drv_internal.h" +#endif +#include "hybridswap_internal.h" +#include "hybridswap.h" + +struct compress_info_s { + struct list_head free_page_head; + spinlock_t free_lock; + unsigned int free_cnt; + + unsigned int max_cnt; +} compress_info; + +#define MAX_AKCOMPRESSD_THREADS 4 +#define DEFAULT_CACHE_SIZE_MB 64 +#define DEFAULT_COMPRESS_BATCH_MB 1 +#define DEFAULT_CACHE_COUNT ((DEFAULT_CACHE_SIZE_MB << 20) >> PAGE_SHIFT) +#define WAKEUP_AKCOMPRESSD_WATERMARK ((DEFAULT_COMPRESS_BATCH_MB << 20) >> PAGE_SHIFT) + +static wait_queue_head_t akcompressd_wait; +static struct task_struct *akc_task[MAX_AKCOMPRESSD_THREADS]; +static atomic64_t akc_cnt[MAX_AKCOMPRESSD_THREADS]; +static int akcompressd_threads = 0; +static atomic64_t cached_cnt; +static struct zram *zram_info; +static DEFINE_MUTEX(akcompress_init_lock); + +struct idr cached_idr = IDR_INIT(cached_idr); +DEFINE_SPINLOCK(cached_idr_lock); + +static void wake_all_akcompressd(void); + +void clear_page_memcg(struct cgroup_cache_page *cache) +{ + struct list_head *pos; + struct page *page; + + spin_lock(&cache->lock); + if (list_empty(&cache->head)) + goto out; + + list_for_each(pos, &cache->head) { + page = list_entry(pos, struct page, lru); + if (!page->mem_cgroup) + BUG(); + page->mem_cgroup = NULL; + } + +out: + cache->dead = 1; + spin_unlock(&cache->lock); +} + +static inline struct page * fetch_free_page(void) +{ + struct page *page = NULL; + + spin_lock(&compress_info.free_lock); + if (compress_info.free_cnt > 0) { + if (list_empty(&compress_info.free_page_head)) + BUG(); + page = lru_to_page(&compress_info.free_page_head); + list_del(&page->lru); + compress_info.free_cnt--; + } + spin_unlock(&compress_info.free_lock); + + return page; +} + +void put_free_page(struct page *page) +{ + set_page_private(page, 0); + spin_lock(&compress_info.free_lock); + list_add_tail(&page->lru, &compress_info.free_page_head); + compress_info.free_cnt++; + spin_unlock(&compress_info.free_lock); +} + +static inline struct cgroup_cache_page *find_and_fetch_memcg_cache(int cache_id) +{ + struct cgroup_cache_page *cache; + + spin_lock(&cached_idr_lock); + cache = (struct cgroup_cache_page *)idr_find(&cached_idr, cache_id); + if (unlikely(!cache)) { + spin_unlock(&cached_idr_lock); + pr_err("cache_id %d cache not find.\n", cache_id); + + return NULL; + } + fetch_memcg_cache(container_of(cache, memcg_hybs_t, cache)); + spin_unlock(&cached_idr_lock); + + return cache; +} + +void del_page_from_cache(struct page *page) +{ + int cache_id; + struct cgroup_cache_page *cache; + + if (!page) + return; + + cache_id = fetch_cache_id(page); + if (unlikely(cache_id < 0 || cache_id > MEM_CGROUP_ID_MAX)) { + hybp(HYB_ERR, "page %p cache_id %d index %u is invalid.\n", + page, cache_id, fetch_zram_index(page)); + return; + } + + cache = find_and_fetch_memcg_cache(cache_id); + if (!cache) + return; + + spin_lock(&cache->lock); + list_del(&page->lru); + cache->cnt--; + spin_unlock(&cache->lock); + put_memcg_cache(container_of(cache, memcg_hybs_t, cache)); + atomic64_dec(&cached_cnt); +} + +void del_page_from_cache_with_cache(struct page *page, + struct cgroup_cache_page *cache) +{ + spin_lock(&cache->lock); + list_del(&page->lru); + cache->cnt--; + spin_unlock(&cache->lock); + atomic64_dec(&cached_cnt); +} + +void put_anon_pages(struct page *page) +{ + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(page->mem_cgroup); + + spin_lock(&hybs->cache.lock); + list_add(&page->lru, &hybs->cache.head); + hybs->cache.cnt++; + spin_unlock(&hybs->cache.lock); +} + +static inline bool can_stop_working(struct cgroup_cache_page *cache, int index) +{ + spin_lock(&cache->lock); + if (unlikely(!list_empty(&cache->head))) { + spin_unlock(&cache->lock); + return false; + } + spin_unlock(&cache->lock); + return 1; +} + +static int check_cache_state(struct cgroup_cache_page *cache) +{ + if (cache->cnt == 0 || cache->compressing == 1) + return 0; + + spin_lock(&cache->lock); + if (cache->cnt == 0 || cache->compressing) { + spin_unlock(&cache->lock); + return 0; + } + cache->compressing = 1; + spin_unlock(&cache->lock); + fetch_memcg_cache(container_of(cache, memcg_hybs_t, cache)); + return 1; +} + +struct cgroup_cache_page *fetch_one_cache(void) +{ + struct cgroup_cache_page *cache = NULL; + int id; + + spin_lock(&cached_idr_lock); + idr_for_each_entry(&cached_idr, cache, id) { + if (check_cache_state(cache)) + break; + } + spin_unlock(&cached_idr_lock); + + return cache; +} + +void mark_compressing_stop(struct cgroup_cache_page *cache) +{ + spin_lock(&cache->lock); + if (cache->dead) + hybp(HYB_WARN, "stop compressing, may be cgroup is delelted\n"); + cache->compressing = 0; + spin_unlock(&cache->lock); + put_memcg_cache(container_of(cache, memcg_hybs_t, cache)); +} + +static inline struct page *fetch_anon_page(struct zram *zram, + struct cgroup_cache_page *cache) +{ + struct page *page, *prev_page; + int index; + + if (compress_info.free_cnt == 0) + return NULL; + + prev_page = NULL; +try_again: + page = NULL; + + spin_lock(&cache->lock); + if (!list_empty(&cache->head)) { + page = lru_to_page(&cache->head); + index = fetch_zram_index(page); + } + spin_unlock(&cache->lock); + + if (page) { + if (prev_page && (page == prev_page)) { + hybp(HYB_ERR, "zram %p index %d page %p\n", + zram, index, page); + BUG(); + } + + zram_slot_lock(zram, index); + if (!zram_test_flag(zram, index, ZRAM_CACHED)) { + zram_slot_unlock(zram, index); + prev_page = page; + goto try_again; + } + + prev_page = NULL; + zram_clear_flag(zram, index, ZRAM_CACHED); + del_page_from_cache_with_cache(page, cache); + zram_set_flag(zram, index, ZRAM_CACHED_COMPRESS); + zram_slot_unlock(zram, index); + } + + return page; +} + +int add_anon_page2cache(struct zram * zram, u32 index, struct page *page) +{ + struct page *dst_page; + void *src, *dst; + struct mem_cgroup *memcg; + struct cgroup_cache_page *cache; + memcg_hybs_t *hybs; + + if (akcompressd_threads == 0) + return 0; + + memcg = page->mem_cgroup; + if (!memcg || !MEMCGRP_ITEM_DATA(memcg)) + return 0; + + hybs = MEMCGRP_ITEM_DATA(memcg); + cache = &hybs->cache; + if (find_and_fetch_memcg_cache(cache->id) != cache) + return 0; + + spin_lock(&cache->lock); + if (cache->dead == 1) { + spin_unlock(&cache->lock); + return 0; + } + spin_unlock(&cache->lock); + + dst_page = fetch_free_page(); + if (!dst_page) + return 0; + + src = kmap_atomic(page); + dst = kmap_atomic(dst_page); + memcpy(dst, src, PAGE_SIZE); + kunmap_atomic(src); + kunmap_atomic(dst); + + dst_page->mem_cgroup = memcg; + set_page_private(dst_page, mk_page_val(cache->id, index)); + update_zram_index(zram, index, (unsigned long)dst_page); + atomic64_inc(&cached_cnt); + wake_all_akcompressd(); + hybp(HYB_DEBUG, "add_anon_page2cache index %u page %p passed\n", + index, dst_page); + return 1; +} + +static inline void akcompressd_try_to_sleep(wait_queue_head_t *waitq) +{ + DEFINE_WAIT(wait); + + prepare_to_wait(waitq, &wait, TASK_INTERRUPTIBLE); + freezable_schedule(); + finish_wait(waitq, &wait); +} + +static int akcompressd_func(void *data) +{ + struct page *page; + int ret, thread_index; + struct list_head compress_fail_list; + struct cgroup_cache_page *cache = NULL; + + thread_index = (int)data; + if (thread_index < 0 || thread_index >= MAX_AKCOMPRESSD_THREADS) { + hybp(HYB_ERR, "akcompress task index %d is invalid.\n", thread_index); + return -EINVAL; + } + + set_freezable(); + while (!kthread_should_stop()) { + akcompressd_try_to_sleep(&akcompressd_wait); + count_swapd_event(AKCOMPRESSD_WAKEUP); + + cache = fetch_one_cache(); + if (!cache) + continue; + +finish_last_jobs: + INIT_LIST_HEAD(&compress_fail_list); + page = fetch_anon_page(zram_info, cache); + while (page) { + ret = async_compress_page(zram_info, page); + put_memcg_cache(container_of(cache, memcg_hybs_t, cache)); + + if (ret) + list_add(&page->lru, &compress_fail_list); + else { + atomic64_inc(&akc_cnt[thread_index]); + page->mem_cgroup = NULL; + put_free_page(page); + } + page = fetch_anon_page(zram_info, cache); + } + + if (!list_empty(&compress_fail_list)) + hybp(HYB_ERR, "have some compress failed pages.\n"); + + if (kthread_should_stop()) { + if (!can_stop_working(cache, thread_index)) + goto finish_last_jobs; + } + mark_compressing_stop(cache); + } + + return 0; +} + +static int update_akcompressd_threads(int thread_count, struct zram *zram) +{ + int drop, increase; + int last_index, start_index, hid; + static DEFINE_MUTEX(update_lock); + + if (thread_count < 0 || thread_count > MAX_AKCOMPRESSD_THREADS) { + hybp(HYB_ERR, "thread_count %d is invalid\n", thread_count); + return -EINVAL; + } + + mutex_lock(&update_lock); + if (!zram_info || zram_info != zram) + zram_info = zram; + + if (thread_count == akcompressd_threads) { + mutex_unlock(&update_lock); + return thread_count; + } + + last_index = akcompressd_threads - 1; + if (thread_count < akcompressd_threads) { + drop = akcompressd_threads - thread_count; + for (hid = last_index; hid > (last_index - drop); hid--) { + if (akc_task[hid]) { + kthread_stop(akc_task[hid]); + akc_task[hid] = NULL; + } + } + } else { + increase = thread_count - akcompressd_threads; + start_index = last_index + 1; + for (hid = start_index; hid < (start_index + increase); hid++) { + if (unlikely(akc_task[hid])) + BUG(); + akc_task[hid]= kthread_run(akcompressd_func, + (void*)(unsigned long)hid, "akcompressd:%d", hid); + if (IS_ERR(akc_task[hid])) { + pr_err("Failed to start akcompressd%d\n", hid); + akc_task[hid] = NULL; + break; + } + } + } + + hybp(HYB_INFO, "akcompressd_threads count changed, old:%d new:%d\n", + akcompressd_threads, thread_count); + akcompressd_threads = thread_count; + mutex_unlock(&update_lock); + + return thread_count; +} + +static void wake_all_akcompressd(void) +{ + if (atomic64_read(&cached_cnt) < WAKEUP_AKCOMPRESSD_WATERMARK) + return; + + if (!waitqueue_active(&akcompressd_wait)) + return; + + wake_up_interruptible(&akcompressd_wait); +} + +int create_akcompressd_task(struct zram *zram) +{ + return update_akcompressd_threads(1, zram) != 1; +} + +void destroy_akcompressd_task(struct zram *zram) +{ + (void)update_akcompressd_threads(0, zram); +} + +ssize_t hybridswap_akcompress_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned long val; + struct zram *zram = dev_to_zram(dev); + + ret = kstrtoul(buf, 0, &val); + if (unlikely(ret)) { + hybp(HYB_ERR, "val is error!\n"); + return -EINVAL; + } + + ret = update_akcompressd_threads(val, zram); + if (ret < 0) { + hybp(HYB_ERR, "create task failed, val %d\n", val); + return ret; + } + + return len; +} + +ssize_t hybridswap_akcompress_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int len = 0, id, i; + struct cgroup_cache_page *cache = NULL; + unsigned long cnt = atomic64_read(&cached_cnt); + memcg_hybs_t *hybs; + + len += sprintf(buf + len, "akcompressd_threads: %d\n", akcompressd_threads); + len += sprintf(buf + len, "cached page cnt: %lu\n", cnt); + len += sprintf(buf + len, "free page cnt: %u\n", compress_info.free_cnt); + + for (i = 0; i < MAX_AKCOMPRESSD_THREADS; i++) + len += sprintf(buf + len, "%-d %-d\n", i, atomic64_read(&akc_cnt[i])); + + if (cnt == 0) + return len; + + spin_lock(&cached_idr_lock); + idr_for_each_entry(&cached_idr, cache, id) { + hybs = container_of(cache, memcg_hybs_t, cache); + if (cache->cnt == 0) + continue; + + len += scnprintf(buf + len, PAGE_SIZE - len, "%s %d\n", + hybs->name, cache->cnt); + + if (len >= PAGE_SIZE) + break; + } + spin_unlock(&cached_idr_lock); + + return len; +} + +void __init akcompressd_pre_init(void) +{ + int i; + struct page *page; + + mutex_lock(&akcompress_init_lock); + INIT_LIST_HEAD(&compress_info.free_page_head); + spin_lock_init(&compress_info.free_lock); + compress_info.free_cnt = 0; + + init_waitqueue_head(&akcompressd_wait); + + atomic64_set(&cached_cnt, 0); + for (i = 0; i < MAX_AKCOMPRESSD_THREADS; i++) + atomic64_set(&akc_cnt[i], 0); + + for (i = 0; i < DEFAULT_CACHE_COUNT; i ++) { + page = alloc_page(GFP_KERNEL); + + if (page) { + list_add_tail(&page->lru, &compress_info.free_page_head); + } else + break; + } + compress_info.free_cnt = i; + mutex_unlock(&akcompress_init_lock); +} + +void __exit akcompressd_pre_deinit(void) +{ + int i; + struct page *page, *tmp; + + mutex_lock(&akcompress_init_lock); + if (list_empty(&compress_info.free_page_head)) + goto out; + list_for_each_entry_safe(page, tmp, &compress_info.free_page_head , lru) { + list_del(&page->lru); + free_page(page); + } + +out: + compress_info.free_cnt = 0; + mutex_unlock(&akcompress_init_lock); +} + +int akcompress_cache_page_fault(struct zram *zram, + struct page *page, u32 index) +{ + void *src, *dst; + + if (zram_test_flag(zram, index, ZRAM_CACHED)) { + struct page *src_page = (struct page *)zram_fetch_page(zram, index); + + src = kmap_atomic(src_page); + dst = kmap_atomic(page); + memcpy(dst, src, PAGE_SIZE); + kunmap_atomic(src); + kunmap_atomic(dst); + zram_slot_unlock(zram, index); + + hybp(HYB_DEBUG, "read_anon_page_from_cache index %u page %p passed, ZRAM_CACHED\n", + index, src_page); + return 1; + } + + if (zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + struct page *src_page = (struct page *)zram_fetch_page(zram, index); + + src = kmap_atomic(src_page); + dst = kmap_atomic(page); + memcpy(dst, src, PAGE_SIZE); + kunmap_atomic(src); + kunmap_atomic(dst); + zram_slot_unlock(zram, index); + + hybp(HYB_DEBUG, "read_anon_page_from_cache index %u page %p passed, ZRAM_CACHED_COMPRESS\n", + index, src_page); + return 1; + } + + return 0; +} diff --git a/drivers/moto_swap/hybridswap/hybridswap_eswap.c b/drivers/moto_swap/hybridswap/hybridswap_eswap.c new file mode 100644 index 000000000000..299a6c1abbb7 --- /dev/null +++ b/drivers/moto_swap/hybridswap/hybridswap_eswap.c @@ -0,0 +1,5503 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (C) 2020-2022 Oplus. All rights reserved. + */ + +#define pr_fmt(fmt) "moto_swap: " fmt + +#include +#include +#include +#include +#include +#include +#include +#ifdef CONFIG_FG_TASK_UID +#include +#endif +#include + +#ifdef CONFIG_ZRAM_5_4 +#include "../zram-5.4/zram_drv.h" +#include "../zram-5.4/zram_drv_internal.h" +#else +#include "../zram-5.10/zram_drv.h" +#include "../zram-5.10/zram_drv_internal.h" +#endif +#include "hybridswap_internal.h" +#include "hybridswap.h" + +#define PRE_EOL_INFO_OVER_VAL 2 +#define LIFE_TIME_EST_OVER_VAL 8 +#define DEFAULT_STORED_WM_RATIO 90 +#define DEVICE_NAME_LEN 64 +#define MIN_RECLAIM_ZRAM_SZ (1024 * 1024) +#define esentry_extid(e) ((e) >> ESWAP_SHIFT) +#define esentry_pgid(e) (((e) & ((1 << ESWAP_SHIFT) - 1)) >> PAGE_SHIFT) +#define esentry_pgoff(e) ((e) & (PAGE_SIZE - 1)) +#define DUMP_BUF_LEN 512 +#define HYBRIDSWAP_KEY_INDEX 0 +#define HYBRIDSWAP_KEY_SIZE 64 +#define HYBRIDSWAP_KEY_INDEX_SHIFT 3 +#define HYBRIDSWAP_MAX_INFILGHT_NUM 256 +#define HYBRIDSWAP_SECTOR_SHIFT 9 +#define HYBRIDSWAP_PAGE_SIZE_SECTOR (PAGE_SIZE >> HYBRIDSWAP_SECTOR_SHIFT) +#define HYBRIDSWAP_READ_TIME 10 +#define HYBRIDSWAP_WRITE_TIME 100 +#define HYB_FAULT_OUT_TIME 10 +#define CLASS_NAME_LEN 32 +#define MBYTE_SHIFT 20 +#define ENTRY_PTR_SHIFT 23 +#define ENTRY_MCG_SHIFT_HALF 8 +#define ENTRY_LOCK_BIT ENTRY_MCG_SHIFT_HALF +#define ENTRY_DATA_BIT (ENTRY_PTR_SHIFT + ENTRY_MCG_SHIFT_HALF + \ + ENTRY_MCG_SHIFT_HALF + 1) + +struct zs_eswap_para { + struct hybridswap_page_pool *pool; + size_t alloc_size; + bool fast; + bool nofail; +}; + +struct hybridswap_cfg { + atomic_t enable; + atomic_t out_to_eswap_enable; + struct hybstatus *stat; + struct workqueue_struct *reclaim_wq; + struct zram *zram; + + atomic_t dev_life; + unsigned long quota_day; + struct timer_list lpc_timer; + struct work_struct lpc_work; +}; + +struct async_req { + struct mem_cgroup *mcg; + unsigned long size; + unsigned long out_size; + unsigned long reclaimined_sz; + struct work_struct work; + int nice; + bool preload; +}; + +struct io_priv { + struct zram *zram; + enum hybridswap_class class; + struct hybridswap_page_pool page_pool; +}; + +struct io_work_arg { + void *iohandle; + struct hybridswap_entry *ioentry; + struct hybridswap_buffer io_buf; + struct io_priv data; + struct hybridswap_key_point_record record; +}; + +struct hyb_sgm_time { + ktime_t submit_bio; + ktime_t end_io; +}; + +struct hyb_sgm { + struct work_struct stopio_work; + struct hyb_sgm_time time; + struct hybridswap_io_req *req; + struct hybridswap_entry *io_entries_fifo[BIO_MAX_PAGES]; + struct list_head io_entries; + sector_t segment_sector; + int eswap_cnt; + u32 bio_result; + int page_cnt; +}; + +struct hyb_info { + unsigned long size; + int total_objects; + int nr_es; + int memcg_num; + + unsigned long *bitmask; + atomic_t last_alloc_bit; + + struct hyb_entries_table *eswap_table; + struct hyb_entries_head *eswap; + + struct hyb_entries_table *objects; + struct hyb_entries_head *maps; + struct hyb_entries_head *lru; + + atomic_t stored_exts; + atomic_t *eswap_stored_pages; + + unsigned int memcgid_cnt[MEM_CGROUP_ID_MAX + 1]; +}; + +struct hyb_entries_head { + unsigned int mcg_left : 8; + unsigned int lock : 1; + unsigned int prev : 23; + unsigned int mcg_right : 8; + unsigned int data : 1; + unsigned int next : 23; +}; + +struct hyb_entries_table { + struct hyb_entries_head *(*fetch_node)(int, void *); + void *private; +}; + +#define index_node(index, tab) ((tab)->fetch_node((index), (tab)->private)) +#define next_index(index, tab) (index_node((index), (tab))->next) +#define prev_index(index, tab) (index_node((index), (tab))->prev) +#define is_last_index(index, hindex, tab) (next_index(index, tab) == (hindex)) +#define is_first_index(index, hindex, tab) (prev_index(index, tab) == (hindex)) +#define hyb_entries_for_each_entry(index, hindex, tab) \ + for ((index) = next_index((hindex), (tab)); \ + (index) != (hindex); (index) = next_index((index), (tab))) +#define hyb_entries_for_each_entry_safe(index, tmp, hindex, tab) \ + for ((index) = next_index((hindex), (tab)), (tmp) = next_index((index), (tab)); \ + (index) != (hindex); (index) = (tmp), (tmp) = next_index((index), (tab))) +#define hyb_entries_for_each_entry_reverse(index, hindex, tab) \ + for ((index) = prev_index((hindex), (tab)); \ + (index) != (hindex); (index) = prev_index((index), (tab))) +#define hyb_entries_for_each_entry_reverse_safe(index, tmp, hindex, tab) \ + for ((index) = prev_index((hindex), (tab)), (tmp) = prev_index((index), (tab)); \ + (index) != (hindex); (index) = (tmp), (tmp) = prev_index((index), (tab))) + +static unsigned long warn_level[HYB_CLASS_BUTT] = { + 0, 200, 500, 0 +}; + +const char *key_point_name[HYB_KYE_POINT_BUTT] = { + "START", + "INIT", + "IOENTRY_ALLOC", + "FIND_ESWAP", + "IO_ESWAP", + "SEGMENT_ALLOC", + "BIO_ALLOC", + "SUBMIT_BIO", + "END_IO", + "SCHED_WORK", + "END_WORK", + "CALL_BACK", + "WAKE_UP", + "ZRAM_LOCK", + "DONE" +}; + +static const char class_name[HYB_CLASS_BUTT][CLASS_NAME_LEN] = { + "out_to_eswap", + "page_fault", + "batches", + "readahead" +}; + +static const char *fg_bg[2] = {"BG", "FG"}; + +bool hyb_io_work_begin_flag; +struct hybridswap_cfg global_settings; + +static u8 hybridswap_io_key[HYBRIDSWAP_KEY_SIZE]; +static struct workqueue_struct *hybridswap_proc_read_workqueue; +static struct workqueue_struct *hybridswap_proc_write_workqueue; +static char loop_device[DEVICE_NAME_LEN]; + +struct mem_cgroup *find_memcg_by_id(unsigned short memcgid); +int obj_index(struct hyb_info *infos, int index); +int eswap_index(struct hyb_info *infos, int index); +int memcgindex(struct hyb_info *infos, int index); +void free_hyb_info(struct hyb_info *infos); +struct hyb_info *alloc_hyb_info(unsigned long ori_size, unsigned long comp_size); +void hybridswap_check_infos_eswap(struct hyb_info *infos); +void hybridswap_free_eswap(struct hyb_info *infos, int eswapid); +int hybridswap_alloc_eswap(struct hyb_info *infos, struct mem_cgroup *mcg); +int fetch_eswap(struct hyb_info *infos, int eswapid); +void put_eswap(struct hyb_info *infos, int eswapid); +int fetch_memcg_eswap(struct hyb_info *infos, struct mem_cgroup *mcg); +int fetch_memcg_zram_entry(struct hyb_info *infos, struct mem_cgroup *mcg); +int fetch_eswap_zram_entry(struct hyb_info *infos, int eswapid); +struct hyb_entries_table *alloc_table(struct hyb_entries_head *(*fetch_node)(int, void *), + void *private, gfp_t gfp); +void hyb_lock_with_idx(int index, struct hyb_entries_table *table); +void hyb_unlock_with_idx(int index, struct hyb_entries_table *table); +void hyb_entries_init(int index, struct hyb_entries_table *table); +void hyb_entries_add_nolock(int index, int hindex, struct hyb_entries_table *table); +void hyb_entries_add_tail_nolock(int index, int hindex, struct hyb_entries_table *table); +void hyb_entries_del_nolock(int index, int hindex, struct hyb_entries_table *table); +void hyb_entries_add(int index, int hindex, struct hyb_entries_table *table); +void hyb_entries_add_tail(int index, int hindex, struct hyb_entries_table *table); +void hyb_entries_del(int index, int hindex, struct hyb_entries_table *table); +unsigned short hyb_entries_fetch_memcgid(int index, struct hyb_entries_table *table); +void hyb_entries_set_memcgid(int index, struct hyb_entries_table *table, int memcgid); +bool hyb_entries_set_priv(int index, struct hyb_entries_table *table); +bool hyb_entries_clear_priv(int index, struct hyb_entries_table *table); +bool hyb_entries_test_priv(int index, struct hyb_entries_table *table); +bool hyb_entries_empty(int hindex, struct hyb_entries_table *table); +void zram_set_mcg(struct zram *zram, u32 index, int memcgid); +struct mem_cgroup *zram_fetch_mcg(struct zram *zram, u32 index); +int zram_fetch_mcg_last_index(struct hyb_info *infos, + struct mem_cgroup *mcg, + int *index, int max_cnt); +int swap_maps_fetch_eswap_index(struct hyb_info *infos, + int eswapid, int *index); +void swap_sorted_list_add(struct zram *zram, u32 index, struct mem_cgroup *memcg); +void swap_sorted_list_add_tail(struct zram *zram, u32 index, struct mem_cgroup *mcg); +void swap_sorted_list_del(struct zram *zram, u32 index); +void swap_maps_insert(struct zram *zram, u32 index); +void swap_maps_destroy(struct zram *zram, u32 index); + +static void hybridswapiowrkshow(struct seq_file *m, struct hybstatus *stat) +{ + int i; + + for (i = 0; i < HYB_CLASS_BUTT; ++i) { + seq_printf(m, "hybridswap_%s_total_lat: %lld\n", + class_name[i], + atomic64_read(&stat->lat[i].total_lat)); + seq_printf(m, "hybridswap_%s_max_lat: %lld\n", + class_name[i], + atomic64_read(&stat->lat[i].max_lat)); + seq_printf(m, "hybridswap_%s_timeout_cnt: %lld\n", + class_name[i], + atomic64_read(&stat->lat[i].timeout_cnt)); + } + + for (i = 0; i < 2; i++) { + seq_printf(m, "page_fault_timeout_100ms_cnt(%s): %lld\n", + fg_bg[i], + atomic64_read(&stat->fault_stat[i].timeout_100ms_cnt)); + seq_printf(m, "page_fault_timeout_500ms_cnt(%s): %lld\n", + fg_bg[i], + atomic64_read(&stat->fault_stat[i].timeout_500ms_cnt)); + } +} + +static void hybstatuss_show(struct seq_file *m, + struct hybstatus *stat) +{ + seq_printf(m, "hybridswap_out_times: %lld\n", + atomic64_read(&stat->reclaimin_cnt)); + seq_printf(m, "hybridswap_out_comp_size: %lld MB\n", + atomic64_read(&stat->reclaimin_bytes) >> MBYTE_SHIFT); + if (PAGE_SHIFT < MBYTE_SHIFT) + seq_printf(m, "hybridswap_out_ori_size: %lld MB\n", + atomic64_read(&stat->reclaimin_pages) >> + (MBYTE_SHIFT - PAGE_SHIFT)); + seq_printf(m, "hybridswap_in_times: %lld\n", + atomic64_read(&stat->batchout_cnt)); + seq_printf(m, "hybridswap_in_comp_size: %lld MB\n", + atomic64_read(&stat->batchout_bytes) >> MBYTE_SHIFT); + if (PAGE_SHIFT < MBYTE_SHIFT) + seq_printf(m, "hybridswap_in_ori_size: %lld MB\n", + atomic64_read(&stat->batchout_pages) >> + (MBYTE_SHIFT - PAGE_SHIFT)); + seq_printf(m, "hybridswap_all_fault: %lld\n", + atomic64_read(&stat->fault_cnt)); + seq_printf(m, "hybridswap_fault: %lld\n", + atomic64_read(&stat->hybridswap_fault_cnt)); +} + +static void hyb_info_info_show(struct seq_file *m, + struct hybstatus *stat) +{ + seq_printf(m, "hybridswap_reout_ori_size: %lld MB\n", + atomic64_read(&stat->reout_pages) >> + (MBYTE_SHIFT - PAGE_SHIFT)); + seq_printf(m, "hybridswap_reout_comp_size: %lld MB\n", + atomic64_read(&stat->reout_bytes) >> MBYTE_SHIFT); + seq_printf(m, "hybridswap_store_comp_size: %lld MB\n", + atomic64_read(&stat->stored_size) >> MBYTE_SHIFT); + seq_printf(m, "hybridswap_store_ori_size: %lld MB\n", + atomic64_read(&stat->stored_pages) >> + (MBYTE_SHIFT - PAGE_SHIFT)); + seq_printf(m, "hybridswap_notify_free_size: %lld MB\n", + atomic64_read(&stat->notify_free) >> + (MBYTE_SHIFT - ESWAP_SHIFT)); + seq_printf(m, "hybridswap_store_memcg_cnt: %lld\n", + atomic64_read(&stat->mcg_cnt)); + seq_printf(m, "hybridswap_store_eswap_cnt: %lld\n", + atomic64_read(&stat->eswap_cnt)); + seq_printf(m, "hybridswap_store_frag_info_cnt: %lld\n", + atomic64_read(&stat->frag_cnt)); +} + +static void hybridswap_fail_show(struct seq_file *m, + struct hybstatus *stat) +{ + int i; + + for (i = 0; i < HYB_CLASS_BUTT; ++i) { + seq_printf(m, "hybridswap_%s_io_fail_cnt: %lld\n", + class_name[i], + atomic64_read(&stat->io_fail_cnt[i])); + seq_printf(m, "hybridswap_%s_alloc_fail_cnt: %lld\n", + class_name[i], + atomic64_read(&stat->alloc_fail_cnt[i])); + } +} + +int hybridswap_psi_show(struct seq_file *m, void *v) +{ + struct hybstatus *stat = NULL; + + if (!hybridswap_core_enabled()) + return -EINVAL; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't fetch stat obj!\n"); + return -EINVAL; + } + + hybstatuss_show(m, stat); + hyb_info_info_show(m, stat); + hybridswapiowrkshow(m, stat); + hybridswap_fail_show(m, stat); + + return 0; +} + +unsigned long hybridswap_fetch_zram_used_pages(void) +{ + struct hybstatus *stat = NULL; + + if (!hybridswap_core_enabled()) + return 0; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't get stat obj!\n"); + + return 0; + } + + return atomic64_read(&stat->zram_stored_pages); +} + +unsigned long long hybridswap_fetch_zram_pagefault(void) +{ + struct hybstatus *stat = NULL; + + if (!hybridswap_core_enabled()) + return 0; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't fetch stat obj!\n"); + + return 0; + } + + return atomic64_read(&stat->fault_cnt); +} + +bool hybridswap_reclaim_work_running(void) +{ + struct hybstatus *stat = NULL; + + if (!hybridswap_core_enabled()) + return false; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't fetch stat obj!\n"); + + return 0; + } + + return atomic64_read(&stat->reclaimin_infight) ? true : false; +} + +unsigned long long hybridswap_read_mcg_stats(struct mem_cgroup *mcg, + enum hybridswap_mcg_member mcg_member) +{ + struct mem_cgroup_hybridswap *mcg_hybs; + + unsigned long long val = 0; + int extcnt; + + if (!hybridswap_core_enabled()) + return 0; + + mcg_hybs = MEMCGRP_ITEM_DATA(mcg); + if (!mcg_hybs) { + hybp(HYB_DEBUG, "NULL mcg_hybs\n"); + return 0; + } + + switch (mcg_member) { + case MCG_ZRAM_STORED_SZ: + val = atomic64_read(&mcg_hybs->zram_stored_size); + break; + case MCG_ZRAM_STORED_PG_SZ: + val = atomic64_read(&mcg_hybs->zram_page_size); + break; + case MCG_DISK_STORED_SZ: + val = atomic64_read(&mcg_hybs->hybridswap_stored_size); + break; + case MCG_DISK_STORED_PG_SZ: + val = atomic64_read(&mcg_hybs->hybridswap_stored_pages); + break; + case MCG_ANON_FAULT_CNT: + val = atomic64_read(&mcg_hybs->hybridswap_allfaultcnt); + break; + case MCG_DISK_FAULT_CNT: + val = atomic64_read(&mcg_hybs->hybridswap_faultcnt); + break; + case MCG_ESWAPOUT_CNT: + val = atomic64_read(&mcg_hybs->hybridswap_outcnt); + break; + case MCG_ESWAPOUT_SZ: + val = atomic64_read(&mcg_hybs->hybridswap_outextcnt) << ESWAP_SHIFT; + break; + case MCG_ESWAPIN_CNT: + val = atomic64_read(&mcg_hybs->hybridswap_incnt); + break; + case MCG_ESWAPIN_SZ: + val = atomic64_read(&mcg_hybs->hybridswap_inextcnt) << ESWAP_SHIFT; + break; + case MCG_DISK_SPACE: + extcnt = atomic_read(&mcg_hybs->hybridswap_extcnt); + if (extcnt < 0) + extcnt = 0; + val = ((unsigned long long) extcnt) << ESWAP_SHIFT; + break; + case MCG_DISK_SPACE_PEAK: + extcnt = atomic_read(&mcg_hybs->hybridswap_peakextcnt); + if (extcnt < 0) + extcnt = 0; + val = ((unsigned long long) extcnt) << ESWAP_SHIFT; + break; + default: + break; + } + + return val; +} + +void hybridswap_fail_record(enum hybridswap_fail_point point, + u32 index, int eswapid, unsigned char *task_comm) +{ + struct hybstatus *stat = NULL; + unsigned long flags; + unsigned int copylen = strlen(task_comm) + 1; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't fetch stat obj!\n"); + return; + } + + if (copylen > TASK_COMM_LEN) { + hybp(HYB_ERR, "task_comm len %d is err\n", copylen); + return; + } + + spin_lock_irqsave(&stat->record.lock, flags); + if (stat->record.num < MAX_FAIL_RECORD_NUM) { + stat->record.record[stat->record.num].point = point; + stat->record.record[stat->record.num].index = index; + stat->record.record[stat->record.num].eswapid = eswapid; + stat->record.record[stat->record.num].time = ktime_get(); + memcpy(stat->record.record[stat->record.num].task_comm, + task_comm, copylen); + stat->record.num++; + } + spin_unlock_irqrestore(&stat->record.lock, flags); +} + +static void hybridswap_fail_record_fetch( + struct hybridswap_fail_record_info *record_info) +{ + struct hybstatus *stat = NULL; + unsigned long flags; + + if (!hybridswap_core_enabled()) + return; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't fetch stat obj!\n"); + return; + } + + spin_lock_irqsave(&stat->record.lock, flags); + memcpy(record_info, &stat->record, + sizeof(struct hybridswap_fail_record_info)); + stat->record.num = 0; + spin_unlock_irqrestore(&stat->record.lock, flags); +} + +static ssize_t hybridswap_fail_record_show(char *buf) +{ + int i; + ssize_t size = 0; + struct hybridswap_fail_record_info record_info = { 0 }; + + hybridswap_fail_record_fetch(&record_info); + + size += scnprintf(buf + size, PAGE_SIZE, + "hybridswap_fail_record_num: %d\n", record_info.num); + for (i = 0; i < record_info.num; ++i) + size += scnprintf(buf + size, PAGE_SIZE - size, + "point[%u]time[%lld]taskname[%s]index[%u]eswapid[%d]\n", + record_info.record[i].point, + ktime_us_delta(ktime_get(), + record_info.record[i].time), + record_info.record[i].task_comm, + record_info.record[i].index, + record_info.record[i].eswapid); + + return size; +} + +ssize_t hybridswap_report_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return hybridswap_fail_record_show(buf); +} + +static inline meminfo_show(struct hybstatus *stat, char *buf, ssize_t len) +{ + unsigned long eswap_total_pages = 0, eswap_compressed_pages = 0; + unsigned long eswap_used_pages = 0; + unsigned long zram_total_pags, zram_used_pages, zram_compressed; + ssize_t size = 0; + + if (!stat || !buf || !len) + return 0; + + (void)hybridswap_stored_info(&eswap_total_pages, &eswap_compressed_pages); + eswap_used_pages = atomic64_read(&stat->stored_pages); +#ifdef CONFIG_HYBRIDSWAP_SWAPD + zram_total_pags = fetch_nr_zram_total(); +#else + zram_total_pags = 0; +#endif + zram_compressed = atomic64_read(&stat->zram_stored_size); + zram_used_pages = atomic64_read(&stat->zram_stored_pages); + + size += scnprintf(buf + size, len - size, "%-32s %12lu KB\n", + "EST:", eswap_total_pages << (PAGE_SHIFT - 10)); + size += scnprintf(buf + size, len - size, "%-32s %12lu KB\n", + "ESU_C:", eswap_compressed_pages << (PAGE_SHIFT - 10)); + size += scnprintf(buf + size, len - size, "%-32s %12lu KB\n", + "ESU_O:", eswap_used_pages << (PAGE_SHIFT - 10)); + size += scnprintf(buf + size, len - size, "%-32s %12lu KB\n", + "ZST:", zram_total_pags << (PAGE_SHIFT - 10)); + size += scnprintf(buf + size, len - size, "%-32s %12lu KB\n", + "ZSU_C:", zram_compressed >> 10); + size += scnprintf(buf + size, len - size, "%-32s %12lu KB\n", + "ZSU_O:", zram_used_pages << (PAGE_SHIFT - 10)); + + return size; +} + +ssize_t hybridswap_stat_snap_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + ssize_t size = 0; + struct hybstatus *stat = NULL; + + if (!hybridswap_core_enabled()) + return 0; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_INFO, "can't fetch stat obj!\n"); + return 0; + } + + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "reclaimin_cnt:", atomic64_read(&stat->reclaimin_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reclaimin_bytes:", atomic64_read(&stat->reclaimin_bytes) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reclaimin_real_load:", atomic64_read(&stat->reclaimin_real_load) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reclaimin_bytes_daily:", atomic64_read(&stat->reclaimin_bytes_daily) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reclaimin_pages:", atomic64_read(&stat->reclaimin_pages) * PAGE_SIZE / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "reclaimin_infight:", atomic64_read(&stat->reclaimin_infight)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "batchout_cnt:", atomic64_read(&stat->batchout_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "batchout_bytes:", atomic64_read(&stat->batchout_bytes) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "batchout_real_load:", atomic64_read(&stat->batchout_real_load) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "batchout_pages:", atomic64_read(&stat->batchout_pages) * PAGE_SIZE / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "batchout_inflight:", atomic64_read(&stat->batchout_inflight)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "fault_cnt:", atomic64_read(&stat->fault_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "hybridswap_fault_cnt:", atomic64_read(&stat->hybridswap_fault_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reout_pages:", atomic64_read(&stat->reout_pages) * PAGE_SIZE / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reout_bytes:", atomic64_read(&stat->reout_bytes) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "zram_stored_pages:", atomic64_read(&stat->zram_stored_pages) * PAGE_SIZE / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "zram_stored_size:", atomic64_read(&stat->zram_stored_size) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "stored_pages:", atomic64_read(&stat->stored_pages) * PAGE_SIZE / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "stored_size:", atomic64_read(&stat->stored_size) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu KB\n", + "reclain-batchout:", (atomic64_read(&stat->reclaimin_real_load) - + atomic64_read(&stat->batchout_real_load)) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12lld KB\n", + "reclain-batchout-stored:", + (atomic64_read(&stat->reclaimin_real_load) - + atomic64_read(&stat->batchout_real_load) - + atomic64_read(&stat->stored_size)) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12lld KB\n", + "dropped_eswap_size:", atomic64_read(&stat->dropped_eswap_size) / SZ_1K); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "notify_free:", atomic64_read(&stat->notify_free)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "frag_cnt:", atomic64_read(&stat->frag_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "mcg_cnt:", atomic64_read(&stat->mcg_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "ext_cnt:", atomic64_read(&stat->eswap_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "miss_free:", atomic64_read(&stat->miss_free)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "memcgid_clear:", atomic64_read(&stat->memcgid_clear)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "skip_track_cnt:", atomic64_read(&stat->skip_track_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "null_memcg_skip_track_cnt:", + atomic64_read(&stat->null_memcg_skip_track_cnt)); + size += scnprintf(buf + size, PAGE_SIZE - size, "%-32s %12llu\n", + "used_swap_pages:", atomic64_read(&stat->used_swap_pages) * PAGE_SIZE / SZ_1K); + size += meminfo_show(stat, buf + size, PAGE_SIZE - size); + + return size; +} + +ssize_t hybridswap_meminfo_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct hybstatus *stat = NULL; + + if (!hybridswap_core_enabled()) + return 0; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_INFO, "can't fetch stat obj!\n"); + return 0; + } + + return meminfo_show(stat, buf, PAGE_SIZE); +} + +static void hybridswap_iostatus_bytes(struct hybridswap_io_req *req) +{ + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat || !req->page_cnt) + return; + + if (req->io_para.class == HYB_RECLAIM_IN) { + atomic64_add(req->page_cnt * PAGE_SIZE, &stat->reclaimin_bytes); + atomic64_add(req->page_cnt * PAGE_SIZE, &stat->reclaimin_bytes_daily); + atomic64_add(atomic64_read(&req->real_load), &stat->reclaimin_real_load); + atomic64_inc(&stat->reclaimin_cnt); + } else { + atomic64_add(req->page_cnt * PAGE_SIZE, &stat->batchout_bytes); + atomic64_inc(&stat->batchout_cnt); + } +} + +static void hybridswap_key_init(void) +{ + get_random_bytes(hybridswap_io_key, HYBRIDSWAP_KEY_SIZE); +} + +static void hybridswap_io_req_release(struct kref *ref) +{ + struct hybridswap_io_req *req = + container_of(ref, struct hybridswap_io_req, refcount); + + if (req->io_para.complete_notify && req->io_para.private) + req->io_para.complete_notify(req->io_para.private); + + kfree(req); +} + +static void hyb_sgm_free(struct hybridswap_io_req *req, + struct hyb_sgm *segment) +{ + int i; + + for (i = 0; i < segment->eswap_cnt; ++i) { + INIT_LIST_HEAD(&segment->io_entries_fifo[i]->list); + req->io_para.done_callback(segment->io_entries_fifo[i], -EIO, req); + } + kfree(segment); +} + +static void hybridswap_limit_doing(struct hybridswap_io_req *req) +{ + int ret; + + if (!req->limit_doing_flag) + return; + + if (atomic_read(&req->eswap_doing) >= HYBRIDSWAP_MAX_INFILGHT_NUM) { + do { + hybp(HYB_DEBUG, "wait doing start\n"); + ret = wait_event_timeout(req->io_wait, + atomic_read(&req->eswap_doing) < + HYBRIDSWAP_MAX_INFILGHT_NUM, + msecs_to_jiffies(100)); + } while (!ret); + } +} + +static void hybridswap_wait_io_finish(struct hybridswap_io_req *req) +{ + int ret; + unsigned int wait_time; + + if (!req->wait_io_finish_flag || !req->page_cnt) + return; + + if (req->io_para.class == HYB_FAULT_OUT) { + hybp(HYB_DEBUG, "fault out wait finish start\n"); + wait_for_completion_io_timeout(&req->io_end_flag, + MAX_SCHEDULE_TIMEOUT); + + return; + } + + wait_time = (req->io_para.class == HYB_RECLAIM_IN) ? + HYBRIDSWAP_WRITE_TIME : HYBRIDSWAP_READ_TIME; + + do { + hybp(HYB_DEBUG, "wait finish start\n"); + ret = wait_event_timeout(req->io_wait, + (!atomic_read(&req->eswap_doing)), + msecs_to_jiffies(wait_time)); + } while (!ret); +} + +static void hybridswap_doing_inc(struct hyb_sgm *segment) +{ + mutex_lock(&segment->req->refmutex); + kref_get(&segment->req->refcount); + mutex_unlock(&segment->req->refmutex); + atomic_add(segment->page_cnt, &segment->req->eswap_doing); +} + +static void hybridswap_doing_dec(struct hybridswap_io_req *req, + int num) +{ + if ((atomic_sub_return(num, &req->eswap_doing) < + HYBRIDSWAP_MAX_INFILGHT_NUM) && req->limit_doing_flag && + wq_has_sleeper(&req->io_wait)) + wake_up(&req->io_wait); +} + +static void hybridswap_io_end_wake_up(struct hybridswap_io_req *req) +{ + if (req->io_para.class == HYB_FAULT_OUT) { + complete(&req->io_end_flag); + return; + } + + if (wq_has_sleeper(&req->io_wait)) + wake_up(&req->io_wait); +} + +static void hybridswap_ioentry_proc(struct hyb_sgm *segment) +{ + int i; + struct hybridswap_io_req *req = segment->req; + struct hybridswap_key_point_record *record = req->io_para.record; + int page_num; + ktime_t callback_start; + unsigned long long callback_start_ravg_sum; + + for (i = 0; i < segment->eswap_cnt; ++i) { + INIT_LIST_HEAD(&segment->io_entries_fifo[i]->list); + page_num = segment->io_entries_fifo[i]->pages_sz; + hybp(HYB_DEBUG, "eswap_id[%d] %d page_num %d\n", + i, segment->io_entries_fifo[i]->eswapid, page_num); + callback_start = ktime_get(); + callback_start_ravg_sum = hybridswap_fetch_ravg_sum(); + if (req->io_para.done_callback) + req->io_para.done_callback(segment->io_entries_fifo[i], + 0, req); + hybperf_async_perf(record, HYB_CALL_BACK, + callback_start, callback_start_ravg_sum); + hybridswap_doing_dec(req, page_num); + } +} + +static void hybridswap_errio_record(enum hybridswap_fail_point point, + struct hybridswap_io_req *req, int eswapid) +{ + if (req->io_para.class == HYB_FAULT_OUT) + hybridswap_fail_record(point, 0, eswapid, + req->io_para.record->task_comm); +} + +static void hybridswap_iostatus_fail(enum hybridswap_class class) +{ + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat || (class >= HYB_CLASS_BUTT)) + return; + + atomic64_inc(&stat->io_fail_cnt[class]); +} + +static void hybridswap_errio_proc(struct hybridswap_io_req *req, + struct hyb_sgm *segment) +{ + hybp(HYB_ERR, "segment sector 0x%llx, eswap_cnt %d\n", + segment->segment_sector, segment->eswap_cnt); + hybp(HYB_ERR, "class %u, bio_result %u\n", + req->io_para.class, segment->bio_result); + hybridswap_iostatus_fail(req->io_para.class); + hybridswap_errio_record(HYB_FAULT_OUT_IO_FAIL, req, + segment->io_entries_fifo[0]->eswapid); + hybridswap_doing_dec(req, segment->page_cnt); + hybridswap_io_end_wake_up(req); + hyb_sgm_free(req, segment); + kref_put_mutex(&req->refcount, hybridswap_io_req_release, + &req->refmutex); +} + +static void hybridswap_io_end_work(struct work_struct *work) +{ + struct hyb_sgm *segment = + container_of(work, struct hyb_sgm, stopio_work); + struct hybridswap_io_req *req = segment->req; + struct hybridswap_key_point_record *record = req->io_para.record; + int old_nice = task_nice(current); + ktime_t work_start; + unsigned long long work_start_ravg_sum; + + if (unlikely(segment->bio_result)) { + hybridswap_errio_proc(req, segment); + return; + } + + hybp(HYB_DEBUG, "segment sector 0x%llx, eswap_cnt %d passed\n", + segment->segment_sector, segment->eswap_cnt); + hybp(HYB_DEBUG, "class %u, bio_result %u passed\n", + req->io_para.class, segment->bio_result); + + set_user_nice(current, req->nice); + + hybperf_async_perf(record, HYB_SCHED_WORK, + segment->time.end_io, 0); + work_start = ktime_get(); + work_start_ravg_sum = hybridswap_fetch_ravg_sum(); + + hybridswap_ioentry_proc(segment); + + hybperf_async_perf(record, HYB_END_WORK, work_start, + work_start_ravg_sum); + + hybridswap_io_end_wake_up(req); + + kref_put_mutex(&req->refcount, hybridswap_io_req_release, + &req->refmutex); + kfree(segment); + + set_user_nice(current, old_nice); +} + +static void hybridswap_end_io(struct bio *bio) +{ + struct hyb_sgm *segment = bio->bi_private; + struct hybridswap_io_req *req = NULL; + struct workqueue_struct *workqueue = NULL; + struct hybridswap_key_point_record *record = NULL; + + if (unlikely(!segment || !(segment->req))) { + hybp(HYB_ERR, "segment or req null\n"); + bio_put(bio); + + return; + } + + req = segment->req; + record = req->io_para.record; + + hybperf_async_perf(record, HYB_END_IO, + segment->time.submit_bio, 0); + + workqueue = (req->io_para.class == HYB_RECLAIM_IN) ? + hybridswap_proc_write_workqueue : hybridswap_proc_read_workqueue; + segment->time.end_io = ktime_get(); + segment->bio_result = bio->bi_status; + + queue_work(workqueue, &segment->stopio_work); + bio_put(bio); +} + +static bool hybridswap_eswap_merge_back( + struct hyb_sgm *segment, + struct hybridswap_entry *ioentry) +{ + struct hybridswap_entry *tail_ioentry = + list_last_entry(&segment->io_entries, + struct hybridswap_entry, list); + + return ((tail_ioentry->addr + + tail_ioentry->pages_sz * HYBRIDSWAP_PAGE_SIZE_SECTOR) == + ioentry->addr); +} + +static bool hybridswap_eswap_merge_front( + struct hyb_sgm *segment, + struct hybridswap_entry *ioentry) +{ + struct hybridswap_entry *head_ioentry = + list_first_entry(&segment->io_entries, + struct hybridswap_entry, list); + + return (head_ioentry->addr == + (ioentry->addr + + ioentry->pages_sz * HYBRIDSWAP_PAGE_SIZE_SECTOR)); +} + +static bool hybridswap_eswap_merge(struct hybridswap_io_req *req, + struct hybridswap_entry *ioentry) +{ + struct hyb_sgm *segment = req->segment; + + if (segment == NULL) + return false; + + if ((segment->page_cnt + ioentry->pages_sz) > BIO_MAX_PAGES) + return false; + + if (hybridswap_eswap_merge_front(segment, ioentry)) { + list_add(&ioentry->list, &segment->io_entries); + segment->io_entries_fifo[segment->eswap_cnt++] = ioentry; + segment->segment_sector = ioentry->addr; + segment->page_cnt += ioentry->pages_sz; + return true; + } + + if (hybridswap_eswap_merge_back(segment, ioentry)) { + list_add_tail(&ioentry->list, &segment->io_entries); + segment->io_entries_fifo[segment->eswap_cnt++] = ioentry; + segment->page_cnt += ioentry->pages_sz; + return true; + } + + return false; +} + +static struct bio *hybridswap_bio_alloc(enum hybridswap_class class) +{ + gfp_t gfp = (class != HYB_RECLAIM_IN) ? GFP_ATOMIC : GFP_NOIO; + struct bio *bio = bio_alloc(gfp, BIO_MAX_PAGES); + + if (!bio && (class == HYB_FAULT_OUT)) + bio = bio_alloc(GFP_NOIO, BIO_MAX_PAGES); + + return bio; +} + +static int hybridswap_bio_add_page(struct bio *bio, + struct hyb_sgm *segment) +{ + int i; + int k = 0; + struct hybridswap_entry *ioentry = NULL; + struct hybridswap_entry *tmp = NULL; + + list_for_each_entry_safe(ioentry, tmp, &segment->io_entries, list) { + for (i = 0; i < ioentry->pages_sz; i++) { + ioentry->dest_pages[i]->index = + bio->bi_iter.bi_sector + k; + if (unlikely(!bio_add_page(bio, + ioentry->dest_pages[i], PAGE_SIZE, 0))) { + return -EIO; + } + k += HYBRIDSWAP_PAGE_SIZE_SECTOR; + } + } + + return 0; +} + +static void hybridswap_set_bio_opf(struct bio *bio, + struct hyb_sgm *segment) +{ + if (segment->req->io_para.class == HYB_RECLAIM_IN) { + bio->bi_opf |= REQ_BACKGROUND; + return; + } + + bio->bi_opf |= REQ_SYNC; +} + +int hybridswap_submit_bio(struct hyb_sgm *segment) +{ + unsigned int op = + (segment->req->io_para.class == HYB_RECLAIM_IN) ? + REQ_OP_WRITE : REQ_OP_READ; + struct hybridswap_entry *head_ioentry = + list_first_entry(&segment->io_entries, + struct hybridswap_entry, list); + struct hybridswap_key_point_record *record = + segment->req->io_para.record; + struct bio *bio = NULL; + + hybperfiowrkstart(record, HYB_BIO_ALLOC); + bio = hybridswap_bio_alloc(segment->req->io_para.class); + hybperfiowrkend(record, HYB_BIO_ALLOC); + if (unlikely(!bio)) { + hybp(HYB_ERR, "bio is null.\n"); + hybridswap_errio_record(HYB_FAULT_OUT_BIO_ALLOC_FAIL, + segment->req, segment->io_entries_fifo[0]->eswapid); + + return -ENOMEM; + } + + bio->bi_iter.bi_sector = segment->segment_sector; + bio_set_dev(bio, segment->req->io_para.bdev); + bio->bi_private = segment; + bio_set_op_attrs(bio, op, 0); + bio->bi_end_io = hybridswap_end_io; + hybridswap_set_bio_opf(bio, segment); + + if (unlikely(hybridswap_bio_add_page(bio, segment))) { + bio_put(bio); + hybp(HYB_ERR, "bio_add_page fail\n"); + hybridswap_errio_record(HYB_FAULT_OUT_BIO_ADD_FAIL, + segment->req, segment->io_entries_fifo[0]->eswapid); + + return -EIO; + } + + hybridswap_doing_inc(segment); + hybp(HYB_DEBUG, "submit bio sector %llu eswapid %d\n", + segment->segment_sector, head_ioentry->eswapid); + hybp(HYB_DEBUG, "eswap_cnt %d class %u\n", + segment->eswap_cnt, segment->req->io_para.class); + + segment->req->page_cnt += segment->page_cnt; + segment->req->segment_cnt++; + segment->time.submit_bio = ktime_get(); + + hybperfiowrkstart(record, HYB_SUBMIT_BIO); + submit_bio(bio); + hybperfiowrkend(record, HYB_SUBMIT_BIO); + + return 0; +} + +static int hybridswap_new_segment_init(struct hybridswap_io_req *req, + struct hybridswap_entry *ioentry) +{ + gfp_t gfp = (req->io_para.class != HYB_RECLAIM_IN) ? + GFP_ATOMIC : GFP_NOIO; + struct hyb_sgm *segment = NULL; + struct hybridswap_key_point_record *record = req->io_para.record; + + hybperfiowrkstart(record, HYB_SEGMENT_ALLOC); + segment = kzalloc(sizeof(struct hyb_sgm), gfp); + if (!segment && (req->io_para.class == HYB_FAULT_OUT)) + segment = kzalloc(sizeof(struct hyb_sgm), GFP_NOIO); + hybperfiowrkend(record, HYB_SEGMENT_ALLOC); + if (unlikely(!segment)) { + hybridswap_errio_record(HYB_FAULT_OUT_SEGMENT_ALLOC_FAIL, + req, ioentry->eswapid); + + return -ENOMEM; + } + + segment->req = req; + INIT_LIST_HEAD(&segment->io_entries); + list_add_tail(&ioentry->list, &segment->io_entries); + segment->io_entries_fifo[segment->eswap_cnt++] = ioentry; + segment->page_cnt = ioentry->pages_sz; + INIT_WORK(&segment->stopio_work, hybridswap_io_end_work); + segment->segment_sector = ioentry->addr; + req->segment = segment; + + return 0; +} + +static int hybridswap_io_submit(struct hybridswap_io_req *req, + bool merge_flag) +{ + int ret; + struct hyb_sgm *segment = req->segment; + + if (!segment || ((merge_flag) && (segment->page_cnt < BIO_MAX_PAGES))) + return 0; + + hybridswap_limit_doing(req); + + ret = hybridswap_submit_bio(segment); + if (unlikely(ret)) { + hybp(HYB_WARN, "submit bio failed, ret %d\n", ret); + hyb_sgm_free(req, segment); + } + req->segment = NULL; + + return ret; +} + +static bool hybridswap_check_io_para_err(struct hybridswap_io *io_para) +{ + if (unlikely(!io_para)) { + hybp(HYB_ERR, "io_para null\n"); + + return true; + } + + if (unlikely(!io_para->bdev || + (io_para->class >= HYB_CLASS_BUTT))) { + hybp(HYB_ERR, "io_para err, class %u\n", + io_para->class); + + return true; + } + + if (unlikely(!io_para->done_callback)) { + hybp(HYB_ERR, "done_callback err\n"); + + return true; + } + + return false; +} + +static bool hybridswap_check_entry_err( + struct hybridswap_entry *ioentry) +{ + int i; + + if (unlikely(!ioentry)) { + hybp(HYB_ERR, "ioentry null\n"); + + return true; + } + + if (unlikely((!ioentry->dest_pages) || + (ioentry->eswapid < 0) || + (ioentry->pages_sz > BIO_MAX_PAGES) || + (ioentry->pages_sz <= 0))) { + hybp(HYB_ERR, "eswapid %d, page_sz %d\n", ioentry->eswapid, + ioentry->pages_sz); + + return true; + } + + for (i = 0; i < ioentry->pages_sz; ++i) { + if (!ioentry->dest_pages[i]) { + hybp(HYB_ERR, "dest_pages[%d] is null\n", i); + return true; + } + } + + return false; +} + +static int hybridswap_io_eswapent(void *iohandle, + struct hybridswap_entry *ioentry) +{ + int ret; + struct hybridswap_io_req *req = (struct hybridswap_io_req *)iohandle; + + if (unlikely(hybridswap_check_entry_err(ioentry))) { + hybridswap_errio_record(HYB_FAULT_OUT_IO_ENTRY_PARA_FAIL, + req, ioentry ? ioentry->eswapid : -EINVAL); + req->io_para.done_callback(ioentry, -EIO, req); + + return -EFAULT; + } + + hybp(HYB_DEBUG, "eswap id %d, pages_sz %d, addr %llx\n", + ioentry->eswapid, ioentry->pages_sz, + ioentry->addr); + + if (hybridswap_eswap_merge(req, ioentry)) + return hybridswap_io_submit(req, true); + + ret = hybridswap_io_submit(req, false); + if (unlikely(ret)) { + hybp(HYB_ERR, "submit fail %d\n", ret); + req->io_para.done_callback(ioentry, -EIO, req); + + return ret; + } + + ret = hybridswap_new_segment_init(req, ioentry); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap_new_segment_init fail %d\n", ret); + req->io_para.done_callback(ioentry, -EIO, req); + + return ret; + } + + return 0; +} + +int hyb_io_work_begin(void) +{ + if (hyb_io_work_begin_flag) + return 0; + + hybridswap_proc_read_workqueue = alloc_workqueue("proc_hybridswap_read", + WQ_HIGHPRI | WQ_UNBOUND, 0); + if (unlikely(!hybridswap_proc_read_workqueue)) + return -EFAULT; + + hybridswap_proc_write_workqueue = alloc_workqueue("proc_hybridswap_write", + WQ_CPU_INTENSIVE, 0); + if (unlikely(!hybridswap_proc_write_workqueue)) { + destroy_workqueue(hybridswap_proc_read_workqueue); + + return -EFAULT; + } + + hybridswap_key_init(); + + hyb_io_work_begin_flag = true; + + return 0; +} + +void *hybridswap_plug_start(struct hybridswap_io *io_para) +{ + gfp_t gfp; + struct hybridswap_io_req *req = NULL; + + if (unlikely(hybridswap_check_io_para_err(io_para))) + return NULL; + + gfp = (io_para->class != HYB_RECLAIM_IN) ? + GFP_ATOMIC : GFP_NOIO; + req = kzalloc(sizeof(struct hybridswap_io_req), gfp); + if (!req && (io_para->class == HYB_FAULT_OUT)) + req = kzalloc(sizeof(struct hybridswap_io_req), GFP_NOIO); + + if (unlikely(!req)) { + hybp(HYB_ERR, "io_req null\n"); + + return NULL; + } + + kref_init(&req->refcount); + mutex_init(&req->refmutex); + atomic_set(&req->eswap_doing, 0); + init_waitqueue_head(&req->io_wait); + req->io_para.bdev = io_para->bdev; + req->io_para.class = io_para->class; + req->io_para.done_callback = io_para->done_callback; + req->io_para.complete_notify = io_para->complete_notify; + req->io_para.private = io_para->private; + req->io_para.record = io_para->record; + req->limit_doing_flag = + (io_para->class == HYB_RECLAIM_IN) || + (io_para->class == HYB_PRE_OUT); + req->wait_io_finish_flag = + (io_para->class == HYB_RECLAIM_IN) || + (io_para->class == HYB_FAULT_OUT); + req->nice = task_nice(current); + init_completion(&req->io_end_flag); + + return (void *)req; +} + +int hybridswap_read_eswap(void *iohandle, + struct hybridswap_entry *ioentry) +{ + return hybridswap_io_eswapent(iohandle, ioentry); +} + +int hybridswap_write_eswap(void *iohandle, + struct hybridswap_entry *ioentry) +{ + return hybridswap_io_eswapent(iohandle, ioentry); +} + +int hybridswap_plug_finish(void *iohandle) +{ + int ret; + struct hybridswap_io_req *req = (struct hybridswap_io_req *)iohandle; + + hybperfiowrkstart(req->io_para.record, HYB_IO_ESWAP); + ret = hybridswap_io_submit(req, false); + if (unlikely(ret)) + hybp(HYB_ERR, "submit fail %d\n", ret); + + hybperfiowrkend(req->io_para.record, HYB_IO_ESWAP); + hybridswap_wait_io_finish(req); + hybperfiowrkpoint(req->io_para.record, HYB_WAKE_UP); + + hybridswap_iostatus_bytes(req); + hybperf_io_stat(req->io_para.record, req->page_cnt, + req->segment_cnt); + + kref_put_mutex(&req->refcount, hybridswap_io_req_release, + &req->refmutex); + + hybp(HYB_DEBUG, "io schedule finish succ\n"); + + return ret; +} + +static void hybridswap_dump_point_lat( + struct hybridswap_key_point_record *record, ktime_t start) +{ + int i; + + for (i = 0; i < HYB_KYE_POINT_BUTT; ++i) { + if (!record->key_point[i].record_cnt) + continue; + + hybp(HYB_ERR, + "%s diff %lld cnt %u end %u lat %lld ravg_sum %llu\n", + key_point_name[i], + ktime_us_delta(record->key_point[i].first_time, start), + record->key_point[i].record_cnt, + record->key_point[i].end_cnt, + record->key_point[i].proc_total_time, + record->key_point[i].proc_ravg_sum); + } +} + +static void hybridswap_dump_no_record_point( + struct hybridswap_key_point_record *record, char *log, + unsigned int *count) +{ + int i; + unsigned int point = 0; + + for (i = 0; i < HYB_KYE_POINT_BUTT; ++i) + if (record->key_point[i].record_cnt) + point = i; + + point++; + if (point < HYB_KYE_POINT_BUTT) + *count += snprintf(log + *count, + (size_t)(DUMP_BUF_LEN - *count), + " no_record_point %s", key_point_name[point]); + else + *count += snprintf(log + *count, + (size_t)(DUMP_BUF_LEN - *count), " all_point_record"); +} + +static long long hybridswap_calc_speed(s64 page_cnt, s64 time) +{ + s64 size; + + if (!page_cnt) + return 0; + + size = page_cnt * PAGE_SIZE * BITS_PER_BYTE; + if (time) + return size * USEC_PER_SEC / time; + else + return S64_MAX; +} + +static void hybridswap_dump_lat( + struct hybridswap_key_point_record *record, ktime_t curr_time, + bool perf_end_flag) +{ + char log[DUMP_BUF_LEN] = { 0 }; + unsigned int count = 0; + ktime_t start; + s64 total_time; + + start = record->key_point[HYB_START].first_time; + total_time = ktime_us_delta(curr_time, start); + count += snprintf(log + count, + (size_t)(DUMP_BUF_LEN - count), + "totaltime(us) %lld class %u task %s nice %d", + total_time, record->class, record->task_comm, record->nice); + + if (perf_end_flag) + count += snprintf(log + count, (size_t)(DUMP_BUF_LEN - count), + " page %d segment %d speed(bps) %lld level %lu", + record->page_cnt, record->segment_cnt, + hybridswap_calc_speed(record->page_cnt, total_time), + record->warn_level); + else + count += snprintf(log + count, (size_t)(DUMP_BUF_LEN - count), + " state %c", task_state_to_char(record->task)); + + hybridswap_dump_no_record_point(record, log, &count); + + hybp(HYB_ERR, "perf end flag %u %s\n", perf_end_flag, log); + hybridswap_dump_point_lat(record, start); + dump_stack(); +} + +static unsigned long hybperf_warn_level( + enum hybridswap_class class) +{ + if (unlikely(class >= HYB_CLASS_BUTT)) + return 0; + + return warn_level[class]; +} + +void hybperf_warning(struct timer_list *t) +{ + struct hybridswap_key_point_record *record = + from_timer(record, t, lat_monitor); + static unsigned long last_dumpiowrkjiffies = 0; + + if (!record->warn_level) + return; + + if (jiffies_to_msecs(jiffies - last_dumpiowrkjiffies) <= 60000) + return; + + hybridswap_dump_lat(record, ktime_get(), false); + +#if LINUX_VERSION_CODE >= KERNEL_VERSION(5, 10, 0) + if (likely(record->task)) + sched_show_task(record->task); +#endif + last_dumpiowrkjiffies = jiffies; + record->warn_level <<= 2; + record->timeout_flag = true; + mod_timer(&record->lat_monitor, + jiffies + msecs_to_jiffies(record->warn_level)); +} + +static void hybperf_init_monitor( + struct hybridswap_key_point_record *record, + enum hybridswap_class class) +{ + record->warn_level = hybperf_warn_level(class); + + if (!record->warn_level) + return; + + record->task = current; + get_task_struct(record->task); + timer_setup(&record->lat_monitor, hybperf_warning, 0); + mod_timer(&record->lat_monitor, + jiffies + msecs_to_jiffies(record->warn_level)); +} + +static void hybperf_stop_monitor( + struct hybridswap_key_point_record *record) +{ + if (!record->warn_level) + return; + + del_timer_sync(&record->lat_monitor); + put_task_struct(record->task); +} + +static void hybperf_init(struct hybridswap_key_point_record *record, + enum hybridswap_class class) +{ + int i; + + for (i = 0; i < HYB_KYE_POINT_BUTT; ++i) + spin_lock_init(&record->key_point[i].time_lock); + + record->nice = task_nice(current); + record->class = class; + get_task_comm(record->task_comm, current); + hybperf_init_monitor(record, class); +} + +void hybperf_start_proc( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type, ktime_t curr_time, + unsigned long long current_ravg_sum) +{ + struct hybridswap_key_point_info *key_point = + &record->key_point[type]; + + if (!key_point->record_cnt) + key_point->first_time = curr_time; + + key_point->record_cnt++; + key_point->last_time = curr_time; + key_point->last_ravg_sum = current_ravg_sum; +} + +void hybperf_end_proc( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type, ktime_t curr_time, + unsigned long long current_ravg_sum) +{ + struct hybridswap_key_point_info *key_point = + &record->key_point[type]; + s64 diff_time = ktime_us_delta(curr_time, key_point->last_time); + + key_point->proc_total_time += diff_time; + if (diff_time > key_point->proc_max_time) + key_point->proc_max_time = diff_time; + + key_point->proc_ravg_sum += current_ravg_sum - + key_point->last_ravg_sum; + key_point->end_cnt++; + key_point->last_time = curr_time; + key_point->last_ravg_sum = current_ravg_sum; +} + +void hybperf_async_perf( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type, ktime_t start, + unsigned long long start_ravg_sum) +{ + unsigned long long current_ravg_sum = ((type == HYB_CALL_BACK) || + (type == HYB_END_WORK)) ? hybridswap_fetch_ravg_sum() : 0; + unsigned long flags; + + spin_lock_irqsave(&record->key_point[type].time_lock, flags); + hybperf_start_proc(record, type, start, start_ravg_sum); + hybperf_end_proc(record, type, ktime_get(), + current_ravg_sum); + spin_unlock_irqrestore(&record->key_point[type].time_lock, flags); +} + +void hybperfiowrkpoint( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type) +{ + hybperf_start_proc(record, type, ktime_get(), + hybridswap_fetch_ravg_sum()); + record->key_point[type].end_cnt++; +} + +void hybperf_start( + struct hybridswap_key_point_record *record, + ktime_t stsrt, unsigned long long start_ravg_sum, + enum hybridswap_class class) +{ + hybperf_init(record, class); + hybperf_start_proc(record, HYB_START, stsrt, + start_ravg_sum); + record->key_point[HYB_START].end_cnt++; +} + +void hybperfiowrkstat( + struct hybridswap_key_point_record *record) +{ + int task_is_fg = 0; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + s64 curr_lat; + s64 timeout_value[HYB_CLASS_BUTT] = { + 2000000, 100000, 500000, 2000000 + }; + + if (!stat || (record->class >= HYB_CLASS_BUTT)) + return; + + curr_lat = ktime_us_delta(record->key_point[HYB_DONE].first_time, + record->key_point[HYB_START].first_time); + atomic64_add(curr_lat, &stat->lat[record->class].total_lat); + if (curr_lat > atomic64_read(&stat->lat[record->class].max_lat)) + atomic64_set(&stat->lat[record->class].max_lat, curr_lat); + if (curr_lat > timeout_value[record->class]) + atomic64_inc(&stat->lat[record->class].timeout_cnt); + if (record->class == HYB_FAULT_OUT) { + if (curr_lat <= timeout_value[HYB_FAULT_OUT]) + return; +#ifdef CONFIG_FG_TASK_UID + task_is_fg = current_is_fg() ? 1 : 0; +#endif + if (curr_lat > 500000) + atomic64_inc(&stat->fault_stat[task_is_fg].timeout_500ms_cnt); + else if (curr_lat > 100000) + atomic64_inc(&stat->fault_stat[task_is_fg].timeout_100ms_cnt); + hybp(HYB_INFO, "task %s:%d fault out timeout us %llu fg %d\n", + current->comm, current->pid, curr_lat, task_is_fg); + } +} + +void hybperf_end(struct hybridswap_key_point_record *record) +{ + int loglevel; + + hybperf_stop_monitor(record); + hybperfiowrkpoint(record, HYB_DONE); + hybperfiowrkstat(record); + + loglevel = record->timeout_flag ? HYB_ERR : HYB_DEBUG; + if (loglevel > hybridswap_loglevel()) + return; + + hybridswap_dump_lat(record, + record->key_point[HYB_DONE].first_time, true); +} + +void hybperfiowrkstart( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type) +{ + hybperf_start_proc(record, type, ktime_get(), + hybridswap_fetch_ravg_sum()); +} + +void hybperfiowrkend( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type) +{ + hybperf_end_proc(record, type, ktime_get(), + hybridswap_fetch_ravg_sum()); +} + +void hybperf_io_stat( + struct hybridswap_key_point_record *record, int page_cnt, + int segment_cnt) +{ + record->page_cnt = page_cnt; + record->segment_cnt = segment_cnt; +} + +static struct io_eswapent *alloc_io_eswapent(struct hybridswap_page_pool *pool, + bool fast, bool nofail) +{ + int i; + struct io_eswapent *io_eswap = hybridswap_malloc(sizeof(struct io_eswapent), + fast, nofail); + + if (!io_eswap) { + hybp(HYB_ERR, "alloc io_eswap failed\n"); + return NULL; + } + + io_eswap->eswapid = -EINVAL; + io_eswap->pool = pool; + for (i = 0; i < (int)ESWAP_PG_CNT; i++) { + io_eswap->pages[i] = hybridswap_alloc_page(pool, GFP_ATOMIC, + fast, nofail); + if (!io_eswap->pages[i]) { + hybp(HYB_ERR, "alloc page[%d] failed\n", i); + goto page_free; + } + } + return io_eswap; +page_free: + for (i = 0; i < (int)ESWAP_PG_CNT; i++) + if (io_eswap->pages[i]) + hybridswap_page_recycle(io_eswap->pages[i], pool); + hybridswap_free(io_eswap); + + return NULL; +} + +static void discard_io_eswapent(struct io_eswapent *io_eswap, unsigned int op) +{ + struct zram *zram = NULL; + int i; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return; + } + if (!io_eswap->mcg) + zram = io_eswap->zram; + else + zram = MEMCGRP_ITEM(io_eswap->mcg, zram); + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + goto out; + } + for (i = 0; i < (int)ESWAP_PG_CNT; i++) + if (io_eswap->pages[i]) + hybridswap_page_recycle(io_eswap->pages[i], io_eswap->pool); + if (io_eswap->eswapid < 0) + goto out; + hybp(HYB_DEBUG, "eswap = %d, op = %d\n", io_eswap->eswapid, op); + if (op == REQ_OP_READ) { + put_eswap(zram->infos, io_eswap->eswapid); + goto out; + } + for (i = 0; i < io_eswap->cnt; i++) { + u32 index = io_eswap->index[i]; + + zram_slot_lock(zram, index); + if (io_eswap->mcg) + swap_sorted_list_add_tail(zram, index, io_eswap->mcg); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_slot_unlock(zram, index); + } + hybridswap_free_eswap(zram->infos, io_eswap->eswapid); +out: + hybridswap_free(io_eswap); +} + +static void copy_to_pages(u8 *src, struct page *pages[], + unsigned long eswpentry, int size) +{ + u8 *dst = NULL; + int pg_id = esentry_pgid(eswpentry); + int offset = esentry_pgoff(eswpentry); + + if (!src) { + hybp(HYB_ERR, "NULL src\n"); + return; + } + if (!pages) { + hybp(HYB_ERR, "NULL pages\n"); + return; + } + if (size < 0 || size > (int)PAGE_SIZE) { + hybp(HYB_ERR, "size = %d invalid\n", size); + return; + } + dst = page_to_virt(pages[pg_id]); + if (offset + size <= (int)PAGE_SIZE) { + memcpy(dst + offset, src, size); + return; + } + if (pg_id == ESWAP_PG_CNT - 1) { + hybp(HYB_ERR, "eswap overflow, addr = %lx, size = %d\n", + eswpentry, size); + return; + } + memcpy(dst + offset, src, PAGE_SIZE - offset); + dst = page_to_virt(pages[pg_id + 1]); + memcpy(dst, src + PAGE_SIZE - offset, offset + size - PAGE_SIZE); +} + +static void copy_from_pages(u8 *dst, struct page *pages[], + unsigned long eswpentry, int size) +{ + u8 *src = NULL; + int pg_id = esentry_pgid(eswpentry); + int offset = esentry_pgoff(eswpentry); + + if (!dst) { + hybp(HYB_ERR, "NULL dst\n"); + return; + } + if (!pages) { + hybp(HYB_ERR, "NULL pages\n"); + return; + } + if (size < 0 || size > (int)PAGE_SIZE) { + hybp(HYB_ERR, "size = %d invalid\n", size); + return; + } + + src = page_to_virt(pages[pg_id]); + if (offset + size <= (int)PAGE_SIZE) { + memcpy(dst, src + offset, size); + return; + } + if (pg_id == ESWAP_PG_CNT - 1) { + hybp(HYB_ERR, "eswap overflow, addr = %lx, size = %d\n", + eswpentry, size); + return; + } + memcpy(dst, src + offset, PAGE_SIZE - offset); + src = page_to_virt(pages[pg_id + 1]); + memcpy(dst + PAGE_SIZE - offset, src, offset + size - PAGE_SIZE); +} + +static bool zram_test_skip(struct zram *zram, u32 index, struct mem_cgroup *mcg) +{ + if (zram_test_flag(zram, index, ZRAM_WB)) + return true; + if (zram_test_flag(zram, index, ZRAM_UNDER_WB)) + return true; + if (zram_test_flag(zram, index, ZRAM_BATCHING_OUT)) + return true; + if (zram_test_flag(zram, index, ZRAM_SAME)) + return true; + if (mcg != zram_fetch_mcg(zram, index)) + return true; + if (!zram_get_obj_size(zram, index)) + return true; + + return false; +} + +static bool zram_test_overwrite(struct zram *zram, u32 index, int eswapid) +{ + if (!zram_test_flag(zram, index, ZRAM_WB)) + return true; + if (eswapid != esentry_extid(zram_get_handle(zram, index))) + return true; + + return false; +} + +static void update_size_info(struct zram *zram, u32 index) +{ + struct hybstatus *stat; + int size = zram_get_obj_size(zram, index); + struct mem_cgroup *mcg; + memcg_hybs_t *hybs; + int eswapid; + + if (!zram_test_flag(zram, index, ZRAM_IN_BD)) + return; + + eswapid = esentry_extid(zram_get_handle(zram, index)); + hybp(HYB_DEBUG, "eswapid %d index %d\n", eswapid, index); + + if (eswapid >= 0 && eswapid < zram->infos->nr_es) + atomic_dec(&zram->infos->eswap_stored_pages[eswapid]); + else { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + eswapid = -1; + } + + stat = hybridswap_fetch_stat_obj(); + if (stat) { + atomic64_add(size, &stat->dropped_eswap_size); + atomic64_sub(size, &stat->stored_size); + atomic64_dec(&stat->stored_pages); + } else + hybp(HYB_ERR, "NULL stat\n"); + + mcg = zram_fetch_mcg(zram, index); + if (mcg) { + hybs = MEMCGRP_ITEM_DATA(mcg); + + if (hybs) { + atomic64_sub(size, &hybs->hybridswap_stored_size); + atomic64_dec(&hybs->hybridswap_stored_pages); + } else + hybp(HYB_ERR, "NULL hybs\n"); + } else + hybp(HYB_ERR, "NULL mcg\n"); + zram_clear_flag(zram, index, ZRAM_IN_BD); +} + +static void move_to_hybridswap(struct zram *zram, u32 index, + unsigned long eswpentry, struct mem_cgroup *mcg) +{ + int size; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return; + } + + size = zram_get_obj_size(zram, index); + + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + + zs_free(zram->mem_pool, zram_get_handle(zram, index)); + atomic64_sub(size, &zram->stats.compr_data_size); + atomic64_dec(&zram->stats.pages_stored); + + zram_set_mcg(zram, index, mcg->id.id); + zram_set_flag(zram, index, ZRAM_IN_BD); + zram_set_flag(zram, index, ZRAM_WB); + zram_set_obj_size(zram, index, size); + if (size == PAGE_SIZE) + zram_set_flag(zram, index, ZRAM_HUGE); + zram_set_handle(zram, index, eswpentry); + swap_maps_insert(zram, index); + + atomic64_add(size, &stat->stored_size); + atomic64_add(size, &MEMCGRP_ITEM(mcg, hybridswap_stored_size)); + atomic64_inc(&stat->stored_pages); + atomic_inc(&zram->infos->eswap_stored_pages[esentry_extid(eswpentry)]); + atomic64_inc(&MEMCGRP_ITEM(mcg, hybridswap_stored_pages)); +} + +static void __move_to_zram(struct zram *zram, u32 index, unsigned long handle, + struct io_eswapent *io_eswap) +{ + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + struct mem_cgroup *mcg = io_eswap->mcg; + int size = zram_get_obj_size(zram, index); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + + zram_slot_lock(zram, index); + if (zram_test_overwrite(zram, index, io_eswap->eswapid)) { + zram_slot_unlock(zram, index); + zs_free(zram->mem_pool, handle); + return; + } + swap_maps_destroy(zram, index); + zram_set_handle(zram, index, handle); + zram_clear_flag(zram, index, ZRAM_WB); + if (mcg) + swap_sorted_list_add_tail(zram, index, mcg); + zram_set_flag(zram, index, ZRAM_FROM_HYBRIDSWAP); + atomic64_add(size, &zram->stats.compr_data_size); + atomic64_inc(&zram->stats.pages_stored); + zram_clear_flag(zram, index, ZRAM_IN_BD); + zram_slot_unlock(zram, index); + + atomic64_inc(&stat->batchout_pages); + atomic64_sub(size, &stat->stored_size); + atomic64_dec(&stat->stored_pages); + atomic64_add(size, &stat->batchout_real_load); + atomic_dec(&zram->infos->eswap_stored_pages[io_eswap->eswapid]); + if (mcg) { + atomic64_sub(size, &MEMCGRP_ITEM(mcg, hybridswap_stored_size)); + atomic64_dec(&MEMCGRP_ITEM(mcg, hybridswap_stored_pages)); + } +} + +static int move_to_zram(struct zram *zram, u32 index, struct io_eswapent *io_eswap) +{ + unsigned long handle, eswpentry; + struct mem_cgroup *mcg = NULL; + int size, i; + u8 *dst = NULL; + + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return -EINVAL; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return -EINVAL; + } + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return -EINVAL; + } + + mcg = io_eswap->mcg; + zram_slot_lock(zram, index); + eswpentry = zram_get_handle(zram, index); + if (zram_test_overwrite(zram, index, io_eswap->eswapid)) { + zram_slot_unlock(zram, index); + return 0; + } + size = zram_get_obj_size(zram, index); + zram_slot_unlock(zram, index); + + for (i = esentry_pgid(eswpentry) - 1; i >= 0 && io_eswap->pages[i]; i--) { + hybridswap_page_recycle(io_eswap->pages[i], io_eswap->pool); + io_eswap->pages[i] = NULL; + } + handle = hybridswap_zsmalloc(zram->mem_pool, size, io_eswap->pool); + if (!handle) { + hybp(HYB_ERR, "alloc handle failed, size = %d\n", size); + return -ENOMEM; + } + dst = zs_map_object(zram->mem_pool, handle, ZS_MM_WO); + copy_from_pages(dst, io_eswap->pages, eswpentry, size); + zs_unmap_object(zram->mem_pool, handle); + __move_to_zram(zram, index, handle, io_eswap); + + return 0; +} + +static int eswap_unlock(struct io_eswapent *io_eswap) +{ + int eswapid; + struct mem_cgroup *mcg = NULL; + struct zram *zram = NULL; + int k; + unsigned long eswpentry; + int real_load = 0, size; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + goto out; + } + + mcg = io_eswap->mcg; + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + goto out; + } + zram = MEMCGRP_ITEM(mcg, zram); + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + goto out; + } + eswapid = io_eswap->eswapid; + if (eswapid < 0) + goto out; + + eswapid = io_eswap->eswapid; + if (MEMCGRP_ITEM(mcg, in_swapin)) + goto out; + hybp(HYB_DEBUG, "add eswapid = %d, cnt = %d.\n", + eswapid, io_eswap->cnt); + eswpentry = ((unsigned long)eswapid) << ESWAP_SHIFT; + for (k = 0; k < io_eswap->cnt; k++) + zram_slot_lock(zram, io_eswap->index[k]); + for (k = 0; k < io_eswap->cnt; k++) { + move_to_hybridswap(zram, io_eswap->index[k], eswpentry, mcg); + size = zram_get_obj_size(zram, io_eswap->index[k]); + eswpentry += size; + real_load += size; + } + put_eswap(zram->infos, eswapid); + io_eswap->eswapid = -EINVAL; + for (k = 0; k < io_eswap->cnt; k++) + zram_slot_unlock(zram, io_eswap->index[k]); + hybp(HYB_DEBUG, "add eswap OK.\n"); +out: + discard_io_eswapent(io_eswap, REQ_OP_WRITE); + if (mcg) + css_put(&mcg->css); + + return real_load; +} + +static void eswap_add(struct io_eswapent *io_eswap, + enum hybridswap_class class) +{ + struct mem_cgroup *mcg = NULL; + struct zram *zram = NULL; + int eswapid; + int k; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return; + } + + mcg = io_eswap->mcg; + if (!mcg) + zram = io_eswap->zram; + else + zram = MEMCGRP_ITEM(mcg, zram); + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + goto out; + } + + eswapid = io_eswap->eswapid; + if (eswapid < 0) + goto out; + + io_eswap->cnt = swap_maps_fetch_eswap_index(zram->infos, + eswapid, + io_eswap->index); + hybp(HYB_DEBUG, "eswapid = %d, cnt = %d.\n", eswapid, io_eswap->cnt); + for (k = 0; k < io_eswap->cnt; k++) { + int ret = move_to_zram(zram, io_eswap->index[k], io_eswap); + + if (ret < 0) + goto out; + } + hybp(HYB_DEBUG, "eswap add OK, free eswapid = %d.\n", eswapid); + hybridswap_free_eswap(zram->infos, io_eswap->eswapid); + io_eswap->eswapid = -EINVAL; + if (mcg) { + atomic64_inc(&MEMCGRP_ITEM(mcg, hybridswap_inextcnt)); + atomic_dec(&MEMCGRP_ITEM(mcg, hybridswap_extcnt)); + } +out: + discard_io_eswapent(io_eswap, REQ_OP_READ); + if (mcg) + css_put(&mcg->css); +} + +static void eswap_clear(struct zram *zram, int eswapid) +{ + int *index = NULL; + int cnt; + int k; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + + index = kzalloc(sizeof(int) * ESWAP_MAX_OBJ_CNT, GFP_NOIO); + if (!index) + index = kzalloc(sizeof(int) * ESWAP_MAX_OBJ_CNT, + GFP_NOIO | __GFP_NOFAIL); + + cnt = swap_maps_fetch_eswap_index(zram->infos, eswapid, index); + + for (k = 0; k < cnt; k++) { + zram_slot_lock(zram, index[k]); + if (zram_test_overwrite(zram, index[k], eswapid)) { + zram_slot_unlock(zram, index[k]); + continue; + } + zram_set_mcg(zram, index[k], 0); + zram_set_flag(zram, index[k], ZRAM_MCGID_CLEAR); + atomic64_inc(&stat->memcgid_clear); + zram_slot_unlock(zram, index[k]); + } + + kfree(index); +} + +static int shrink_entry(struct zram *zram, u32 index, struct io_eswapent *io_eswap, + unsigned long eswap_off) +{ + unsigned long handle; + int size; + u8 *src = NULL; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return -EINVAL; + } + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return -EINVAL; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return -EINVAL; + } + + zram_slot_lock(zram, index); + handle = zram_get_handle(zram, index); + if (!handle || zram_test_skip(zram, index, io_eswap->mcg)) { + zram_slot_unlock(zram, index); + return 0; + } + size = zram_get_obj_size(zram, index); + if (eswap_off + size > ESWAP_SIZE) { + zram_slot_unlock(zram, index); + return -ENOSPC; + } + + src = zs_map_object(zram->mem_pool, handle, ZS_MM_RO); + copy_to_pages(src, io_eswap->pages, eswap_off, size); + zs_unmap_object(zram->mem_pool, handle); + io_eswap->index[io_eswap->cnt++] = index; + + swap_sorted_list_del(zram, index); + zram_set_flag(zram, index, ZRAM_UNDER_WB); + if (zram_test_flag(zram, index, ZRAM_FROM_HYBRIDSWAP)) { + atomic64_inc(&stat->reout_pages); + atomic64_add(size, &stat->reout_bytes); + } + zram_slot_unlock(zram, index); + atomic64_inc(&stat->reclaimin_pages); + + return size; +} + +static int shrink_entry_list(struct io_eswapent *io_eswap) +{ + struct mem_cgroup *mcg = NULL; + struct zram *zram = NULL; + unsigned long stored_size; + int *swap_index = NULL; + int swap_cnt, k; + int swap_size = 0; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return -EINVAL; + } + + mcg = io_eswap->mcg; + zram = MEMCGRP_ITEM(mcg, zram); + hybp(HYB_DEBUG, "mcg = %d\n", mcg->id.id); + stored_size = atomic64_read(&MEMCGRP_ITEM(mcg, zram_stored_size)); + hybp(HYB_DEBUG, "zram_stored_size = %ld\n", stored_size); + if (stored_size < ESWAP_SIZE) { + hybp(HYB_INFO, "%lu is smaller than ESWAP_SIZE\n", stored_size); + return -ENOENT; + } + + swap_index = kzalloc(sizeof(int) * ESWAP_MAX_OBJ_CNT, GFP_NOIO); + if (!swap_index) + return -ENOMEM; + io_eswap->eswapid = hybridswap_alloc_eswap(zram->infos, mcg); + if (io_eswap->eswapid < 0) { + kfree(swap_index); + return io_eswap->eswapid; + } + swap_cnt = zram_fetch_mcg_last_index(zram->infos, mcg, swap_index, + ESWAP_MAX_OBJ_CNT); + io_eswap->cnt = 0; + for (k = 0; k < swap_cnt && swap_size < (int)ESWAP_SIZE; k++) { + int size = shrink_entry(zram, swap_index[k], io_eswap, swap_size); + + if (size < 0) + break; + swap_size += size; + } + kfree(swap_index); + hybp(HYB_DEBUG, "fill eswap = %d, cnt = %d, overhead = %ld.\n", + io_eswap->eswapid, io_eswap->cnt, ESWAP_SIZE - swap_size); + if (swap_size == 0) { + hybp(HYB_ERR, "swap_size = 0, free eswapid = %d.\n", + io_eswap->eswapid); + hybridswap_free_eswap(zram->infos, io_eswap->eswapid); + io_eswap->eswapid = -EINVAL; + return -ENOENT; + } + + return swap_size; +} + +void hybridswap_manager_deinit(struct zram *zram) +{ + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + + free_hyb_info(zram->infos); + zram->infos = NULL; +} + +int hybridswap_manager_init(struct zram *zram) +{ + int ret; + + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + ret = -EINVAL; + goto out; + } + + zram->infos = alloc_hyb_info(zram->disksize, + zram->nr_pages << PAGE_SHIFT); + if (!zram->infos) { + ret = -ENOMEM; + goto out; + } + return 0; +out: + hybridswap_manager_deinit(zram); + + return ret; +} + +void hybridswap_manager_memcg_init(struct zram *zram, + struct mem_cgroup *memcg) +{ + memcg_hybs_t *hybs; + + if (!memcg || !zram || !zram->infos) { + hybp(HYB_ERR, "invalid zram or mcg_hyb\n"); + return; + } + + hyb_entries_init(memcgindex(zram->infos, memcg->id.id), zram->infos->objects); + hyb_entries_init(memcgindex(zram->infos, memcg->id.id), zram->infos->eswap_table); + + hybs = MEMCGRP_ITEM_DATA(memcg); + hybs->in_swapin = false; + atomic64_set(&hybs->zram_stored_size, 0); + atomic64_set(&hybs->zram_page_size, 0); + atomic64_set(&hybs->hybridswap_stored_pages, 0); + atomic64_set(&hybs->hybridswap_stored_size, 0); + atomic64_set(&hybs->hybridswap_allfaultcnt, 0); + atomic64_set(&hybs->hybridswap_outcnt, 0); + atomic64_set(&hybs->hybridswap_incnt, 0); + atomic64_set(&hybs->hybridswap_faultcnt, 0); + atomic64_set(&hybs->hybridswap_outextcnt, 0); + atomic64_set(&hybs->hybridswap_inextcnt, 0); + atomic_set(&hybs->hybridswap_extcnt, 0); + atomic_set(&hybs->hybridswap_peakextcnt, 0); + mutex_init(&hybs->swap_lock); + + smp_wmb(); + hybs->zram = zram; + hybp(HYB_DEBUG, "new memcg in zram, id = %d.\n", memcg->id.id); +} + +void hybridswap_manager_memcg_deinit(struct mem_cgroup *mcg) +{ + struct zram *zram = NULL; + struct hyb_info *infos = NULL; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + int last_index = -1; + memcg_hybs_t *hybs; + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + + hybs = MEMCGRP_ITEM_DATA(mcg); + if (!hybs->zram) + return; + + zram = hybs->zram; + if (!zram->infos) { + hybp(HYB_WARN, "mcg %p name %s id %d zram %p infos is NULL\n", + mcg, hybs->name, mcg->id.id, zram); + return; + } + + hybp(HYB_DEBUG, "deinit mcg %d %s\n", mcg->id.id, hybs->name); + if (mcg->id.id == 0) + return; + + infos = zram->infos; + while (1) { + int index = fetch_memcg_zram_entry(infos, mcg); + + if (index == -ENOENT) + break; + + if (index < 0) { + hybp(HYB_ERR, "invalid index\n"); + return; + } + + if (last_index == index) { + hybp(HYB_ERR, "dup index %d\n", index); + dump_stack(); + } + + zram_slot_lock(zram, index); + if (index == last_index || mcg == zram_fetch_mcg(zram, index)) { + hyb_entries_del(obj_index(zram->infos, index), + memcgindex(zram->infos, mcg->id.id), + zram->infos->objects); + zram_set_mcg(zram, index, 0); + zram_set_flag(zram, index, ZRAM_MCGID_CLEAR); + atomic64_inc(&stat->memcgid_clear); + } + zram_slot_unlock(zram, index); + last_index = index; + } + + hybp(HYB_DEBUG, "deinit mcg %d %s, entry done\n", mcg->id.id, hybs->name); + while (1) { + int eswapid = fetch_memcg_eswap(infos, mcg); + + if (eswapid == -ENOENT) + break; + + eswap_clear(zram, eswapid); + hyb_entries_set_memcgid(eswap_index(infos, eswapid), infos->eswap_table, 0); + put_eswap(infos, eswapid); + } + hybp(HYB_DEBUG, "deinit mcg %d %s, eswap done\n", mcg->id.id, hybs->name); + hybs->zram = NULL; +} +void hybridswap_swap_sorted_list_add(struct zram *zram, + u32 index, struct mem_cgroup *memcg) +{ + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + + swap_sorted_list_add(zram, index, memcg); +} + +void hybridswap_swap_sorted_list_del(struct zram *zram, u32 index) +{ + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + + zram_clear_flag(zram, index, ZRAM_FROM_HYBRIDSWAP); + if (zram_test_flag(zram, index, ZRAM_MCGID_CLEAR)) { + zram_clear_flag(zram, index, ZRAM_MCGID_CLEAR); + atomic64_dec(&stat->memcgid_clear); + } + + if (zram_test_flag(zram, index, ZRAM_WB)) { + update_size_info(zram, index); + swap_maps_destroy(zram, index); + zram_clear_flag(zram, index, ZRAM_WB); + zram_set_mcg(zram, index, 0); + zram_set_handle(zram, index, 0); + } else { + swap_sorted_list_del(zram, index); + } +} + +unsigned long hybridswap_eswap_create(struct mem_cgroup *mcg, + int *eswapid, + struct hybridswap_buffer *buf, + void **private) +{ + struct io_eswapent *io_eswap = NULL; + int reclaim_size; + + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return 0; + } + if (!eswapid) { + hybp(HYB_ERR, "NULL eswapid\n"); + return 0; + } + (*eswapid) = -EINVAL; + if (!buf) { + hybp(HYB_ERR, "NULL buf\n"); + return 0; + } + if (!private) { + hybp(HYB_ERR, "NULL private\n"); + return 0; + } + + io_eswap = alloc_io_eswapent(buf->pool, false, true); + if (!io_eswap) + return 0; + io_eswap->mcg = mcg; + reclaim_size = shrink_entry_list(io_eswap); + if (reclaim_size < 0) { + discard_io_eswapent(io_eswap, REQ_OP_WRITE); + (*eswapid) = reclaim_size; + return 0; + } + io_eswap->real_load = reclaim_size; + css_get(&mcg->css); + (*eswapid) = io_eswap->eswapid; + buf->dest_pages = io_eswap->pages; + (*private) = io_eswap; + hybp(HYB_DEBUG, "mcg = %d, eswapid = %d\n", mcg->id.id, io_eswap->eswapid); + + return reclaim_size; +} + +void hybridswap_eswap_register(void *private, struct hybridswap_io_req *req) +{ + struct io_eswapent *io_eswap = private; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return; + } + hybp(HYB_DEBUG, "eswapid = %d\n", io_eswap->eswapid); + atomic64_add(eswap_unlock(io_eswap), &req->real_load); +} + +void hybridswap_eswap_objs_del(struct zram *zram, u32 index) +{ + int eswapid; + struct mem_cgroup *mcg = NULL; + unsigned long eswpentry; + int size; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + if (!zram || !zram->infos) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + if (!zram_test_flag(zram, index, ZRAM_WB)) { + hybp(HYB_ERR, "not WB object\n"); + return; + } + + eswpentry = zram_get_handle(zram, index); + size = zram_get_obj_size(zram, index); + atomic64_sub(size, &stat->stored_size); + atomic64_dec(&stat->stored_pages); + atomic64_add(size, &stat->dropped_eswap_size); + mcg = zram_fetch_mcg(zram, index); + if (mcg) { + atomic64_sub(size, &MEMCGRP_ITEM(mcg, hybridswap_stored_size)); + atomic64_dec(&MEMCGRP_ITEM(mcg, hybridswap_stored_pages)); + } + + zram_clear_flag(zram, index, ZRAM_IN_BD); + if (!atomic_dec_and_test( + &zram->infos->eswap_stored_pages[esentry_extid(eswpentry)])) + return; + eswapid = fetch_eswap(zram->infos, esentry_extid(eswpentry)); + if (eswapid < 0) + return; + + atomic64_inc(&stat->notify_free); + if (mcg) + atomic64_inc(&MEMCGRP_ITEM(mcg, hybridswap_eswap_notify_free)); + hybp(HYB_DEBUG, "free eswapid = %d\n", eswapid); + hybridswap_free_eswap(zram->infos, eswapid); +} + +int hybridswap_find_eswap_by_index(unsigned long eswpentry, + struct hybridswap_buffer *buf, + void **private) +{ + int eswapid; + struct io_eswapent *io_eswap = NULL; + struct zram *zram = NULL; + + if (!buf) { + hybp(HYB_ERR, "NULL buf\n"); + return -EINVAL; + } + if (!private) { + hybp(HYB_ERR, "NULL private\n"); + return -EINVAL; + } + + zram = buf->zram; + eswapid = fetch_eswap(zram->infos, esentry_extid(eswpentry)); + if (eswapid < 0) + return eswapid; + io_eswap = alloc_io_eswapent(buf->pool, true, true); + if (!io_eswap) { + hybp(HYB_ERR, "io_eswap alloc failed\n"); + put_eswap(zram->infos, eswapid); + return -ENOMEM; + } + + io_eswap->eswapid = eswapid; + io_eswap->zram = zram; + io_eswap->mcg = find_memcg_by_id( + hyb_entries_fetch_memcgid(eswap_index(zram->infos, eswapid), + zram->infos->eswap_table)); + if (io_eswap->mcg) + css_get(&io_eswap->mcg->css); + buf->dest_pages = io_eswap->pages; + (*private) = io_eswap; + hybp(HYB_DEBUG, "fetch entry = %lx eswap = %d\n", eswpentry, eswapid); + + return eswapid; +} + +int hybridswap_find_eswap_by_memcg(struct mem_cgroup *mcg, + struct hybridswap_buffer *buf, + void **private) +{ + int eswapid; + struct io_eswapent *io_eswap = NULL; + + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return -EINVAL; + } + if (!buf) { + hybp(HYB_ERR, "NULL buf\n"); + return -EINVAL; + } + if (!private) { + hybp(HYB_ERR, "NULL private\n"); + return -EINVAL; + } + + eswapid = fetch_memcg_eswap(MEMCGRP_ITEM(mcg, zram)->infos, mcg); + if (eswapid < 0) + return eswapid; + io_eswap = alloc_io_eswapent(buf->pool, true, false); + if (!io_eswap) { + hybp(HYB_ERR, "io_eswap alloc failed\n"); + put_eswap(MEMCGRP_ITEM(mcg, zram)->infos, eswapid); + return -ENOMEM; + } + io_eswap->eswapid = eswapid; + io_eswap->mcg = mcg; + css_get(&mcg->css); + buf->dest_pages = io_eswap->pages; + (*private) = io_eswap; + hybp(HYB_DEBUG, "fetch mcg = %d, eswap = %d\n", mcg->id.id, eswapid); + + return eswapid; +} + +void hybridswap_eswap_destroy(void *private, enum hybridswap_class class) +{ + struct io_eswapent *io_eswap = private; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return; + } + + hybp(HYB_DEBUG, "eswapid = %d\n", io_eswap->eswapid); + eswap_add(io_eswap, class); +} + +void hybridswap_eswap_exception(enum hybridswap_class class, + void *private) +{ + struct io_eswapent *io_eswap = private; + struct mem_cgroup *mcg = NULL; + unsigned int op = (class == HYB_RECLAIM_IN) ? + REQ_OP_WRITE : REQ_OP_READ; + + if (!io_eswap) { + hybp(HYB_ERR, "NULL io_eswap\n"); + return; + } + + hybp(HYB_DEBUG, "eswapid = %d, op = %d\n", io_eswap->eswapid, op); + mcg = io_eswap->mcg; + discard_io_eswapent(io_eswap, op); + if (mcg) + css_put(&mcg->css); +} + +struct mem_cgroup *hybridswap_zram_fetch_mcg(struct zram *zram, u32 index) +{ + return zram_fetch_mcg(zram, index); +} + +void zram_set_mcg(struct zram *zram, u32 index, int memcgid) +{ + hyb_entries_set_memcgid(obj_index(zram->infos, index), + zram->infos->objects, memcgid); +} + +struct mem_cgroup *zram_fetch_mcg(struct zram *zram, u32 index) +{ + unsigned short memcgid; + + memcgid = hyb_entries_fetch_memcgid(obj_index(zram->infos, index), + zram->infos->objects); + + return find_memcg_by_id(memcgid); +} + +int zram_fetch_mcg_last_index(struct hyb_info *infos, + struct mem_cgroup *mcg, + int *index, int max_cnt) +{ + int cnt = 0; + u32 i, tmp; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return 0; + } + if (!infos->objects) { + hybp(HYB_ERR, "NULL table\n"); + return 0; + } + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return 0; + } + if (!index) { + hybp(HYB_ERR, "NULL index\n"); + return 0; + } + + hyb_lock_with_idx(memcgindex(infos, mcg->id.id), infos->objects); + hyb_entries_for_each_entry_reverse_safe(i, tmp, + memcgindex(infos, mcg->id.id), infos->objects) { + if (i >= (u32)infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", i); + continue; + } + index[cnt++] = i; + if (cnt >= max_cnt) + break; + } + hyb_unlock_with_idx(memcgindex(infos, mcg->id.id), infos->objects); + + return cnt; +} + +int swap_maps_fetch_eswap_index(struct hyb_info *infos, + int eswapid, int *index) +{ + int cnt = 0; + u32 i; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return 0; + } + if (!infos->objects) { + hybp(HYB_ERR, "NULL table\n"); + return 0; + } + if (!index) { + hybp(HYB_ERR, "NULL index\n"); + return 0; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + return 0; + } + + hyb_lock_with_idx(eswap_index(infos, eswapid), infos->objects); + hyb_entries_for_each_entry(i, eswap_index(infos, eswapid), infos->objects) { + if (cnt >= (int)ESWAP_MAX_OBJ_CNT) { + WARN_ON_ONCE(1); + break; + } + index[cnt++] = i; + } + hyb_unlock_with_idx(eswap_index(infos, eswapid), infos->objects); + + return cnt; +} + +void swap_sorted_list_add(struct zram *zram, u32 index, struct mem_cgroup *memcg) +{ + unsigned long size; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + if (zram_test_flag(zram, index, ZRAM_WB)) { + hybp(HYB_ERR, "WB object, index = %d\n", index); + return; + } +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (zram_test_flag(zram, index, ZRAM_CACHED)) { + hybp(HYB_ERR, "CACHED object, index = %d\n", index); + return; + } + if (zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + hybp(HYB_ERR, "CACHED_COMPRESS object, index = %d\n", index); + return; + } +#endif + if (zram_test_flag(zram, index, ZRAM_SAME)) + return; + + zram_set_mcg(zram, index, memcg->id.id); + hyb_entries_add(obj_index(zram->infos, index), + memcgindex(zram->infos, memcg->id.id), + zram->infos->objects); + + size = zram_get_obj_size(zram, index); + + atomic64_add(size, &MEMCGRP_ITEM(memcg, zram_stored_size)); + atomic64_inc(&MEMCGRP_ITEM(memcg, zram_page_size)); + atomic64_add(size, &stat->zram_stored_size); + atomic64_inc(&stat->zram_stored_pages); +} + +void swap_sorted_list_add_tail(struct zram *zram, u32 index, struct mem_cgroup *mcg) +{ + unsigned long size; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (!mcg || !MEMCGRP_ITEM(mcg, zram) || !MEMCGRP_ITEM(mcg, zram)->infos) { + hybp(HYB_ERR, "invalid mcg\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + if (zram_test_flag(zram, index, ZRAM_WB)) { + hybp(HYB_ERR, "WB object, index = %d\n", index); + return; + } +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (zram_test_flag(zram, index, ZRAM_CACHED)) { + hybp(HYB_ERR, "CACHED object, index = %d\n", index); + return; + } + if (zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + hybp(HYB_ERR, "CACHED_COMPRESS object, index = %d\n", index); + return; + } +#endif + if (zram_test_flag(zram, index, ZRAM_SAME)) + return; + + zram_set_mcg(zram, index, mcg->id.id); + hyb_entries_add_tail(obj_index(zram->infos, index), + memcgindex(zram->infos, mcg->id.id), + zram->infos->objects); + + size = zram_get_obj_size(zram, index); + + atomic64_add(size, &MEMCGRP_ITEM(mcg, zram_stored_size)); + atomic64_inc(&MEMCGRP_ITEM(mcg, zram_page_size)); + atomic64_add(size, &stat->zram_stored_size); + atomic64_inc(&stat->zram_stored_pages); +} + +void swap_sorted_list_del(struct zram *zram, u32 index) +{ + struct mem_cgroup *mcg = NULL; + unsigned long size; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + if (!zram || !zram->infos) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + if (zram_test_flag(zram, index, ZRAM_WB)) { + hybp(HYB_ERR, "WB object, index = %d\n", index); + return; + } + + mcg = zram_fetch_mcg(zram, index); + if (!mcg || !MEMCGRP_ITEM(mcg, zram) || !MEMCGRP_ITEM(mcg, zram)->infos) + return; + if (zram_test_flag(zram, index, ZRAM_SAME)) + return; + + size = zram_get_obj_size(zram, index); + hyb_entries_del(obj_index(zram->infos, index), + memcgindex(zram->infos, mcg->id.id), + zram->infos->objects); + zram_set_mcg(zram, index, 0); + + atomic64_sub(size, &MEMCGRP_ITEM(mcg, zram_stored_size)); + atomic64_dec(&MEMCGRP_ITEM(mcg, zram_page_size)); + atomic64_sub(size, &stat->zram_stored_size); + atomic64_dec(&stat->zram_stored_pages); +} + +void swap_maps_insert(struct zram *zram, u32 index) +{ + unsigned long eswpentry; + u32 eswapid; + + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + + eswpentry = zram_get_handle(zram, index); + eswapid = esentry_extid(eswpentry); + hyb_entries_add_tail(obj_index(zram->infos, index), + eswap_index(zram->infos, eswapid), + zram->infos->objects); +} + +void swap_maps_destroy(struct zram *zram, u32 index) +{ + unsigned long eswpentry; + u32 eswapid; + + if (!zram) { + hybp(HYB_ERR, "NULL zram\n"); + return; + } + if (index >= (u32)zram->infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return; + } + + eswpentry = zram_get_handle(zram, index); + eswapid = esentry_extid(eswpentry); + hyb_entries_del(obj_index(zram->infos, index), + eswap_index(zram->infos, eswapid), + zram->infos->objects); +} + +static struct hyb_entries_head *fetch_node_default(int index, void *private) +{ + struct hyb_entries_head *table = private; + + return &table[index]; +} + +struct hyb_entries_table *alloc_table(struct hyb_entries_head *(*fetch_node)(int, void *), + void *private, gfp_t gfp) +{ + struct hyb_entries_table *table = + kmalloc(sizeof(struct hyb_entries_table), gfp); + + if (!table) + return NULL; + table->fetch_node = fetch_node ? fetch_node : fetch_node_default; + table->private = private; + + return table; +} + +void hyb_lock_with_idx(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK\n", index, table); + return; + } + bit_spin_lock(ENTRY_LOCK_BIT, (unsigned long *)node); +} + +void hyb_unlock_with_idx(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK\n", index, table); + return; + } + bit_spin_unlock(ENTRY_LOCK_BIT, (unsigned long *)node); +} + +bool hyb_entries_empty(int hindex, struct hyb_entries_table *table) +{ + bool ret = false; + + hyb_lock_with_idx(hindex, table); + ret = (prev_index(hindex, table) == hindex) && (next_index(hindex, table) == hindex); + hyb_unlock_with_idx(hindex, table); + + return ret; +} + +void hyb_entries_init(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pS func %pS\n", + index, table, table->fetch_node); + return; + } + memset(node, 0, sizeof(struct hyb_entries_head)); + node->prev = index; + node->next = index; +} + +void hyb_entries_add_nolock(int index, int hindex, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = NULL; + struct hyb_entries_head *head = NULL; + struct hyb_entries_head *next = NULL; + int nindex; + + node = index_node(index, table); + if (!node) { + hybp(HYB_ERR, + "NULL node, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + head = index_node(hindex, table); + if (!head) { + hybp(HYB_ERR, + "NULL head, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + next = index_node(head->next, table); + if (!next) { + hybp(HYB_ERR, + "NULL next, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + + nindex = head->next; + if (index != hindex) + hyb_lock_with_idx(index, table); + node->prev = hindex; + node->next = nindex; + if (index != hindex) + hyb_unlock_with_idx(index, table); + head->next = index; + if (nindex != hindex) + hyb_lock_with_idx(nindex, table); + next->prev = index; + if (nindex != hindex) + hyb_unlock_with_idx(nindex, table); +} + +void hyb_entries_add_tail_nolock(int index, int hindex, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = NULL; + struct hyb_entries_head *head = NULL; + struct hyb_entries_head *tail = NULL; + int tindex; + + node = index_node(index, table); + if (!node) { + hybp(HYB_ERR, + "NULL node, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + head = index_node(hindex, table); + if (!head) { + hybp(HYB_ERR, + "NULL head, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + tail = index_node(head->prev, table); + if (!tail) { + hybp(HYB_ERR, + "NULL tail, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + + tindex = head->prev; + if (index != hindex) + hyb_lock_with_idx(index, table); + node->prev = tindex; + node->next = hindex; + if (index != hindex) + hyb_unlock_with_idx(index, table); + head->prev = index; + if (tindex != hindex) + hyb_lock_with_idx(tindex, table); + tail->next = index; + if (tindex != hindex) + hyb_unlock_with_idx(tindex, table); +} + +void hyb_entries_del_nolock(int index, int hindex, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = NULL; + struct hyb_entries_head *prev = NULL; + struct hyb_entries_head *next = NULL; + int pindex, nindex; + + node = index_node(index, table); + if (!node) { + hybp(HYB_ERR, + "NULL node, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + prev = index_node(node->prev, table); + if (!prev) { + hybp(HYB_ERR, + "NULL prev, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + next = index_node(node->next, table); + if (!next) { + hybp(HYB_ERR, + "NULL next, index = %d, hindex = %d, table = %pK\n", + index, hindex, table); + return; + } + + if (index != hindex) + hyb_lock_with_idx(index, table); + pindex = node->prev; + nindex = node->next; + node->prev = index; + node->next = index; + if (index != hindex) + hyb_unlock_with_idx(index, table); + if (pindex != hindex) + hyb_lock_with_idx(pindex, table); + prev->next = nindex; + if (pindex != hindex) + hyb_unlock_with_idx(pindex, table); + if (nindex != hindex) + hyb_lock_with_idx(nindex, table); + next->prev = pindex; + if (nindex != hindex) + hyb_unlock_with_idx(nindex, table); +} + +void hyb_entries_add(int index, int hindex, struct hyb_entries_table *table) +{ + hyb_lock_with_idx(hindex, table); + hyb_entries_add_nolock(index, hindex, table); + hyb_unlock_with_idx(hindex, table); +} + +void hyb_entries_add_tail(int index, int hindex, struct hyb_entries_table *table) +{ + hyb_lock_with_idx(hindex, table); + hyb_entries_add_tail_nolock(index, hindex, table); + hyb_unlock_with_idx(hindex, table); +} + +void hyb_entries_del(int index, int hindex, struct hyb_entries_table *table) +{ + hyb_lock_with_idx(hindex, table); + hyb_entries_del_nolock(index, hindex, table); + hyb_unlock_with_idx(hindex, table); +} + +unsigned short hyb_entries_fetch_memcgid(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + int memcgid; + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK\n", index, table); + return 0; + } + + hyb_lock_with_idx(index, table); + memcgid = (node->mcg_left << ENTRY_MCG_SHIFT_HALF) | node->mcg_right; + hyb_unlock_with_idx(index, table); + + return memcgid; +} + +void hyb_entries_set_memcgid(int index, struct hyb_entries_table *table, int memcgid) +{ + struct hyb_entries_head *node = index_node(index, table); + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK, mcg = %d\n", + index, table, memcgid); + return; + } + + hyb_lock_with_idx(index, table); + node->mcg_left = (u32)memcgid >> ENTRY_MCG_SHIFT_HALF; + node->mcg_right = (u32)memcgid & ((1 << ENTRY_MCG_SHIFT_HALF) - 1); + hyb_unlock_with_idx(index, table); +} + +bool hyb_entries_set_priv(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + bool ret = false; + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK\n", index, table); + return false; + } + hyb_lock_with_idx(index, table); + ret = !test_and_set_bit(ENTRY_DATA_BIT, (unsigned long *)node); + hyb_unlock_with_idx(index, table); + + return ret; +} + +bool hyb_entries_test_priv(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + bool ret = false; + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK\n", index, table); + return false; + } + hyb_lock_with_idx(index, table); + ret = test_bit(ENTRY_DATA_BIT, (unsigned long *)node); + hyb_unlock_with_idx(index, table); + + return ret; +} + +bool hyb_entries_clear_priv(int index, struct hyb_entries_table *table) +{ + struct hyb_entries_head *node = index_node(index, table); + bool ret = false; + + if (!node) { + hybp(HYB_ERR, "index = %d, table = %pK\n", index, table); + return false; + } + + hyb_lock_with_idx(index, table); + ret = test_and_clear_bit(ENTRY_DATA_BIT, (unsigned long *)node); + hyb_unlock_with_idx(index, table); + + return ret; +} + +struct mem_cgroup *find_memcg_by_id(unsigned short memcgid) +{ + struct mem_cgroup *mcg = NULL; + + rcu_read_lock(); + mcg = mem_cgroup_from_id(memcgid); + rcu_read_unlock(); + + return mcg; +} + +static bool frag_info_dec(bool prev_flag, bool next_flag, + struct hybstatus *stat) +{ + if (prev_flag && next_flag) { + atomic64_inc(&stat->frag_cnt); + return false; + } + + if (prev_flag || next_flag) + return false; + + return true; +} + +static bool frag_info_inc(bool prev_flag, bool next_flag, + struct hybstatus *stat) +{ + if (prev_flag && next_flag) { + atomic64_dec(&stat->frag_cnt); + return false; + } + + if (prev_flag || next_flag) + return false; + + return true; +} + +static bool pre_is_conted(struct hyb_info *infos, int eswapid, int memcgid) +{ + int prev; + + if (is_first_index(eswap_index(infos, eswapid), memcgindex(infos, memcgid), + infos->eswap_table)) + return false; + prev = prev_index(eswap_index(infos, eswapid), infos->eswap_table); + + return (prev >= 0) && (eswap_index(infos, eswapid) == prev + 1); +} + +static bool ne_is_conted(struct hyb_info *infos, int eswapid, int memcgid) +{ + int next; + + if (is_last_index(eswap_index(infos, eswapid), memcgindex(infos, memcgid), + infos->eswap_table)) + return false; + next = next_index(eswap_index(infos, eswapid), infos->eswap_table); + + return (next >= 0) && (eswap_index(infos, eswapid) + 1 == next); +} + +static void eswap_frag_info_sub(struct hyb_info *infos, int eswapid) +{ + bool prev_flag = false; + bool next_flag = false; + int memcgid; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + + if (!infos->eswap_table) { + hybp(HYB_ERR, "NULL table\n"); + return; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + return; + } + + memcgid = hyb_entries_fetch_memcgid(eswap_index(infos, eswapid), infos->eswap_table); + if (memcgid <= 0 || memcgid >= infos->memcg_num) { + hybp(HYB_ERR, "memcgid = %d invalid\n", memcgid); + return; + } + + atomic64_dec(&stat->eswap_cnt); + infos->memcgid_cnt[memcgid]--; + if (infos->memcgid_cnt[memcgid] == 0) { + atomic64_dec(&stat->mcg_cnt); + atomic64_dec(&stat->frag_cnt); + return; + } + + prev_flag = pre_is_conted(infos, eswapid, memcgid); + next_flag = ne_is_conted(infos, eswapid, memcgid); + + if (frag_info_dec(prev_flag, next_flag, stat)) + atomic64_dec(&stat->frag_cnt); +} + +static void eswap_frag_info_add(struct hyb_info *infos, int eswapid) +{ + bool prev_flag = false; + bool next_flag = false; + int memcgid; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat) { + hybp(HYB_ERR, "NULL stat\n"); + return; + } + + if (!infos->eswap_table) { + hybp(HYB_ERR, "NULL table\n"); + return; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + return; + } + + memcgid = hyb_entries_fetch_memcgid(eswap_index(infos, eswapid), infos->eswap_table); + if (memcgid <= 0 || memcgid >= infos->memcg_num) { + hybp(HYB_ERR, "memcgid = %d invalid\n", memcgid); + return; + } + + atomic64_inc(&stat->eswap_cnt); + if (infos->memcgid_cnt[memcgid] == 0) { + infos->memcgid_cnt[memcgid]++; + atomic64_inc(&stat->frag_cnt); + atomic64_inc(&stat->mcg_cnt); + return; + } + infos->memcgid_cnt[memcgid]++; + + prev_flag = pre_is_conted(infos, eswapid, memcgid); + next_flag = ne_is_conted(infos, eswapid, memcgid); + + if (frag_info_inc(prev_flag, next_flag, stat)) + atomic64_inc(&stat->frag_cnt); +} + +static int eswap_bit2id(struct hyb_info *infos, int bit) +{ + if (bit < 0 || bit >= infos->nr_es) { + hybp(HYB_ERR, "bit = %d invalid\n", bit); + return -EINVAL; + } + + return infos->nr_es - bit - 1; +} + +static int eswap_id2bit(struct hyb_info *infos, int id) +{ + if (id < 0 || id >= infos->nr_es) { + hybp(HYB_ERR, "id = %d invalid\n", id); + return -EINVAL; + } + + return infos->nr_es - id - 1; +} + +int obj_index(struct hyb_info *infos, int index) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (index < 0 || index >= infos->total_objects) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return -EINVAL; + } + + return index; +} + +int eswap_index(struct hyb_info *infos, int index) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (index < 0 || index >= infos->nr_es) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return -EINVAL; + } + + return index + infos->total_objects; +} + +int memcgindex(struct hyb_info *infos, int index) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (index <= 0 || index >= infos->memcg_num) { + hybp(HYB_ERR, "index = %d invalid, memcg_num %d\n", index, + infos->memcg_num); + return -EINVAL; + } + + return index + infos->total_objects + infos->nr_es; +} + +static struct hyb_entries_head *fetch_objects_node(int index, void *private) +{ + struct hyb_info *infos = private; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return NULL; + } + if (index < 0) { + hybp(HYB_ERR, "index = %d invalid\n", index); + return NULL; + } + if (index < infos->total_objects) + return &infos->lru[index]; + index -= infos->total_objects; + if (index < infos->nr_es) + return &infos->maps[index]; + index -= infos->nr_es; + if (index > 0 && index < infos->memcg_num) { + struct mem_cgroup *mcg = find_memcg_by_id(index); + + if (!mcg) + goto err_out; + return (struct hyb_entries_head *)(&MEMCGRP_ITEM(mcg, swap_sorted_list)); + } +err_out: + hybp(HYB_ERR, "index = %d invalid, mcg is NULL\n", index); + + return NULL; +} + +static void free_obj_list_table(struct hyb_info *infos) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return; + } + + if (infos->lru) { + vfree(infos->lru); + infos->lru = NULL; + } + if (infos->maps) { + vfree(infos->maps); + infos->maps = NULL; + } + + kfree(infos->objects); + infos->objects = NULL; +} + +static int init_obj_list_table(struct hyb_info *infos) +{ + int i; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + + infos->lru = vzalloc(sizeof(struct hyb_entries_head) * infos->total_objects); + if (!infos->lru) { + hybp(HYB_ERR, "infos->lru alloc failed\n"); + goto err_out; + } + infos->maps = vzalloc(sizeof(struct hyb_entries_head) * infos->nr_es); + if (!infos->maps) { + hybp(HYB_ERR, "infos->maps alloc failed\n"); + goto err_out; + } + infos->objects = alloc_table(fetch_objects_node, infos, GFP_KERNEL); + if (!infos->objects) { + hybp(HYB_ERR, "infos->objects alloc failed\n"); + goto err_out; + } + for (i = 0; i < infos->total_objects; i++) + hyb_entries_init(obj_index(infos, i), infos->objects); + for (i = 0; i < infos->nr_es; i++) + hyb_entries_init(eswap_index(infos, i), infos->objects); + + hybp(HYB_INFO, "hybridswap obj list table init OK.\n"); + return 0; +err_out: + free_obj_list_table(infos); + hybp(HYB_ERR, "hybridswap obj list table init failed.\n"); + + return -ENOMEM; +} + +static struct hyb_entries_head *fetch_eswap_table_node(int index, void *private) +{ + struct hyb_info *infos = private; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return NULL; + } + + if (index < infos->total_objects) + goto err_out; + index -= infos->total_objects; + if (index < infos->nr_es) + return &infos->eswap[index]; + index -= infos->nr_es; + if (index > 0 && index < infos->memcg_num) { + struct mem_cgroup *mcg = find_memcg_by_id(index); + + if (!mcg) + return NULL; + return (struct hyb_entries_head *)(&MEMCGRP_ITEM(mcg, eswap_lru)); + } +err_out: + hybp(HYB_ERR, "index = %d invalid\n", index); + + return NULL; +} + +static void free_eswap_list_table(struct hyb_info *infos) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return; + } + + if (infos->eswap) + vfree(infos->eswap); + + kfree(infos->eswap_table); +} + +static int init_eswap_list_table(struct hyb_info *infos) +{ + int i; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + infos->eswap = vzalloc(sizeof(struct hyb_entries_head) * infos->nr_es); + if (!infos->eswap) + goto err_out; + infos->eswap_table = alloc_table(fetch_eswap_table_node, infos, GFP_KERNEL); + if (!infos->eswap_table) + goto err_out; + for (i = 0; i < infos->nr_es; i++) + hyb_entries_init(eswap_index(infos, i), infos->eswap_table); + hybp(HYB_INFO, "hybridswap eswap list table init OK.\n"); + return 0; +err_out: + free_eswap_list_table(infos); + hybp(HYB_ERR, "hybridswap eswap list table init failed.\n"); + + return -ENOMEM; +} + +void free_hyb_info(struct hyb_info *infos) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return; + } + + vfree(infos->bitmask); + vfree(infos->eswap_stored_pages); + free_obj_list_table(infos); + free_eswap_list_table(infos); + vfree(infos); +} + +struct hyb_info *alloc_hyb_info(unsigned long ori_size, + unsigned long comp_size) +{ + struct hyb_info *infos = vzalloc(sizeof(struct hyb_info)); + + if (!infos) { + hybp(HYB_ERR, "infos alloc failed\n"); + goto err_out; + } + if (comp_size & (ESWAP_SIZE - 1)) { + hybp(HYB_ERR, "disksize = %ld align invalid (32K align needed)\n", + comp_size); + goto err_out; + } + infos->size = comp_size; + infos->nr_es = comp_size >> ESWAP_SHIFT; + infos->memcg_num = MEM_CGROUP_ID_MAX; + infos->total_objects = ori_size >> PAGE_SHIFT; + infos->bitmask = vzalloc(BITS_TO_LONGS(infos->nr_es) * sizeof(long)); + if (!infos->bitmask) { + hybp(HYB_ERR, "infos->bitmask alloc failed, %lu\n", + BITS_TO_LONGS(infos->nr_es) * sizeof(long)); + goto err_out; + } + infos->eswap_stored_pages = vzalloc(sizeof(atomic_t) * infos->nr_es); + if (!infos->eswap_stored_pages) { + hybp(HYB_ERR, "infos->eswap_stored_pages alloc failed\n"); + goto err_out; + } + if (init_obj_list_table(infos)) { + hybp(HYB_ERR, "init obj list table failed\n"); + goto err_out; + } + if (init_eswap_list_table(infos)) { + hybp(HYB_ERR, "init eswap list table failed\n"); + goto err_out; + } + hybp(HYB_INFO, "infos %p size %lu nr_es %lu memcg_num %lu total_objects %lu\n", + infos, (unsigned long)infos->size, (unsigned long)infos->nr_es, (unsigned long)infos->memcg_num, + (unsigned long)infos->total_objects); + hybp(HYB_INFO, "hyb_info init OK.\n"); + return infos; +err_out: + free_hyb_info(infos); + hybp(HYB_ERR, "hyb_info init failed.\n"); + + return NULL; +} + +void hybridswap_check_infos_eswap(struct hyb_info *infos) +{ + int i; + + if (!infos) + return; + + for (i = 0; i < infos->nr_es; i++) { + int cnt = atomic_read(&infos->eswap_stored_pages[i]); + int eswapid = eswap_index(infos, i); + bool data = hyb_entries_test_priv(eswapid, infos->eswap_table); + int memcgid = hyb_entries_fetch_memcgid(eswapid, infos->eswap_table); + + if (cnt < 0 || (cnt > 0 && memcgid == 0)) + hybp(HYB_ERR, "%8d %8d %8d %8d %4d\n", i, cnt, eswapid, + memcgid, data); + } +} + +void hybridswap_free_eswap(struct hyb_info *infos, int eswapid) +{ + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "INVALID eswap %d\n", eswapid); + return; + } + hybp(HYB_DEBUG, "free eswap id = %d.\n", eswapid); + + hyb_entries_set_memcgid(eswap_index(infos, eswapid), infos->eswap_table, 0); + if (!test_and_clear_bit(eswap_id2bit(infos, eswapid), infos->bitmask)) { + hybp(HYB_ERR, "bit not set, eswap = %d\n", eswapid); + WARN_ON_ONCE(1); + } + atomic_dec(&infos->stored_exts); +} + +static int alloc_bitmask(unsigned long *bitmask, int max, int last_bit) +{ + int bit; + + if (!bitmask) { + hybp(HYB_ERR, "NULL bitmask.\n"); + return -EINVAL; + } +retry: + bit = find_next_zero_bit(bitmask, max, last_bit); + if (bit == max) { + if (last_bit == 0) { + hybp(HYB_ERR, "alloc bitmask failed.\n"); + return -ENOSPC; + } + last_bit = 0; + goto retry; + } + if (test_and_set_bit(bit, bitmask)) + goto retry; + + return bit; +} + +int hybridswap_alloc_eswap(struct hyb_info *infos, struct mem_cgroup *mcg) +{ + int last_bit; + int bit; + int eswapid; + int memcgid; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return -EINVAL; + } + + last_bit = atomic_read(&infos->last_alloc_bit); + hybp(HYB_DEBUG, "last_bit = %d.\n", last_bit); + bit = alloc_bitmask(infos->bitmask, infos->nr_es, last_bit); + if (bit < 0) { + hybp(HYB_ERR, "alloc bitmask failed.\n"); + return bit; + } + eswapid = eswap_bit2id(infos, bit); + memcgid = hyb_entries_fetch_memcgid(eswap_index(infos, eswapid), infos->eswap_table); + if (memcgid) { + hybp(HYB_ERR, "already has mcg %d, eswap %d\n", + memcgid, eswapid); + goto err_out; + } + hyb_entries_set_memcgid(eswap_index(infos, eswapid), infos->eswap_table, mcg->id.id); + + atomic_set(&infos->last_alloc_bit, bit); + atomic_inc(&infos->stored_exts); + hybp(HYB_DEBUG, "eswap %d init OK.\n", eswapid); + hybp(HYB_DEBUG, "memcgid = %d, eswap id = %d\n", mcg->id.id, eswapid); + + return eswapid; +err_out: + clear_bit(bit, infos->bitmask); + WARN_ON_ONCE(1); + return -EBUSY; +} + +int fetch_eswap(struct hyb_info *infos, int eswapid) +{ + int memcgid; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + return -EINVAL; + } + + if (!hyb_entries_clear_priv(eswap_index(infos, eswapid), infos->eswap_table)) + return -EBUSY; + memcgid = hyb_entries_fetch_memcgid(eswap_index(infos, eswapid), infos->eswap_table); + if (memcgid) { + eswap_frag_info_sub(infos, eswapid); + hyb_entries_del(eswap_index(infos, eswapid), memcgindex(infos, memcgid), + infos->eswap_table); + } + hybp(HYB_DEBUG, "eswap id = %d\n", eswapid); + + return eswapid; +} + +void put_eswap(struct hyb_info *infos, int eswapid) +{ + int memcgid; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + return; + } + + memcgid = hyb_entries_fetch_memcgid(eswap_index(infos, eswapid), infos->eswap_table); + if (memcgid) { + hyb_lock_with_idx(memcgindex(infos, memcgid), infos->eswap_table); + hyb_entries_add_nolock(eswap_index(infos, eswapid), memcgindex(infos, memcgid), + infos->eswap_table); + eswap_frag_info_add(infos, eswapid); + hyb_unlock_with_idx(memcgindex(infos, memcgid), infos->eswap_table); + } + if (!hyb_entries_set_priv(eswap_index(infos, eswapid), infos->eswap_table)) { + hybp(HYB_ERR, "private not set, eswap = %d\n", eswapid); + WARN_ON_ONCE(1); + return; + } + hybp(HYB_DEBUG, "put eswap %d.\n", eswapid); +} + +int fetch_memcg_eswap(struct hyb_info *infos, struct mem_cgroup *mcg) +{ + int memcgid; + int eswapid = -ENOENT; + int index; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (!infos->eswap_table) { + hybp(HYB_ERR, "NULL table\n"); + return -EINVAL; + } + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return -EINVAL; + } + + memcgid = mcg->id.id; + hyb_lock_with_idx(memcgindex(infos, memcgid), infos->eswap_table); + hyb_entries_for_each_entry(index, memcgindex(infos, memcgid), infos->eswap_table) + if (hyb_entries_clear_priv(index, infos->eswap_table)) { + eswapid = index - infos->total_objects; + break; + } + if (eswapid >= 0 && eswapid < infos->nr_es) { + eswap_frag_info_sub(infos, eswapid); + hyb_entries_del_nolock(index, memcgindex(infos, memcgid), infos->eswap_table); + hybp(HYB_DEBUG, "eswap id = %d\n", eswapid); + } + hyb_unlock_with_idx(memcgindex(infos, memcgid), infos->eswap_table); + + return eswapid; +} + +int fetch_memcg_zram_entry(struct hyb_info *infos, struct mem_cgroup *mcg) +{ + int memcgid, idx; + int index = -ENOENT; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (!infos->objects) { + hybp(HYB_ERR, "NULL table\n"); + return -EINVAL; + } + if (!mcg) { + hybp(HYB_ERR, "NULL mcg\n"); + return -EINVAL; + } + + memcgid = mcg->id.id; + hyb_lock_with_idx(memcgindex(infos, memcgid), infos->objects); + hyb_entries_for_each_entry(idx, memcgindex(infos, memcgid), infos->objects) { + index = idx; + break; + } + hyb_unlock_with_idx(memcgindex(infos, memcgid), infos->objects); + + return index; +} + +int fetch_eswap_zram_entry(struct hyb_info *infos, int eswapid) +{ + int index = -ENOENT; + int idx; + + if (!infos) { + hybp(HYB_ERR, "NULL infos\n"); + return -EINVAL; + } + if (!infos->objects) { + hybp(HYB_ERR, "NULL table\n"); + return -EINVAL; + } + if (eswapid < 0 || eswapid >= infos->nr_es) { + hybp(HYB_ERR, "eswap = %d invalid\n", eswapid); + return -EINVAL; + } + + hyb_lock_with_idx(eswap_index(infos, eswapid), infos->objects); + hyb_entries_for_each_entry(idx, eswap_index(infos, eswapid), infos->objects) { + index = idx; + break; + } + hyb_unlock_with_idx(eswap_index(infos, eswapid), infos->objects); + + return index; +} + +void *hybridswap_malloc(size_t size, bool fast, bool nofail) +{ + void *mem = NULL; + + if (likely(fast)) { + mem = kzalloc(size, GFP_ATOMIC); + if (likely(mem || !nofail)) + return mem; + } + + mem = kzalloc(size, GFP_NOIO); + + return mem; +} + +void hybridswap_free(const void *mem) +{ + kfree(mem); +} + +struct page *hybridswap_alloc_page_common(void *data, gfp_t gfp) +{ + struct page *page = NULL; + struct zs_eswap_para *eswap_para = (struct zs_eswap_para *)data; + + if (eswap_para->pool) { + spin_lock(&eswap_para->pool->page_pool_lock); + if (!list_empty(&eswap_para->pool->page_pool_list)) { + page = list_first_entry( + &eswap_para->pool->page_pool_list, + struct page, lru); + list_del(&page->lru); + } + spin_unlock(&eswap_para->pool->page_pool_lock); + } + + if (!page) { + if (eswap_para->fast) { + page = alloc_page(GFP_ATOMIC); + if (likely(page)) + goto out; + } + if (eswap_para->nofail) + page = alloc_page(GFP_NOIO); + else + page = alloc_page(gfp); + } +out: + return page; +} + +unsigned long hybridswap_zsmalloc(struct zs_pool *zs_pool, + size_t size, struct hybridswap_page_pool *pool) +{ + gfp_t gfp = __GFP_DIRECT_RECLAIM | __GFP_KSWAPD_RECLAIM | + __GFP_NOWARN | __GFP_HIGHMEM | __GFP_MOVABLE; + return zs_malloc(zs_pool, size, gfp); +} + +unsigned long zram_zsmalloc(struct zs_pool *zs_pool, size_t size, gfp_t gfp) +{ + return zs_malloc(zs_pool, size, gfp); +} + +struct page *hybridswap_alloc_page(struct hybridswap_page_pool *pool, + gfp_t gfp, bool fast, bool nofail) +{ + struct zs_eswap_para eswap_para; + + eswap_para.pool = pool; + eswap_para.fast = fast; + eswap_para.nofail = nofail; + + return hybridswap_alloc_page_common((void *)&eswap_para, gfp); +} + +void hybridswap_page_recycle(struct page *page, struct hybridswap_page_pool *pool) +{ + if (pool) { + spin_lock(&pool->page_pool_lock); + list_add(&page->lru, &pool->page_pool_list); + spin_unlock(&pool->page_pool_lock); + } else { + __free_page(page); + } +} + +bool hybridswap_out_to_eswap_enable(void) +{ + return !!atomic_read(&global_settings.out_to_eswap_enable); +} + +void hybridswap_set_out_to_eswap_disable(void) +{ + atomic_set(&global_settings.out_to_eswap_enable, false); +} + +void hybridswap_set_out_to_eswap_enable(bool en) +{ + atomic_set(&global_settings.out_to_eswap_enable, en ? 1 : 0); +} + +bool hybridswap_core_enabled(void) +{ + return !!atomic_read(&global_settings.enable); +} + +void hybridswap_set_enable(bool en) +{ + hybridswap_set_out_to_eswap_enable(en); + + if (!hybridswap_core_enabled()) + atomic_set(&global_settings.enable, en ? 1 : 0); +} + +struct hybstatus *hybridswap_fetch_stat_obj(void) +{ + return global_settings.stat; +} + +bool hybridswap_dev_life(void) +{ + return !!atomic_read(&global_settings.dev_life); +} + +void hybridswap_set_dev_life(bool en) +{ + atomic_set(&global_settings.dev_life, en ? 1 : 0); +} + +unsigned long hybridswap_quota_day(void) +{ + return global_settings.quota_day; +} + +void hybridswap_set_quota_day(unsigned long val) +{ + global_settings.quota_day = val; +} + +bool hybridswap_reach_life_protect(void) +{ + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + unsigned long quota = hybridswap_quota_day(); + + if (hybridswap_dev_life()) + quota /= 10; + return atomic64_read(&stat->reclaimin_bytes_daily) > quota; +} + +static void hybridswap_life_protect_ctrl_work(struct work_struct *work) +{ + struct tm tm; + struct timespec64 ts; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + ktime_get_real_ts64(&ts); + time64_to_tm(ts.tv_sec - sys_tz.tz_minuteswest * 60, 0, &tm); + + if (tm.tm_hour > 2) + atomic64_set(&stat->reclaimin_bytes_daily, 0); +} + +static void hybridswap_life_protect_ctrl_timer(struct timer_list *t) +{ + schedule_work(&global_settings.lpc_work); + mod_timer(&global_settings.lpc_timer, + jiffies + HYBRIDSWAP_CHECK_GAP * HZ); +} + +void hybridswap_close_bdev(struct block_device *bdev, struct file *backing_dev) +{ + if (bdev) + blkdev_put(bdev, FMODE_READ | FMODE_WRITE | FMODE_EXCL); + + if (backing_dev) + filp_close(backing_dev, NULL); +} + +struct file *hybridswap_open_bdev(const char *file_name) +{ + struct file *backing_dev = NULL; + +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 10, 0) + backing_dev = filp_open(file_name, O_RDWR|O_LARGEFILE, 0); +#else + backing_dev = filp_open_block(file_name, O_RDWR|O_LARGEFILE, 0); +#endif + if (unlikely(IS_ERR(backing_dev))) { + hybp(HYB_ERR, "open the %s failed! eno = %ld\n", + file_name, PTR_ERR(backing_dev)); + backing_dev = NULL; + return NULL; + } + + if (unlikely(!S_ISBLK(backing_dev->f_mapping->host->i_mode))) { + hybp(HYB_ERR, "%s isn't a blk device\n", file_name); + hybridswap_close_bdev(NULL, backing_dev); + return NULL; + } + + return backing_dev; +} + +int hybridswap_bind(struct zram *zram, const char *file_name) +{ + struct file *backing_dev = NULL; + struct inode *inode = NULL; + unsigned long nr_pages; + struct block_device *bdev = NULL; + int err; + + backing_dev = hybridswap_open_bdev(file_name); + if (unlikely(!backing_dev)) + return -EINVAL; + + inode = backing_dev->f_mapping->host; + bdev = blkdev_get_by_dev(inode->i_rdev, + FMODE_READ | FMODE_WRITE | FMODE_EXCL, zram); + if (IS_ERR(bdev)) { + hybp(HYB_ERR, "%s blkdev_fetch failed!\n", file_name); + err = PTR_ERR(bdev); + bdev = NULL; + goto out; + } + + nr_pages = (unsigned long)i_size_read(inode) >> PAGE_SHIFT; + err = set_blocksize(bdev, PAGE_SIZE); + if (unlikely(err)) { + hybp(HYB_ERR, + "%s set blocksize failed! eno = %d\n", file_name, err); + goto out; + } + + zram->bdev = bdev; + zram->backing_dev = backing_dev; + zram->nr_pages = nr_pages; + return 0; + +out: + hybridswap_close_bdev(bdev, backing_dev); + + return err; +} + +static inline unsigned long fetch_original_used_swap(void) +{ + struct sysinfo val; + + si_swapinfo(&val); + + return val.totalswap - val.freeswap; +} + +void hybstatus_init(struct hybstatus *stat) +{ + int i; + + atomic64_set(&stat->reclaimin_cnt, 0); + atomic64_set(&stat->reclaimin_bytes, 0); + atomic64_set(&stat->reclaimin_real_load, 0); + atomic64_set(&stat->dropped_eswap_size, 0); + atomic64_set(&stat->reclaimin_bytes_daily, 0); + atomic64_set(&stat->reclaimin_pages, 0); + atomic64_set(&stat->reclaimin_infight, 0); + atomic64_set(&stat->batchout_cnt, 0); + atomic64_set(&stat->batchout_bytes, 0); + atomic64_set(&stat->batchout_real_load, 0); + atomic64_set(&stat->batchout_pages, 0); + atomic64_set(&stat->batchout_inflight, 0); + atomic64_set(&stat->fault_cnt, 0); + atomic64_set(&stat->hybridswap_fault_cnt, 0); + atomic64_set(&stat->reout_pages, 0); + atomic64_set(&stat->reout_bytes, 0); + atomic64_set(&stat->zram_stored_pages, 0); + atomic64_set(&stat->zram_stored_size, 0); + atomic64_set(&stat->stored_pages, 0); + atomic64_set(&stat->stored_size, 0); + atomic64_set(&stat->notify_free, 0); + atomic64_set(&stat->frag_cnt, 0); + atomic64_set(&stat->mcg_cnt, 0); + atomic64_set(&stat->eswap_cnt, 0); + atomic64_set(&stat->miss_free, 0); + atomic64_set(&stat->memcgid_clear, 0); + atomic64_set(&stat->skip_track_cnt, 0); + atomic64_set(&stat->null_memcg_skip_track_cnt, 0); + atomic64_set(&stat->used_swap_pages, fetch_original_used_swap()); + atomic64_set(&stat->stored_wm_scale, DEFAULT_STORED_WM_RATIO); + + for (i = 0; i < HYB_CLASS_BUTT; ++i) { + atomic64_set(&stat->io_fail_cnt[i], 0); + atomic64_set(&stat->alloc_fail_cnt[i], 0); + atomic64_set(&stat->lat[i].total_lat, 0); + atomic64_set(&stat->lat[i].max_lat, 0); + } + + stat->record.num = 0; + spin_lock_init(&stat->record.lock); +} + +static bool hybridswap_global_setting_init(struct zram *zram) +{ + if (unlikely(global_settings.stat)) + return false; + + global_settings.zram = zram; + hybridswap_set_enable(false); + global_settings.stat = hybridswap_malloc( + sizeof(struct hybstatus), false, true); + if (unlikely(!global_settings.stat)) { + hybp(HYB_ERR, "global stat allocation failed!\n"); + return false; + } + + hybstatus_init(global_settings.stat); + global_settings.reclaim_wq = alloc_workqueue("hybridswap_reclaim", + WQ_CPU_INTENSIVE, 0); + if (unlikely(!global_settings.reclaim_wq)) { + hybp(HYB_ERR, "reclaim workqueue allocation failed!\n"); + hybridswap_free(global_settings.stat); + global_settings.stat = NULL; + + return false; + } + + global_settings.quota_day = HYBRIDSWAP_QUOTA_DAY; + INIT_WORK(&global_settings.lpc_work, hybridswap_life_protect_ctrl_work); + global_settings.lpc_timer.expires = jiffies + HYBRIDSWAP_CHECK_GAP * HZ; + timer_setup(&global_settings.lpc_timer, hybridswap_life_protect_ctrl_timer, 0); + add_timer(&global_settings.lpc_timer); + + hybp(HYB_DEBUG, "global settings init success\n"); + return true; +} + +void hybridswap_global_setting_deinit(void) +{ + destroy_workqueue(global_settings.reclaim_wq); + hybridswap_free(global_settings.stat); + global_settings.stat = NULL; + global_settings.zram = NULL; + global_settings.reclaim_wq = NULL; +} + +struct workqueue_struct *hybridswap_fetch_reclaim_workqueue(void) +{ + return global_settings.reclaim_wq; +} + +static int hybridswap_core_init(struct zram *zram) +{ + int ret; + + if (loop_device[0] == '\0') { + hybp(HYB_ERR, "please setting loop_device first\n"); + return -EINVAL; + } + + if (!hybridswap_global_setting_init(zram)) + return -EINVAL; + + ret = hybridswap_bind(zram, loop_device); + if (unlikely(ret)) { + hybp(HYB_ERR, "bind storage device failed! %d\n", ret); + hybridswap_global_setting_deinit(); + } + + return 0; +} + +int hybridswap_set_enable_init(bool en) +{ + int ret; + + if (hybridswap_core_enabled() || !en) + return 0; + + if (!global_settings.stat) { + hybp(HYB_ERR, "global_settings.stat is null!\n"); + + return -EINVAL; + } + + ret = hybridswap_manager_init(global_settings.zram); + if (unlikely(ret)) { + hybp(HYB_ERR, "init manager failed! %d\n", ret); + + return -EINVAL; + } + + ret = hyb_io_work_begin(); + if (unlikely(ret)) { + hybp(HYB_ERR, "init schedule failed! %d\n", ret); + hybridswap_manager_deinit(global_settings.zram); + + return -EINVAL; + } + + return 0; +} + +ssize_t hybridswap_core_enable_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned long val; + + ret = kstrtoul(buf, 0, &val); + if (unlikely(ret)) { + hybp(HYB_ERR, "val is error!\n"); + + return -EINVAL; + } + + if (hybridswap_set_enable_init(!!val)) + return -EINVAL; + + hybridswap_set_enable(!!val); + + return len; +} + +ssize_t hybridswap_core_enable_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int len = snprintf(buf, PAGE_SIZE, "hybridswap %s out_to_eswap %s\n", + hybridswap_core_enabled() ? "enable" : "disable", + hybridswap_out_to_eswap_enable() ? "enable" : "disable"); + + return len; +} + +ssize_t hybridswap_loop_device_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram; + int ret = 0; + + if (len > (DEVICE_NAME_LEN - 1)) { + hybp(HYB_ERR, "buf %s len %d is too long\n", buf, (int)len); + return -EINVAL; + } + + memcpy(loop_device, buf, len); + loop_device[len] = '\0'; + strstrip(loop_device); + + zram = dev_to_zram(dev); + down_write(&zram->init_lock); + if (zram->disksize == 0) { + hybp(HYB_ERR, "disksize is 0\n"); + goto out; + } + + ret = hybridswap_core_init(zram); + if (ret) + hybp(HYB_ERR, "hybridswap_core_init init failed\n"); + +out: + up_write(&zram->init_lock); + return len; +} + +ssize_t hybridswap_loop_device_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int len = 0; + + len = sprintf(buf, "%s\n", loop_device); + + return len; +} + +ssize_t hybridswap_dev_life_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned long val; + + ret = kstrtoul(buf, 0, &val); + if (unlikely(ret)) { + hybp(HYB_ERR, "val is error!\n"); + + return -EINVAL; + } + + hybridswap_set_dev_life(!!val); + + return len; +} + +ssize_t hybridswap_dev_life_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int len = 0; + + len = sprintf(buf, "%s\n", + hybridswap_dev_life() ? "enable" : "disable"); + + return len; +} + +ssize_t hybridswap_quota_day_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned long val; + + ret = kstrtoul(buf, 0, &val); + if (unlikely(ret)) { + hybp(HYB_ERR, "val is error!\n"); + + return -EINVAL; + } + + hybridswap_set_quota_day(val); + + return len; +} + +ssize_t hybridswap_quota_day_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int len = 0; + + len = sprintf(buf, "%lu\n", hybridswap_quota_day()); + + return len; +} + +ssize_t hybridswap_zram_increase_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + char *type_buf = NULL; + unsigned long val; + struct zram *zram = dev_to_zram(dev); + + type_buf = strstrip((char *)buf); + if (kstrtoul(type_buf, 0, &val)) + return -EINVAL; + + zram->increase_nr_pages = (val << 8); + return len; +} + +ssize_t hybridswap_zram_increase_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + ssize_t size = 0; + struct zram *zram = dev_to_zram(dev); + + size += scnprintf(buf + size, PAGE_SIZE - size, + "%lu\n", zram->increase_nr_pages >> 8); + + return size; +} + +int mem_cgroup_stored_wm_scale_write( + struct cgroup_subsys_state *css, struct cftype *cft, s64 val) +{ + if (val > MAX_RATIO || val < MIN_RATIO) + return -EINVAL; + + if (!global_settings.stat) + return -EINVAL; + + atomic64_set(&global_settings.stat->stored_wm_scale, val); + + return 0; +} + +s64 mem_cgroup_stored_wm_scale_read( + struct cgroup_subsys_state *css, struct cftype *cft) +{ + if (!global_settings.stat) + return -EINVAL; + + return atomic64_read(&global_settings.stat->stored_wm_scale); +} + +int hybridswap_stored_info(unsigned long *total, unsigned long *used) +{ + if (!total || !used) + return -EINVAL; + + if (!global_settings.stat || !global_settings.zram) { + *total = 0; + *used = 0; + return 0; + } + + *used = atomic64_read(&global_settings.stat->eswap_cnt) * ESWAP_PG_CNT; + *total = global_settings.zram->nr_pages; + + return 0; +} + +bool hybridswap_stored_wm_ok(void) +{ + unsigned long scale, stored_pages, total_pages, wm_scale; + int ret; + + if (!hybridswap_core_enabled()) + return false; + + ret = hybridswap_stored_info(&total_pages, &stored_pages); + if (ret) + return false; + + scale = (stored_pages * 100) / (total_pages + 1); + wm_scale = atomic64_read(&global_settings.stat->stored_wm_scale); + + return scale <= wm_scale; +} + +int hybridswap_core_enable(void) +{ + int ret; + + ret = hybridswap_set_enable_init(true); + if (ret) { + hybp(HYB_ERR, "set true failed, ret=%d\n", ret); + return ret; + } + + hybridswap_set_enable(true); + return 0; +} + +void hybridswap_core_disable(void) +{ + (void)hybridswap_set_enable_init(false); + hybridswap_set_enable(false); +} + +static void hybridswap_memcg_iter( + int (*iter)(struct mem_cgroup *, void *), void *data) +{ + struct mem_cgroup *mcg = fetch_next_memcg(NULL); + int ret; + + while (mcg) { + ret = iter(mcg, data); + hybp(HYB_DEBUG, "%pS mcg %d %s %s, ret %d\n", + iter, mcg->id.id, + MEMCGRP_ITEM(mcg, name), + ret ? "failed" : "pass", + ret); + if (ret) { + fetch_next_memcg_break(mcg); + return; + } + mcg = fetch_next_memcg(mcg); + } +} + +void hybridswap_record(struct zram *zram, u32 index, + struct mem_cgroup *memcg) +{ + memcg_hybs_t *hybs; + struct hybstatus *stat; + + if (!hybridswap_core_enabled()) + return; + + if (!memcg || !memcg->id.id) { + stat = hybridswap_fetch_stat_obj(); + if (stat) + atomic64_inc(&stat->null_memcg_skip_track_cnt); + return; + } + + hybs = MEMCGRP_ITEM_DATA(memcg); + if (!hybs) { + hybs = hybridswap_cache_alloc(memcg, false); + if (!hybs) { + stat = hybridswap_fetch_stat_obj(); + if (stat) + atomic64_inc(&stat->skip_track_cnt); + return; + } + } + + if (unlikely(!hybs->zram)) { + spin_lock(&hybs->zram_init_lock); + if (!hybs->zram) + hybridswap_manager_memcg_init(zram, memcg); + spin_unlock(&hybs->zram_init_lock); + } + + hybridswap_swap_sorted_list_add(zram, index, memcg); + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + zram_slot_unlock(zram, index); + if (!zram_watermark_ok()) + wake_all_swapd(); + zram_slot_lock(zram, index); +#endif +} + +void hybridswap_untrack(struct zram *zram, u32 index) +{ + if (!hybridswap_core_enabled()) + return; + + while (zram_test_flag(zram, index, ZRAM_UNDER_WB) || + zram_test_flag(zram, index, ZRAM_BATCHING_OUT)) { + zram_slot_unlock(zram, index); + udelay(50); + zram_slot_lock(zram, index); + } + + hybridswap_swap_sorted_list_del(zram, index); +} + +static unsigned long memcg_reclaim_size(struct mem_cgroup *memcg) +{ + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(memcg); + unsigned long zram_size, cur_size, new_size; + + if (!hybs) + return 0; + + zram_size = atomic64_read(&hybs->zram_stored_size); + if (hybs->force_swapout) { + hybs->can_eswaped = zram_size; + return zram_size; + } + + cur_size = atomic64_read(&hybs->hybridswap_stored_size); + new_size = (zram_size + cur_size) * + atomic_read(&hybs->zram2ufs_scale) / 100; + + hybs->can_eswaped = (new_size > cur_size) ? (new_size - cur_size) : 0; + return hybs->can_eswaped; +} + +static int hybridswap_permcg_sz(struct mem_cgroup *memcg, void *data) +{ + unsigned long *out_size = (unsigned long *)data; + + *out_size += memcg_reclaim_size(memcg); + return 0; +} + +static void hybridswap_flush_cb(enum hybridswap_class class, + void *pri, struct hybridswap_io_req *req) +{ + switch (class) { + case HYB_FAULT_OUT: + case HYB_PRE_OUT: + case HYB_BATCH_OUT: + hybridswap_eswap_destroy(pri, class); + break; + case HYB_RECLAIM_IN: + hybridswap_eswap_register(pri, req); + break; + default: + break; + } +} + +static void hybridswap_flush_done(struct hybridswap_entry *ioentry, + int err, struct hybridswap_io_req *req) +{ + struct io_priv *data; + + if (unlikely(!ioentry)) + return; + + data = (struct io_priv *)(ioentry->private); + if (likely(!err)) { + hybridswap_flush_cb(data->class, + ioentry->manager_private, req); +#ifdef CONFIG_HYBRIDSWAP_SWAPD + if (!zram_watermark_ok()) + wake_all_swapd(); +#endif + } else { + hybridswap_eswap_exception(data->class, + ioentry->manager_private); + } + hybridswap_free(ioentry); +} + +static void hybridswap_free_pagepool(struct io_work_arg *iowork) +{ + struct page *free_page = NULL; + + spin_lock(&iowork->data.page_pool.page_pool_lock); + while (!list_empty(&iowork->data.page_pool.page_pool_list)) { + free_page = list_first_entry( + &iowork->data.page_pool.page_pool_list, + struct page, lru); + list_del_init(&free_page->lru); + __free_page(free_page); + } + spin_unlock(&iowork->data.page_pool.page_pool_lock); +} + +static void hybridswap_plug_complete(void *data) +{ + struct io_work_arg *iowork = (struct io_work_arg *)data; + + hybridswap_free_pagepool(iowork); + + hybperf_end(&iowork->record); + + hybridswap_free(iowork); +} + +static void *hybridswap_init_plug(struct zram *zram, + enum hybridswap_class class, + struct io_work_arg *iowork) +{ + struct hybridswap_io io_para; + + io_para.bdev = zram->bdev; + io_para.class = class; + io_para.private = (void *)iowork; + io_para.record = &iowork->record; + INIT_LIST_HEAD(&iowork->data.page_pool.page_pool_list); + spin_lock_init(&iowork->data.page_pool.page_pool_lock); + io_para.done_callback = hybridswap_flush_done; + switch (io_para.class) { + case HYB_RECLAIM_IN: + io_para.complete_notify = hybridswap_plug_complete; + iowork->io_buf.pool = NULL; + break; + case HYB_BATCH_OUT: + case HYB_PRE_OUT: + io_para.complete_notify = hybridswap_plug_complete; + iowork->io_buf.pool = &iowork->data.page_pool; + break; + case HYB_FAULT_OUT: + io_para.complete_notify = NULL; + iowork->io_buf.pool = NULL; + break; + default: + break; + } + iowork->io_buf.zram = zram; + iowork->data.zram = zram; + iowork->data.class = io_para.class; + return hybridswap_plug_start(&io_para); +} + +static void hybridswap_fill_entry(struct hybridswap_entry *ioentry, + struct hybridswap_buffer *io_buf, + void *private) +{ + ioentry->addr = ioentry->eswapid * ESWAP_SECTOR_SIZE; + ioentry->dest_pages = io_buf->dest_pages; + ioentry->pages_sz = ESWAP_PG_CNT; + ioentry->private = private; +} + +static int hybridswap_reclaim_check(struct mem_cgroup *memcg, + unsigned long *require_size) +{ + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(memcg); + + if (unlikely(!hybs) || unlikely(!hybs->zram)) + return -EAGAIN; + if (unlikely(hybs->in_swapin)) + return -EAGAIN; + if (!hybs->force_swapout && *require_size < MIN_RECLAIM_ZRAM_SZ) + return -EAGAIN; + + return 0; +} + +static int hybridswap_update_reclaim_sz(unsigned long *require_size, + unsigned long *mcg_reclaimed_sz, + unsigned long reclaim_size) +{ + *mcg_reclaimed_sz += reclaim_size; + + if (*require_size > reclaim_size) + *require_size -= reclaim_size; + else + *require_size = 0; + + return 0; +} + +static void hybstatus_alloc_fail(enum hybridswap_class class, + int err) +{ + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (!stat || (err != -ENOMEM) || (class >= HYB_CLASS_BUTT)) + return; + + atomic64_inc(&stat->alloc_fail_cnt[class]); +} + +static int hybridswap_reclaim_eswap(struct mem_cgroup *memcg, + struct io_work_arg *iowork, + unsigned long *require_size, + unsigned long *mcg_reclaimed_sz, + int *errio) +{ + int ret; + unsigned long reclaim_size; + + hybperfiowrkstart(&iowork->record, HYB_IOENTRY_ALLOC); + iowork->ioentry = hybridswap_malloc( + sizeof(struct hybridswap_entry), false, true); + hybperfiowrkend(&iowork->record, HYB_IOENTRY_ALLOC); + if (unlikely(!iowork->ioentry)) { + hybp(HYB_ERR, "alloc io entry failed!\n"); + *require_size = 0; + *errio = -ENOMEM; + hybstatus_alloc_fail(HYB_RECLAIM_IN, -ENOMEM); + + return *errio; + } + + hybperfiowrkstart(&iowork->record, HYB_FIND_ESWAP); + reclaim_size = hybridswap_eswap_create( + memcg, &iowork->ioentry->eswapid, + &iowork->io_buf, &iowork->ioentry->manager_private); + hybperfiowrkend(&iowork->record, HYB_FIND_ESWAP); + if (unlikely(!reclaim_size)) { + if (iowork->ioentry->eswapid != -ENOENT) + *require_size = 0; + hybridswap_free(iowork->ioentry); + return -EAGAIN; + } + + hybridswap_fill_entry(iowork->ioentry, &iowork->io_buf, + (void *)(&iowork->data)); + + hybperfiowrkstart(&iowork->record, HYB_IO_ESWAP); + ret = hybridswap_write_eswap(iowork->iohandle, iowork->ioentry); + hybperfiowrkend(&iowork->record, HYB_IO_ESWAP); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap write failed! %d\n", ret); + *require_size = 0; + *errio = ret; + hybstatus_alloc_fail(HYB_RECLAIM_IN, ret); + + return *errio; + } + + ret = hybridswap_update_reclaim_sz(require_size, mcg_reclaimed_sz, + reclaim_size); + if (MEMCGRP_ITEM(memcg, force_swapout)) + return 0; + return ret; +} + +static int hybridswap_permcg_reclaim(struct mem_cgroup *memcg, + unsigned long require_size, unsigned long *mcg_reclaimed_sz) +{ + int ret, extcnt; + int errio = 0; + unsigned long require_size_before = 0; + struct io_work_arg *iowork = NULL; + ktime_t start = ktime_get(); + unsigned long long start_ravg_sum = hybridswap_fetch_ravg_sum(); + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(memcg); + + ret = hybridswap_reclaim_check(memcg, &require_size); + if (ret) + return ret == -EAGAIN ? 0 : ret; + + iowork = hybridswap_malloc(sizeof(struct io_work_arg), false, true); + if (unlikely(!iowork)) { + hybp(HYB_ERR, "alloc iowork failed!\n"); + hybstatus_alloc_fail(HYB_RECLAIM_IN, -ENOMEM); + + return -ENOMEM; + } + + hybperf_start(&iowork->record, start, start_ravg_sum, + HYB_RECLAIM_IN); + hybperfiowrkstart(&iowork->record, HYB_INIT); + iowork->iohandle = hybridswap_init_plug(hybs->zram, + HYB_RECLAIM_IN, iowork); + hybperfiowrkend(&iowork->record, HYB_INIT); + if (unlikely(!iowork->iohandle)) { + hybp(HYB_ERR, "plug start failed!\n"); + hybperf_end(&iowork->record); + hybridswap_free(iowork); + hybstatus_alloc_fail(HYB_RECLAIM_IN, -ENOMEM); + ret = -EIO; + goto out; + } + + require_size_before = require_size; + while (require_size) { + if (hybridswap_reclaim_eswap(memcg, iowork, + &require_size, mcg_reclaimed_sz, &errio)) + break; + + atomic64_inc(&hybs->hybridswap_outextcnt); + extcnt = atomic_inc_return(&hybs->hybridswap_extcnt); + if (extcnt > atomic_read(&hybs->hybridswap_peakextcnt)) + atomic_set(&hybs->hybridswap_peakextcnt, extcnt); + } + + ret = hybridswap_plug_finish(iowork->iohandle); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap write flush failed! %d\n", ret); + hybstatus_alloc_fail(HYB_RECLAIM_IN, ret); + require_size = 0; + } else { + ret = errio; + } + atomic64_inc(&hybs->hybridswap_outcnt); + +out: + hybp(HYB_INFO, "memcg %s %lu %lu out_to_eswap %lu KB eswap %lu zram %lu %d\n", + hybs->name, (unsigned long)require_size_before, (unsigned long)require_size, + (unsigned long)((require_size_before - require_size) >> 10), + (unsigned long)atomic64_read(&hybs->hybridswap_stored_size), + (unsigned long)atomic64_read(&hybs->zram_stored_size), ret); + return ret; +} + +static void hybridswap_reclaimin_inc(void) +{ + struct hybstatus *stat; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) + return; + atomic64_inc(&stat->reclaimin_infight); +} + +static void hybridswap_reclaimin_dec(void) +{ + struct hybstatus *stat; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) + return; + atomic64_dec(&stat->reclaimin_infight); +} + +static int hybridswap_permcg_reclaimin(struct mem_cgroup *memcg, + void *data) +{ + struct async_req *rq = (struct async_req *)data; + unsigned long mcg_reclaimed_size = 0, require_size; + int ret; + memcg_hybs_t *hybs; + + hybs = MEMCGRP_ITEM_DATA(memcg); + if (!hybs) + return 0; + + require_size = hybs->can_eswaped * rq->size / rq->out_size; + if (require_size < MIN_RECLAIM_ZRAM_SZ) + return 0; + + if (!mutex_trylock(&hybs->swap_lock)) + return 0; + + ret = hybridswap_permcg_reclaim(memcg, require_size, + &mcg_reclaimed_size); + rq->reclaimined_sz += mcg_reclaimed_size; + mutex_unlock(&hybs->swap_lock); + + hybp(HYB_INFO, "memcg %s mcg_reclaimed_size %lu rq->reclaimined_sz %lu rq->size %lu rq->out_size %lu ret %d\n", + hybs->name, mcg_reclaimed_size, rq->reclaimined_sz, + rq->size, rq->out_size, ret); + + if (!ret && rq->reclaimined_sz >= rq->size) + return -EINVAL; + + return ret; +} + +static void hybridswap_reclaim_work(struct work_struct *work) +{ + struct async_req *rq = container_of(work, struct async_req, work); + int old_nice = task_nice(current); + + set_user_nice(current, rq->nice); + hybridswap_reclaimin_inc(); + hybridswap_memcg_iter(hybridswap_permcg_reclaimin, rq); + hybridswap_reclaimin_dec(); + set_user_nice(current, old_nice); + hybp(HYB_INFO, "SWAPOUT want %lu MB real %lu Mb\n", rq->size >> 20, + rq->reclaimined_sz >> 20); + hybridswap_free(rq); +} + +unsigned long hybridswap_out_to_eswap(unsigned long size) +{ + struct async_req *rq = NULL; + unsigned long out_size = 0; + + if (!hybridswap_core_enabled() || !hybridswap_out_to_eswap_enable() + || hybridswap_reach_life_protect() || !size) + return 0; + + hybridswap_memcg_iter(hybridswap_permcg_sz, &out_size); + if (!out_size) + return 0; + + rq = hybridswap_malloc(sizeof(struct async_req), false, true); + if (unlikely(!rq)) { + hybp(HYB_ERR, "alloc async req fail!\n"); + hybstatus_alloc_fail(HYB_RECLAIM_IN, -ENOMEM); + return 0; + } + + if (out_size < size) + size = out_size; + rq->size = size; + rq->out_size = out_size; + rq->reclaimined_sz = 0; + rq->nice = task_nice(current); + INIT_WORK(&rq->work, hybridswap_reclaim_work); + queue_work(hybridswap_fetch_reclaim_workqueue(), &rq->work); + + return out_size > size ? size : out_size; +} + +static int hybridswap_batches_eswap(struct io_work_arg *iowork, + struct mem_cgroup *mcg, + bool preload, + int *errio) +{ + int ret; + + hybperfiowrkstart(&iowork->record, HYB_IOENTRY_ALLOC); + iowork->ioentry = hybridswap_malloc( + sizeof(struct hybridswap_entry), !preload, preload); + hybperfiowrkend(&iowork->record, HYB_IOENTRY_ALLOC); + if (unlikely(!iowork->ioentry)) { + hybp(HYB_ERR, "alloc io entry failed!\n"); + *errio = -ENOMEM; + hybstatus_alloc_fail(HYB_BATCH_OUT, -ENOMEM); + + return *errio; + } + + hybperfiowrkstart(&iowork->record, HYB_FIND_ESWAP); + iowork->ioentry->eswapid = hybridswap_find_eswap_by_memcg( + mcg, &iowork->io_buf, + &iowork->ioentry->manager_private); + hybperfiowrkend(&iowork->record, HYB_FIND_ESWAP); + if (iowork->ioentry->eswapid < 0) { + hybstatus_alloc_fail(HYB_BATCH_OUT, + iowork->ioentry->eswapid); + hybridswap_free(iowork->ioentry); + return -EAGAIN; + } + + hybridswap_fill_entry(iowork->ioentry, &iowork->io_buf, + (void *)(&iowork->data)); + + hybperfiowrkstart(&iowork->record, HYB_IO_ESWAP); + ret = hybridswap_read_eswap(iowork->iohandle, iowork->ioentry); + hybperfiowrkend(&iowork->record, HYB_IO_ESWAP); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap read failed! %d\n", ret); + hybstatus_alloc_fail(HYB_BATCH_OUT, ret); + *errio = ret; + + return *errio; + } + + return 0; +} + +static int hybridswap_do_batches_init(struct io_work_arg **out_sched, + struct mem_cgroup *mcg, bool preload) +{ + struct io_work_arg *iowork = NULL; + ktime_t start = ktime_get(); + unsigned long long start_ravg_sum = hybridswap_fetch_ravg_sum(); + + iowork = hybridswap_malloc(sizeof(struct io_work_arg), + !preload, preload); + if (unlikely(!iowork)) { + hybp(HYB_ERR, "alloc iowork failed!\n"); + hybstatus_alloc_fail(HYB_BATCH_OUT, -ENOMEM); + + return -ENOMEM; + } + + hybperf_start(&iowork->record, start, start_ravg_sum, + preload ? HYB_PRE_OUT : HYB_BATCH_OUT); + + hybperfiowrkstart(&iowork->record, HYB_INIT); + iowork->iohandle = hybridswap_init_plug(MEMCGRP_ITEM(mcg, zram), + preload ? HYB_PRE_OUT : HYB_BATCH_OUT, + iowork); + hybperfiowrkend(&iowork->record, HYB_INIT); + if (unlikely(!iowork->iohandle)) { + hybp(HYB_ERR, "plug start failed!\n"); + hybperf_end(&iowork->record); + hybridswap_free(iowork); + hybstatus_alloc_fail(HYB_BATCH_OUT, -ENOMEM); + + return -EIO; + } + + *out_sched = iowork; + + return 0; +} + +static int hybridswap_do_batches(struct mem_cgroup *mcg, + unsigned long size, bool preload) +{ + int ret = 0; + int errio = 0; + struct io_work_arg *iowork = NULL; + + if (unlikely(!mcg || !MEMCGRP_ITEM(mcg, zram))) { + hybp(HYB_WARN, "no zram in mcg!\n"); + ret = -EINVAL; + goto out; + } + + ret = hybridswap_do_batches_init(&iowork, mcg, preload); + if (unlikely(ret)) + goto out; + + MEMCGRP_ITEM(mcg, in_swapin) = true; + while (size) { + if (hybridswap_batches_eswap(iowork, mcg, preload, &errio)) + break; + size -= ESWAP_SIZE; + } + + ret = hybridswap_plug_finish(iowork->iohandle); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap read flush failed! %d\n", ret); + hybstatus_alloc_fail(HYB_BATCH_OUT, ret); + } else { + ret = errio; + } + + if (atomic64_read(&MEMCGRP_ITEM(mcg, hybridswap_stored_size)) && + hybridswap_loglevel() >= HYB_INFO) + hybridswap_check_infos_eswap((MEMCGRP_ITEM(mcg, zram)->infos)); + + atomic64_inc(&MEMCGRP_ITEM(mcg, hybridswap_incnt)); + MEMCGRP_ITEM(mcg, in_swapin) = false; +out: + return ret; +} + +static void hybridswap_batchout_inc(void) +{ + struct hybstatus *stat; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) + return; + atomic64_inc(&stat->batchout_inflight); +} + +static void hybridswap_batchout_dec(void) +{ + struct hybstatus *stat; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) + return; + atomic64_dec(&stat->batchout_inflight); +} + +int hybridswap_batches(struct mem_cgroup *mcg, + unsigned long size, bool preload) +{ + int ret; + + if (!hybridswap_core_enabled()) + return 0; + + hybridswap_batchout_inc(); + ret = hybridswap_do_batches(mcg, size, preload); + hybridswap_batchout_dec(); + + return ret; +} + +static void hybridswap_fault_stat(struct zram *zram, u32 index) +{ + struct mem_cgroup *mcg = NULL; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (unlikely(!stat)) + return; + + atomic64_inc(&stat->fault_cnt); + + mcg = hybridswap_zram_fetch_mcg(zram, index); + if (mcg) + atomic64_inc(&MEMCGRP_ITEM(mcg, hybridswap_allfaultcnt)); +} + +static void hybridswap_fault2_stat(struct zram *zram, u32 index) +{ + struct mem_cgroup *mcg = NULL; + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (unlikely(!stat)) + return; + + atomic64_inc(&stat->hybridswap_fault_cnt); + + mcg = hybridswap_zram_fetch_mcg(zram, index); + if (mcg) + atomic64_inc(&MEMCGRP_ITEM(mcg, hybridswap_faultcnt)); +} + +static bool hybridswap_page_fault_check(struct zram *zram, + u32 index, unsigned long *zentry) +{ + if (!hybridswap_core_enabled()) + return false; + + hybridswap_fault_stat(zram, index); + + if (!zram_test_flag(zram, index, ZRAM_WB)) + return false; + + zram_set_flag(zram, index, ZRAM_BATCHING_OUT); + *zentry = zram_get_handle(zram, index); + zram_slot_unlock(zram, index); + return true; +} + +static int hybridswap_page_fault_fetch_eswap(struct zram *zram, + struct io_work_arg *iowork, + unsigned long zentry, + u32 index) +{ + int wait_cycle = 0; + + iowork->io_buf.zram = zram; + iowork->data.zram = zram; + iowork->io_buf.pool = NULL; + hybperfiowrkstart(&iowork->record, HYB_FIND_ESWAP); + iowork->ioentry->eswapid = hybridswap_find_eswap_by_index(zentry, + &iowork->io_buf, &iowork->ioentry->manager_private); + hybperfiowrkend(&iowork->record, HYB_FIND_ESWAP); + if (unlikely(iowork->ioentry->eswapid == -EBUSY)) { + while (1) { + zram_slot_lock(zram, index); + if (!zram_test_flag(zram, index, ZRAM_WB)) { + zram_slot_unlock(zram, index); + hybridswap_free(iowork->ioentry); +#ifdef CONFIG_HYBRIDSWAP_SWAPD + if (wait_cycle >= 1000) + atomic_long_dec(&page_fault_pause); +#endif + return -EAGAIN; + } + zram_slot_unlock(zram, index); + + hybperfiowrkstart(&iowork->record, + HYB_FIND_ESWAP); + iowork->ioentry->eswapid = + hybridswap_find_eswap_by_index(zentry, + &iowork->io_buf, + &iowork->ioentry->manager_private); + hybperfiowrkend(&iowork->record, + HYB_FIND_ESWAP); + if (likely(iowork->ioentry->eswapid != -EBUSY)) + break; + + if (wait_cycle < 100) + udelay(50); + else + usleep_range(50, 100); + wait_cycle++; +#ifdef CONFIG_HYBRIDSWAP_SWAPD + if (wait_cycle == 1000) { + atomic_long_inc(&page_fault_pause); + atomic_long_inc(&page_fault_pause_cnt); + } +#endif + } + } +#ifdef CONFIG_HYBRIDSWAP_SWAPD + if (wait_cycle >= 1000) + atomic_long_dec(&page_fault_pause); +#endif + if (iowork->ioentry->eswapid < 0) { + hybstatus_alloc_fail(HYB_FAULT_OUT, + iowork->ioentry->eswapid); + + return iowork->ioentry->eswapid; + } + hybridswap_fault2_stat(zram, index); + hybridswap_fill_entry(iowork->ioentry, &iowork->io_buf, + (void *)(&iowork->data)); + return 0; +} + +static int hybridswap_page_fault_exit_check(struct zram *zram, + u32 index, int ret) +{ + zram_slot_lock(zram, index); + if (likely(!ret)) { + if (unlikely(zram_test_flag(zram, index, ZRAM_WB))) { + hybp(HYB_ERR, "still in WB status!\n"); + ret = -EIO; + } + } + zram_clear_flag(zram, index, ZRAM_BATCHING_OUT); + + return ret; +} + +static int hybridswap_page_fault_eswap(struct zram *zram, u32 index, + struct io_work_arg *iowork, unsigned long zentry) +{ + int ret; + + hybperfiowrkstart(&iowork->record, HYB_IOENTRY_ALLOC); + iowork->ioentry = hybridswap_malloc(sizeof(struct hybridswap_entry), + true, true); + hybperfiowrkend(&iowork->record, HYB_IOENTRY_ALLOC); + if (unlikely(!iowork->ioentry)) { + hybp(HYB_ERR, "alloc io entry failed!\n"); + hybstatus_alloc_fail(HYB_FAULT_OUT, -ENOMEM); + hybridswap_fail_record(HYB_FAULT_OUT_ENTRY_ALLOC_FAIL, + index, 0, iowork->record.task_comm); + return -ENOMEM; + } + + ret = hybridswap_page_fault_fetch_eswap(zram, iowork, zentry, index); + if (ret) + return ret; + + hybperfiowrkstart(&iowork->record, HYB_IO_ESWAP); + ret = hybridswap_read_eswap(iowork->iohandle, iowork->ioentry); + hybperfiowrkend(&iowork->record, HYB_IO_ESWAP); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap read failed! %d\n", ret); + hybstatus_alloc_fail(HYB_FAULT_OUT, ret); + } + + return ret; +} + +int hybridswap_page_fault(struct zram *zram, u32 index) +{ + int ret = 0; + int errio; + struct io_work_arg iowork; + unsigned long zentry; + ktime_t start = ktime_get(); + unsigned long long start_ravg_sum = hybridswap_fetch_ravg_sum(); + + if (!hybridswap_page_fault_check(zram, index, &zentry)) + return ret; + + memset(&iowork.record, 0, sizeof(struct hybridswap_key_point_record)); + hybperf_start(&iowork.record, start, start_ravg_sum, + HYB_FAULT_OUT); + + hybperfiowrkstart(&iowork.record, HYB_INIT); + iowork.iohandle = hybridswap_init_plug(zram, + HYB_FAULT_OUT, &iowork); + hybperfiowrkend(&iowork.record, HYB_INIT); + if (unlikely(!iowork.iohandle)) { + hybp(HYB_ERR, "plug start failed!\n"); + hybstatus_alloc_fail(HYB_FAULT_OUT, -ENOMEM); + ret = -EIO; + hybridswap_fail_record(HYB_FAULT_OUT_INIT_FAIL, + index, 0, iowork.record.task_comm); + + goto out; + } + + errio = hybridswap_page_fault_eswap(zram, index, &iowork, zentry); + ret = hybridswap_plug_finish(iowork.iohandle); + if (unlikely(ret)) { + hybp(HYB_ERR, "hybridswap flush failed! %d\n", ret); + hybstatus_alloc_fail(HYB_FAULT_OUT, ret); + } else { + ret = (errio != -EAGAIN) ? errio : 0; + } +out: + hybperfiowrkstart(&iowork.record, HYB_ZRAM_LOCK); + ret = hybridswap_page_fault_exit_check(zram, index, ret); + hybperfiowrkend(&iowork.record, HYB_ZRAM_LOCK); + hybperf_end(&iowork.record); + + return ret; +} + +bool hybridswap_delete(struct zram *zram, u32 index) +{ + if (!hybridswap_core_enabled()) + return true; + + if (zram_test_flag(zram, index, ZRAM_UNDER_WB) + || zram_test_flag(zram, index, ZRAM_BATCHING_OUT)) { + struct hybstatus *stat = hybridswap_fetch_stat_obj(); + + if (stat) + atomic64_inc(&stat->miss_free); + return false; + } + + if (!zram_test_flag(zram, index, ZRAM_WB)) + return true; + + hybridswap_eswap_objs_del(zram, index); + + return true; +} + +void hybridswap_mem_cgroup_deinit(struct mem_cgroup *memcg) +{ + if (!hybridswap_core_enabled()) + return; + + hybridswap_manager_memcg_deinit(memcg); +} + +void hybridswap_force_reclaim(struct mem_cgroup *mcg) +{ + unsigned long mcg_reclaimed_size = 0, require_size; + memcg_hybs_t *hybs; + + if (!hybridswap_core_enabled() || !hybridswap_out_to_eswap_enable() + || hybridswap_reach_life_protect()) + return; + + if (!mcg) + return; + + hybs = MEMCGRP_ITEM_DATA(mcg); + if (!hybs || !hybs->zram) + return; + + mutex_lock(&hybs->swap_lock); + require_size = atomic64_read(&hybs->zram_stored_size); + hybs->force_swapout = true; + hybridswap_permcg_reclaim(mcg, require_size, &mcg_reclaimed_size); + hybs->force_swapout = false; + mutex_unlock(&hybs->swap_lock); +} + +void mem_cgroup_id_remove_hook(void *data, struct mem_cgroup *memcg) +{ + if (!memcg->android_oem_data1) + return; + + hybridswap_mem_cgroup_deinit(memcg); + hybp(HYB_DEBUG, "hybridswap remove mcg id = %d\n", memcg->id.id); +} diff --git a/drivers/moto_swap/hybridswap/hybridswap_internal.h b/drivers/moto_swap/hybridswap/hybridswap_internal.h new file mode 100644 index 000000000000..9eea587f921f --- /dev/null +++ b/drivers/moto_swap/hybridswap/hybridswap_internal.h @@ -0,0 +1,560 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (C) 2020-2022 Oplus. All rights reserved. + */ + +#ifndef HYBRIDSWAP_INTERNAL_H +#define HYBRIDSWAP_INTERNAL_H + +#include +#include +#include +#include +#include + +#define ESWAP_SHIFT 15 +#define ESWAP_SIZE (1UL << ESWAP_SHIFT) +#define ESWAP_PG_CNT (ESWAP_SIZE >> PAGE_SHIFT) +#define ESWAP_SECTOR_SIZE (ESWAP_PG_CNT << 3) +#define ESWAP_MAX_OBJ_CNT (30 * ESWAP_PG_CNT) +#define ESWAP_MASK (~(ESWAP_SIZE - 1)) +#define ESWAP_ALIGN_UP(size) ((size + ESWAP_SIZE - 1) & ESWAP_MASK) + +#define MAX_FAIL_RECORD_NUM 4 +#define MAX_APP_GRADE 600 + +#define HYBRIDSWAP_QUOTA_DAY 0x280000000 /* 10G bytes */ +#define HYBRIDSWAP_CHECK_GAP 86400 /* 24 hour */ +#define MEM_CGROUP_NAME_MAX_LEN 32 + +#define MAX_RATIO 100 +#define MIN_RATIO 0 + +enum { + HYB_ERR = 0, + HYB_WARN, + HYB_INFO, + HYB_DEBUG, + HYB_MAX +}; + +void hybridswap_loglevel_set(int level); +int hybridswap_loglevel(void); + +#define DUMP_STACK_ON_ERR 0 +#define pt(l, f, ...) pr_err("[%s][%s]:"f, #l, __func__, ##__VA_ARGS__) +static inline void pr_none(void) {} +#define hybp(l, f, ...) do {\ + (l <= hybridswap_loglevel()) ? pt(l, f, ##__VA_ARGS__) : pr_none();\ + if (DUMP_STACK_ON_ERR && l == HYB_ERR) dump_stack();\ +} while (0) + +enum hybridswap_class { + HYB_RECLAIM_IN = 0, + HYB_FAULT_OUT, + HYB_BATCH_OUT, + HYB_PRE_OUT, + HYB_CLASS_BUTT +}; + +enum hybridswap_key_point { + HYB_START = 0, + HYB_INIT, + HYB_IOENTRY_ALLOC, + HYB_FIND_ESWAP, + HYB_IO_ESWAP, + HYB_SEGMENT_ALLOC, + HYB_BIO_ALLOC, + HYB_SUBMIT_BIO, + HYB_END_IO, + HYB_SCHED_WORK, + HYB_END_WORK, + HYB_CALL_BACK, + HYB_WAKE_UP, + HYB_ZRAM_LOCK, + HYB_DONE, + HYB_KYE_POINT_BUTT +}; + +enum hybridswap_mcg_member { + MCG_ZRAM_STORED_SZ = 0, + MCG_ZRAM_STORED_PG_SZ, + MCG_DISK_STORED_SZ, + MCG_DISK_STORED_PG_SZ, + MCG_ANON_FAULT_CNT, + MCG_DISK_FAULT_CNT, + MCG_ESWAPOUT_CNT, + MCG_ESWAPOUT_SZ, + MCG_ESWAPIN_CNT, + MCG_ESWAPIN_SZ, + MCG_DISK_SPACE, + MCG_DISK_SPACE_PEAK, +}; + +enum hybridswap_fail_point { + HYB_FAULT_OUT_INIT_FAIL = 0, + HYB_FAULT_OUT_ENTRY_ALLOC_FAIL, + HYB_FAULT_OUT_IO_ENTRY_PARA_FAIL, + HYB_FAULT_OUT_SEGMENT_ALLOC_FAIL, + HYB_FAULT_OUT_BIO_ALLOC_FAIL, + HYB_FAULT_OUT_BIO_ADD_FAIL, + HYB_FAULT_OUT_IO_FAIL, + HYBRIDSWAP_FAIL_POINT_BUTT +}; + +struct hybridswap_fail_record { + unsigned char task_comm[TASK_COMM_LEN]; + enum hybridswap_fail_point point; + ktime_t time; + u32 index; + int eswapid; +}; + +struct hybridswap_fail_record_info { + int num; + spinlock_t lock; + struct hybridswap_fail_record record[MAX_FAIL_RECORD_NUM]; +}; + +struct hybridswap_key_point_info { + unsigned int record_cnt; + unsigned int end_cnt; + ktime_t first_time; + ktime_t last_time; + s64 proc_total_time; + s64 proc_max_time; + unsigned long long last_ravg_sum; + unsigned long long proc_ravg_sum; + spinlock_t time_lock; +}; + +struct hybridswap_key_point_record { + struct timer_list lat_monitor; + unsigned long warn_level; + int page_cnt; + int segment_cnt; + int nice; + bool timeout_flag; + unsigned char task_comm[TASK_COMM_LEN]; + struct task_struct *task; + enum hybridswap_class class; + struct hybridswap_key_point_info key_point[HYB_KYE_POINT_BUTT]; +}; + +struct hybridswapiowrkstat { + atomic64_t total_lat; + atomic64_t max_lat; + atomic64_t timeout_cnt; +}; + +struct hybridswap_fault_timeout_cnt{ + atomic64_t timeout_100ms_cnt; + atomic64_t timeout_500ms_cnt; +}; + +struct hybstatus { + atomic64_t reclaimin_cnt; + atomic64_t reclaimin_bytes; + atomic64_t reclaimin_real_load; + atomic64_t reclaimin_bytes_daily; + atomic64_t reclaimin_pages; + atomic64_t reclaimin_infight; + atomic64_t batchout_cnt; + atomic64_t batchout_bytes; + atomic64_t batchout_real_load; + atomic64_t batchout_pages; + atomic64_t batchout_inflight; + atomic64_t fault_cnt; + atomic64_t hybridswap_fault_cnt; + atomic64_t reout_pages; + atomic64_t reout_bytes; + atomic64_t zram_stored_pages; + atomic64_t zram_stored_size; + atomic64_t stored_pages; + atomic64_t stored_size; + atomic64_t notify_free; + atomic64_t frag_cnt; + atomic64_t mcg_cnt; + atomic64_t eswap_cnt; + atomic64_t miss_free; + atomic64_t memcgid_clear; + atomic64_t skip_track_cnt; + atomic64_t used_swap_pages; + atomic64_t null_memcg_skip_track_cnt; + atomic64_t stored_wm_scale; + atomic64_t dropped_eswap_size; + atomic64_t io_fail_cnt[HYB_CLASS_BUTT]; + atomic64_t alloc_fail_cnt[HYB_CLASS_BUTT]; + struct hybridswapiowrkstat lat[HYB_CLASS_BUTT]; + struct hybridswap_fault_timeout_cnt fault_stat[2]; /* 0:bg 1:fg */ + struct hybridswap_fail_record_info record; +}; + +struct hybridswap_page_pool { + struct list_head page_pool_list; + spinlock_t page_pool_lock; +}; + +struct io_eswapent { + int eswapid; + struct zram *zram; + struct mem_cgroup *mcg; + struct page *pages[ESWAP_PG_CNT]; + u32 index[ESWAP_MAX_OBJ_CNT]; + int cnt; + int real_load; + + struct hybridswap_page_pool *pool; +}; + +struct hybridswap_buffer { + struct zram *zram; + struct hybridswap_page_pool *pool; + struct page **dest_pages; +}; + +struct hybridswap_entry { + int eswapid; + sector_t addr; + struct page **dest_pages; + int pages_sz; + struct list_head list; + void *private; + void *manager_private; +}; + +struct hybridswap_io_req; +struct hybridswap_io { + struct block_device *bdev; + enum hybridswap_class class; + void (*done_callback)(struct hybridswap_entry *, int, struct hybridswap_io_req *); + void (*complete_notify)(void *); + void *private; + struct hybridswap_key_point_record *record; +}; + +struct hybridswap_io_req { + struct hybridswap_io io_para; + struct kref refcount; + struct mutex refmutex; + struct wait_queue_head io_wait; + atomic_t eswap_doing; + struct completion io_end_flag; + struct hyb_sgm *segment; + bool limit_doing_flag; + bool wait_io_finish_flag; + int page_cnt; + int segment_cnt; + int nice; + atomic64_t real_load; +}; + +/* Change hybridswap_event_item, you should change swapd_text togather*/ +enum hybridswap_event_item { +#ifdef CONFIG_HYBRIDSWAP_SWAPD + SWAPD_WAKEUP, + SWAPD_REFAULT, + SWAPD_MEMCG_RATIO_SKIP, + SWAPD_MEMCG_REFAULT_SKIP, + SWAPD_SHRINK_ANON, + SWAPD_SWAPOUT, + SWAPD_SKIP_SWAPOUT, + SWAPD_EMPTY_ROUND, + SWAPD_OVER_MIN_BUFFER_SKIP_TIMES, + SWAPD_EMPTY_ROUND_SKIP_TIMES, + SWAPD_SNAPSHOT_TIMES, + SWAPD_SKIP_SHRINK_OF_WINDOW, + SWAPD_MANUAL_PAUSE, +#ifdef CONFIG_OPLUS_JANK + SWAPD_CPU_BUSY_SKIP_TIMES, + SWAPD_CPU_BUSY_BREAK_TIMES, +#endif +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + AKCOMPRESSD_WAKEUP, +#endif + NR_EVENT_ITEMS +}; + +struct swapd_event_state { + unsigned long event[NR_EVENT_ITEMS]; +}; + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +struct cgroup_cache_page { + spinlock_t lock; + struct list_head head; + unsigned int cnt; + int id; + char compressing; + char dead; +}; +#endif + +typedef struct mem_cgroup_hybridswap { +#ifdef CONFIG_HYBRIDSWAP + atomic64_t ufs2zram_scale; + atomic_t zram2ufs_scale; + atomic64_t app_grade; + atomic64_t app_uid; + struct list_head grade_node; + char name[MEM_CGROUP_NAME_MAX_LEN]; + struct zram *zram; + struct mem_cgroup *memcg; + refcount_t usage; +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD + atomic_t mem2zram_scale; + atomic_t pagefault_level; + unsigned long long reclaimed_pagefault; + long long can_reclaimed; +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + unsigned long swap_sorted_list; + unsigned long eswap_lru; + struct list_head link_list; + spinlock_t zram_init_lock; + long long can_eswaped; + + atomic64_t zram_stored_size; + atomic64_t zram_page_size; + unsigned long zram_watermark; + + atomic_t hybridswap_extcnt; + atomic_t hybridswap_peakextcnt; + + atomic64_t hybridswap_stored_pages; + atomic64_t hybridswap_stored_size; + atomic64_t hybridswap_eswap_notify_free; + + atomic64_t hybridswap_outcnt; + atomic64_t hybridswap_incnt; + atomic64_t hybridswap_allfaultcnt; + atomic64_t hybridswap_faultcnt; + + atomic64_t hybridswap_outextcnt; + atomic64_t hybridswap_inextcnt; + + struct mutex swap_lock; + bool in_swapin; + bool force_swapout; +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + struct cgroup_cache_page cache; +#endif +}memcg_hybs_t; + +#define MEMCGRP_ITEM_DATA(memcg) ((memcg_hybs_t *)(memcg)->android_oem_data1) +#define MEMCGRP_ITEM(memcg, item) (MEMCGRP_ITEM_DATA(memcg)->item) + +extern void __put_memcg_cache(memcg_hybs_t *hybs); + +static inline memcg_hybs_t *fetch_memcg_cache(memcg_hybs_t *hybs) +{ + refcount_inc(&hybs->usage); + return hybs; +} + +static inline void put_memcg_cache(memcg_hybs_t *hybs) +{ + if (refcount_dec_and_test(&hybs->usage)) + __put_memcg_cache(hybs); +} + +DECLARE_PER_CPU(struct swapd_event_state, swapd_event_states); +extern struct mutex reclaim_para_lock; + +static inline void __count_swapd_event(enum hybridswap_event_item item) +{ + raw_cpu_inc(swapd_event_states.event[item]); +} + +static inline void count_swapd_event(enum hybridswap_event_item item) +{ + this_cpu_inc(swapd_event_states.event[item]); +} + +static inline void __count_swapd_events(enum hybridswap_event_item item, long delta) +{ + raw_cpu_add(swapd_event_states.event[item], delta); +} + +static inline void count_swapd_events(enum hybridswap_event_item item, long delta) +{ + this_cpu_add(swapd_event_states.event[item], delta); +} + +void *hybridswap_malloc(size_t size, bool fast, bool nofail); +void hybridswap_free(const void *mem); +unsigned long hybridswap_zsmalloc(struct zs_pool *zs_pool, + size_t size, struct hybridswap_page_pool *pool); +struct page *hybridswap_alloc_page( + struct hybridswap_page_pool *pool, gfp_t gfp, + bool fast, bool nofail); +void hybridswap_page_recycle(struct page *page, + struct hybridswap_page_pool *pool); +struct hybstatus *hybridswap_fetch_stat_obj(void); +int hybridswap_manager_init(struct zram *zram); +void hybridswap_manager_memcg_init(struct zram *zram, + struct mem_cgroup *memcg); +void hybridswap_manager_memcg_deinit(struct mem_cgroup *mcg); +void hybridswap_swap_sorted_list_add(struct zram *zram, u32 index, + struct mem_cgroup *memcg); +void hybridswap_swap_sorted_list_del(struct zram *zram, u32 index); +unsigned long hybridswap_eswap_create(struct mem_cgroup *memcg, + int *eswapid, + struct hybridswap_buffer *dest_buf, + void **private); +void hybridswap_eswap_register(void *private, struct hybridswap_io_req *req); +void hybridswap_eswap_objs_del(struct zram *zram, u32 index); +int hybridswap_find_eswap_by_index( + unsigned long eswpentry, struct hybridswap_buffer *buf, void **private); +int hybridswap_find_eswap_by_memcg( + struct mem_cgroup *mcg, + struct hybridswap_buffer *dest_buf, void **private); +void hybridswap_eswap_destroy(void *private, enum hybridswap_class class); +void hybridswap_eswap_exception(enum hybridswap_class class, + void *private); +void hybridswap_manager_deinit(struct zram *zram); +struct mem_cgroup *hybridswap_zram_fetch_mcg(struct zram *zram, u32 index); +int hyb_io_work_begin(void); +void *hybridswap_plug_start(struct hybridswap_io *io_para); +int hybridswap_read_eswap(void *iohandle, + struct hybridswap_entry *ioentry); +int hybridswap_write_eswap(void *iohandle, + struct hybridswap_entry *ioentry); + +int hybridswap_plug_finish(void *iohandle); + +void hybperf_start( + struct hybridswap_key_point_record *record, + ktime_t stsrt, unsigned long long start_ravg_sum, + enum hybridswap_class class); + +void hybperf_end(struct hybridswap_key_point_record *record); + +void hybperfiowrkstart( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type); + +void hybperfiowrkend( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type); + +void hybperfiowrkpoint( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type); + +void hybperf_async_perf( + struct hybridswap_key_point_record *record, + enum hybridswap_key_point type, ktime_t start, + unsigned long long start_ravg_sum); + +void hybperf_io_stat( + struct hybridswap_key_point_record *record, int page_cnt, + int segment_cnt); + +static inline unsigned long long hybridswap_fetch_ravg_sum(void) +{ + return 0; +} + +void hybridswap_fail_record(enum hybridswap_fail_point point, + u32 index, int eswapid, unsigned char *task_comm); +bool hybridswap_reach_life_protect(void); +struct workqueue_struct *hybridswap_fetch_reclaim_workqueue(void); +extern struct mem_cgroup *fetch_next_memcg(struct mem_cgroup *prev); +extern void fetch_next_memcg_break(struct mem_cgroup *prev); +extern memcg_hybs_t *hybridswap_cache_alloc(struct mem_cgroup *memcg, bool atomic); +extern void memcg_app_grade_resort(void); +extern unsigned long memcg_anon_pages(struct mem_cgroup *memcg); + +#ifdef CONFIG_HYBRIDSWAP_CORE +extern bool hybridswap_core_enabled(void); +extern bool hybridswap_out_to_eswap_enable(void); +extern void hybridswap_mem_cgroup_deinit(struct mem_cgroup *memcg); +extern unsigned long hybridswap_out_to_eswap(unsigned long size); +extern int hybridswap_batches(struct mem_cgroup *mcg, + unsigned long size, bool preload); +extern unsigned long zram_zsmalloc(struct zs_pool *zs_pool, + size_t size, gfp_t gfp); +extern struct task_struct *fetch_task_from_proc(struct inode *inode); +extern unsigned long hybridswap_fetch_zram_used_pages(void); +extern unsigned long long hybridswap_fetch_zram_pagefault(void); +extern bool hybridswap_reclaim_work_running(void); +extern void hybridswap_force_reclaim(struct mem_cgroup *mcg); +extern bool hybridswap_stored_wm_ok(void); +extern void mem_cgroup_id_remove_hook(void *data, struct mem_cgroup *memcg); +extern int mem_cgroup_stored_wm_scale_write( + struct cgroup_subsys_state *css, struct cftype *cft, s64 val); +extern s64 mem_cgroup_stored_wm_scale_read( + struct cgroup_subsys_state *css, struct cftype *cft); +extern bool hybridswap_delete(struct zram *zram, u32 index); +extern int hybridswap_stored_info(unsigned long *total, unsigned long *used); + +extern unsigned long long hybridswap_read_mcg_stats( + struct mem_cgroup *mcg, enum hybridswap_mcg_member mcg_member); +extern int hybridswap_core_enable(void); +extern void hybridswap_core_disable(void); +extern int hybridswap_psi_show(struct seq_file *m, void *v); +#else +static inline unsigned long long hybridswap_read_mcg_stats( + struct mem_cgroup *mcg, enum hybridswap_mcg_member mcg_member) +{ + return 0; +} + +unsigned long long hybridswap_fetch_zram_pagefault(void) +{ + return 0; +} + +static inline unsigned long long hybridswap_fetch_zram_pagefault(void) +{ + return 0; +} + +static inline bool hybridswap_reclaim_work_running(void) +{ + return false; +} + +static inline bool hybridswap_core_enabled(void) { return false; } +static inline bool hybridswap_out_to_eswap_enable(void) { return false; } +#endif + +#ifdef CONFIG_HYBRIDSWAP_SWAPD +extern atomic_long_t page_fault_pause; +extern atomic_long_t page_fault_pause_cnt; +extern struct cftype mem_cgroup_swapd_legacy_files[]; +extern bool zram_watermark_ok(void); +extern void wake_all_swapd(void); +extern void alloc_pages_slowpath_hook(void *data, gfp_t gfp_mask, + unsigned int order, unsigned long delta); +extern void rmqueue_hook(void *data, struct zone *preferred_zone, + struct zone *zone, unsigned int order, gfp_t gfp_flags, + unsigned int alloc_flags, int migratetype); +extern void __init swapd_pre_init(void); +extern void swapd_pre_deinit(void); +extern void update_swapd_mcg_setup(struct mem_cgroup *memcg); +extern bool free_zram_is_ok(void); +extern bool free_swap_is_low(void); +extern unsigned long fetch_nr_zram_total(void); +extern int swapd_init(struct zram *zram); +extern void swapd_exit(void); +extern bool hybridswap_swapd_enabled(void); +#else +static inline bool hybridswap_swapd_enabled(void) { return false; } +#endif + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +extern spinlock_t cached_idr_lock; +extern struct idr cached_idr; + +extern void __init akcompressd_pre_init(void); +extern void __exit akcompressd_pre_deinit(void); +extern int create_akcompressd_task(struct zram *zram); +extern void clear_page_memcg(struct cgroup_cache_page *cache); +#endif + +#endif /* end of HYBRIDSWAP_INTERNAL_H */ diff --git a/drivers/moto_swap/hybridswap/hybridswap_main.c b/drivers/moto_swap/hybridswap/hybridswap_main.c new file mode 100644 index 000000000000..471657b99ae5 --- /dev/null +++ b/drivers/moto_swap/hybridswap/hybridswap_main.c @@ -0,0 +1,1062 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (C) 2020-2022 Oplus. All rights reserved. + */ + +#define pr_fmt(fmt) "moto_swap: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include + +#ifdef CONFIG_ZRAM_5_4 +#include "../zram-5.4/zram_drv.h" +#include "../zram-5.4/zram_drv_internal.h" +#else +#include "../zram-5.10/zram_drv.h" +#include "../zram-5.10/zram_drv_internal.h" +#endif +#include "hybridswap_internal.h" +#include "hybridswap.h" + + + +static const char *swapd_text[NR_EVENT_ITEMS] = { +#ifdef CONFIG_HYBRIDSWAP_SWAPD + "swapd_wakeup", + "swapd_hit_pagefaults", + "swapd_memcg_scale_skip", + "swapd_memcg_pagefault_skip", + "swapd_shrink_anon", + "swapd_swapout", + "swapd_skip_swapout", + "swapd_nothing_ignore", + "swapd_over_min_buffer_skip_times", + "swapd_nothing_ignore_skip_times", + "swapd_snapshot_times", + "swapd_skip_shrink_of_window", + "swapd_manual_pause", +#ifdef CONFIG_OPLUS_JANK + "swapd_cpu_busy_skip_times", + "swapd_cpu_busy_break_times", +#endif +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + "akcompressd_running", +#endif +}; + +enum scan_balance { + SCAN_EQUAL, + SCAN_FRACT, + SCAN_ANON, + SCAN_FILE, +}; + +static int log_level = HYB_MAX; +static struct kmem_cache *hybridswap_cache; +static struct list_head grade_head; +static DEFINE_SPINLOCK(grade_list_lock); +static DEFINE_MUTEX(hybridswap_enable_lock); +static bool hybridswap_enabled = false; + +DEFINE_MUTEX(reclaim_para_lock); +DEFINE_PER_CPU(struct swapd_event_state, swapd_event_states); + +extern unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, + unsigned long nr_pages, + gfp_t gfp_mask, + bool may_swap); + +void hybridswap_loglevel_set(int level) +{ + log_level = level; +} + +int hybridswap_loglevel(void) +{ + return log_level; +} + +void __put_memcg_cache(memcg_hybs_t *hybs) +{ +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (hybs->cache.id > 0) { + spin_lock(&cached_idr_lock); + idr_replace(&cached_idr, NULL, hybs->cache.id); + idr_remove(&cached_idr, hybs->cache.id); + spin_unlock(&cached_idr_lock); + } + + spin_lock(&hybs->cache.lock); + if (hybs->cache.dead != 1) + BUG(); + spin_unlock(&hybs->cache.lock); +#endif + kmem_cache_free(hybridswap_cache, (void *)hybs); +} + +static inline void sum_hybridswap_vm_events(unsigned long *ret) +{ + int cpu; + int i; + + memset(ret, 0, NR_EVENT_ITEMS * sizeof(unsigned long)); + + for_each_online_cpu(cpu) { + struct swapd_event_state *this = + &per_cpu(swapd_event_states, cpu); + + for (i = 0; i < NR_EVENT_ITEMS; i++) + ret[i] += this->event[i]; + } +} + +static inline void all_hybridswap_vm_events(unsigned long *ret) +{ + get_online_cpus(); + sum_hybridswap_vm_events(ret); + put_online_cpus(); +} + +ssize_t hybridswap_vmstat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + unsigned long *vm_buf = NULL; + int len = 0; + int i = 0; + + vm_buf = kzalloc(sizeof(struct swapd_event_state), GFP_KERNEL); + if (!vm_buf) + return -ENOMEM; + all_hybridswap_vm_events(vm_buf); + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + len += snprintf(buf + len, PAGE_SIZE - len, "%-32s %12lu\n", + "page_fault_pause", atomic_long_read(&page_fault_pause)); + len += snprintf(buf + len, PAGE_SIZE - len, "%-32s %12lu\n", + "page_fault_pause_cnt", atomic_long_read(&page_fault_pause_cnt)); +#endif + + for (;i < NR_EVENT_ITEMS; i++) { + len += snprintf(buf + len, PAGE_SIZE - len, "%-32s %12lu\n", + swapd_text[i], vm_buf[i]); + if (len == PAGE_SIZE) + break; + } + kfree(vm_buf); + + return len; +} + +ssize_t hybridswap_loglevel_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + char *type_buf = NULL; + unsigned long val; + + type_buf = strstrip((char *)buf); + if (kstrtoul(type_buf, 0, &val)) + return -EINVAL; + + if (val >= HYB_MAX) { + hybp(HYB_ERR, "val %lu is not valid\n", val); + return -EINVAL; + } + hybridswap_loglevel_set((int)val); + + return len; +} + +ssize_t hybridswap_loglevel_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + ssize_t size = 0; + + size += scnprintf(buf + size, PAGE_SIZE - size, + "Hybridswap log level: %d\n", hybridswap_loglevel()); + + return size; +} + +/* Make sure the memcg is not NULL in caller */ +memcg_hybs_t *hybridswap_cache_alloc(struct mem_cgroup *memcg, bool atomic) +{ + memcg_hybs_t *hybs; + u64 ret; + gfp_t flags = GFP_KERNEL; + + if (memcg->android_oem_data1) + BUG(); + + if (atomic) + flags &= ~__GFP_DIRECT_RECLAIM; + + hybs = (memcg_hybs_t *)kmem_cache_zalloc(hybridswap_cache, flags); + if (!hybs) + return NULL; + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + spin_lock_init(&hybs->cache.lock); + INIT_LIST_HEAD(&hybs->cache.head); + hybs->cache.cnt = 0; + hybs->cache.compressing = 0; + hybs->cache.dead = 0; + spin_lock(&cached_idr_lock); + hybs->cache.id = idr_alloc(&cached_idr, NULL, 1, MEM_CGROUP_ID_MAX, + GFP_KERNEL); + if (hybs->cache.id < 0) { + spin_unlock(&cached_idr_lock); + kmem_cache_free(hybridswap_cache, (void*)hybs); + return NULL; + } + idr_replace(&cached_idr, &hybs->cache, hybs->cache.id); + spin_unlock(&cached_idr_lock); +#endif + INIT_LIST_HEAD(&hybs->grade_node); +#ifdef CONFIG_HYBRIDSWAP_CORE + spin_lock_init(&hybs->zram_init_lock); +#endif + atomic64_set(&hybs->app_grade, 300); + atomic64_set(&hybs->ufs2zram_scale, 100); +#ifdef CONFIG_HYBRIDSWAP_SWAPD + atomic_set(&hybs->mem2zram_scale, 80); + atomic_set(&hybs->zram2ufs_scale, 50); + atomic_set(&hybs->pagefault_level, 50); +#endif + hybs->memcg = memcg; + refcount_set(&hybs->usage, 1); + + ret = atomic64_cmpxchg((atomic64_t *)&memcg->android_oem_data1, 0, (u64)hybs); + if (ret != 0) { +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + hybs->cache.dead = 1; +#endif + put_memcg_cache(hybs); + return (memcg_hybs_t *)ret; + } + + return hybs; +} + +#ifdef CONFIG_HYBRIDSWAP_SWAPD +static void tune_scan_type_hook(void *data, char *scan_balance) +{ + /*hybrid swapd,scan anon only*/ + if (current_is_mswapd()) { + *scan_balance = SCAN_ANON; + return; + } + +#ifdef CONFIG_HYBRIDSWAP_CORE + if (unlikely(!hybridswap_core_enabled())) + return; + + /*real zram full, scan file only*/ + if (!free_zram_is_ok()) { + *scan_balance = SCAN_FILE; + return; + } +#endif +} +#endif + +static void tune_swappiness_hook(void *data, int *swappiness) +{ + if (current_is_kswapd()) { + *swappiness = *swappiness; +#ifdef CONFIG_HYBRIDSWAP_SWAPD + } else if (current_is_mswapd()) { + *swappiness = 200; + if (free_swap_is_low()) { + *swappiness = 0; + } +#endif + } else { + *swappiness = 60; + } + + return; +} + +static void mem_cgroup_alloc_hook(void *data, struct mem_cgroup *memcg) +{ + if (memcg->android_oem_data1) + BUG(); + + hybridswap_cache_alloc(memcg, true); +} + +static void mem_cgroup_free_hook(void *data, struct mem_cgroup *memcg) +{ + memcg_hybs_t *hybs; + + if (!memcg->android_oem_data1) + return; + + hybs = (memcg_hybs_t *)memcg->android_oem_data1; + memcg->android_oem_data1 = 0; +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + clear_page_memcg(&hybs->cache); +#endif + put_memcg_cache(hybs); +} + +void memcg_app_grade_update(struct mem_cgroup *tarfetch) +{ + struct list_head *pos = NULL; + unsigned long flags; + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + update_swapd_mcg_setup(tarfetch); +#endif + spin_lock_irqsave(&grade_list_lock, flags); + list_for_each(pos, &grade_head) { + memcg_hybs_t *hybs = list_entry(pos, memcg_hybs_t, grade_node); + if (atomic64_read(&hybs->app_grade) < + atomic64_read(&MEMCGRP_ITEM(tarfetch, app_grade))) + break; + } + list_move_tail(&MEMCGRP_ITEM(tarfetch, grade_node), pos); + spin_unlock_irqrestore(&grade_list_lock, flags); +} + +static void mem_cgroup_css_online_hook(void *data, + struct cgroup_subsys_state *css, struct mem_cgroup *memcg) +{ + if (memcg->android_oem_data1) + memcg_app_grade_update(memcg); + + css_get(css); +} + +static void mem_cgroup_css_offline_hook(void *data, + struct cgroup_subsys_state *css, struct mem_cgroup *memcg) +{ + unsigned long flags; + + if (memcg->android_oem_data1) { + spin_lock_irqsave(&grade_list_lock, flags); + list_del_init(&MEMCGRP_ITEM(memcg, grade_node)); + spin_unlock_irqrestore(&grade_list_lock, flags); + } + + css_put(css); +} + +#define REGISTER_HOOK(name) do {\ + rc = register_trace_android_vh_##name(name##_hook, NULL);\ + if (rc) {\ + hybp(HYB_ERR, "%s:%d register hook %s failed", __FILE__, __LINE__, #name);\ + goto err_out_##name;\ + }\ +} while (0) + +#define UNREGISTER_HOOK(name) do {\ + unregister_trace_android_vh_##name(name##_hook, NULL);\ +} while (0) + +#define ERROR_OUT(name) err_out_##name + +static int register_all_hooks(void) +{ + int rc; + + /* mem_cgroup_alloc_hook */ + REGISTER_HOOK(mem_cgroup_alloc); + /* mem_cgroup_free_hook */ + REGISTER_HOOK(mem_cgroup_free); + /* mem_cgroup_css_online_hook */ + REGISTER_HOOK(mem_cgroup_css_online); + /* mem_cgroup_css_offline_hook */ + REGISTER_HOOK(mem_cgroup_css_offline); +#ifdef CONFIG_HYBRIDSWAP_SWAPD + /* rmqueue_hook */ + REGISTER_HOOK(rmqueue); + /* tune_scan_type_hook */ + REGISTER_HOOK(tune_scan_type); +#endif + /* tune_swappiness_hook */ + REGISTER_HOOK(tune_swappiness); +#ifdef CONFIG_HYBRIDSWAP_CORE + /* mem_cgroup_id_remove_hook */ + REGISTER_HOOK(mem_cgroup_id_remove); +#endif + return 0; + +#ifdef CONFIG_HYBRIDSWAP_CORE + UNREGISTER_HOOK(mem_cgroup_id_remove); +ERROR_OUT(mem_cgroup_id_remove): +#endif + UNREGISTER_HOOK(tune_swappiness); +ERROR_OUT(tune_swappiness): +#ifdef CONFIG_HYBRIDSWAP_SWAPD + UNREGISTER_HOOK(tune_scan_type); +ERROR_OUT(tune_scan_type): + UNREGISTER_HOOK(rmqueue); +ERROR_OUT(rmqueue): +#endif + UNREGISTER_HOOK(mem_cgroup_css_offline); +ERROR_OUT(mem_cgroup_css_offline): + UNREGISTER_HOOK(mem_cgroup_css_online); +ERROR_OUT(mem_cgroup_css_online): + UNREGISTER_HOOK(mem_cgroup_free); +ERROR_OUT(mem_cgroup_free): + UNREGISTER_HOOK(mem_cgroup_alloc); +ERROR_OUT(mem_cgroup_alloc): + return rc; +} + +static void unregister_all_hook(void) +{ + UNREGISTER_HOOK(mem_cgroup_alloc); + UNREGISTER_HOOK(mem_cgroup_free); + UNREGISTER_HOOK(mem_cgroup_css_offline); + UNREGISTER_HOOK(mem_cgroup_css_online); +#ifdef CONFIG_HYBRIDSWAP_CORE + UNREGISTER_HOOK(mem_cgroup_id_remove); +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD + UNREGISTER_HOOK(rmqueue); + UNREGISTER_HOOK(tune_scan_type); +#endif + UNREGISTER_HOOK(tune_swappiness); +} + +unsigned long memcg_anon_pages(struct mem_cgroup *memcg) +{ +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 10, 0) + struct lruvec *lruvec = NULL; + struct mem_cgroup_per_node *mz = NULL; +#endif + if (!memcg) + return 0; + +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 10, 0) + mz = mem_cgroup_nodeinfo(memcg, 0); + if (!mz) { + fetch_next_memcg_break(memcg); + return 0; + } + + lruvec = &mz->lruvec; + if (!lruvec) { + fetch_next_memcg_break(memcg); + return 0; + } + +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 4, 0) + return (mem_cgroup_fetch_lru_size(lruvec, LRU_ACTIVE_ANON) + + mem_cgroup_fetch_lru_size(lruvec, LRU_INACTIVE_ANON)); +#else + return (lruvec_page_state(lruvec, NR_ACTIVE_ANON) + + lruvec_page_state(lruvec, NR_INACTIVE_ANON)); +#endif +#else + return (memcg_page_state_local(memcg, NR_ACTIVE_ANON) + + memcg_page_state_local(memcg, NR_INACTIVE_ANON)); +#endif +} + +static unsigned long memcg_inactive_anon_pages(struct mem_cgroup *memcg) +{ +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 10, 0) + struct lruvec *lruvec = NULL; + struct mem_cgroup_per_node *mz = NULL; +#endif + + if (!memcg) + return 0; + +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 10, 0) + mz = mem_cgroup_nodeinfo(memcg, 0); + if (!mz) { + fetch_next_memcg_break(memcg); + return 0; + } + + lruvec = &mz->lruvec; + if (!lruvec) { + fetch_next_memcg_break(memcg); + return 0; + } + +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 4, 0) + return mem_cgroup_fetch_lru_size(lruvec, LRU_INACTIVE_ANON); +#else + return lruvec_page_state(lruvec, NR_INACTIVE_ANON); +#endif +#else + return memcg_page_state_local(memcg, NR_INACTIVE_ANON); +#endif +} + +static ssize_t mem_cgroup_force_shrink_anon(struct kernfs_open_file *of, + char *buf, size_t nbytes, loff_t off) +{ + struct mem_cgroup *memcg; + unsigned long nr_need_reclaim, reclaim_total, nr_reclaimed; + int ret; + + buf = strstrip(buf); + ret = kstrtoul(buf, 0, &reclaim_total); + if (unlikely(ret)) { + hybp(HYB_ERR, "reclaim_total %s value is error!\n", buf); + return -EINVAL; + } + + memcg = mem_cgroup_from_css(of_css(of)); + + if (reclaim_total) + nr_need_reclaim = memcg_anon_pages(memcg); + else + nr_need_reclaim = memcg_inactive_anon_pages(memcg); + + hybp(HYB_INFO, "FORCE SHRINK +\n"); + nr_reclaimed = try_to_free_mem_cgroup_pages(memcg, nr_need_reclaim, + GFP_KERNEL, true); + hybp(HYB_INFO, "FORCE SHRINK - to_reclaim %lu reclaimed %lu\n", nr_need_reclaim, nr_reclaimed); + return nbytes; +} + +static int memcg_total_info_per_app_show(struct seq_file *m, void *v) +{ + struct mem_cgroup *memcg = NULL; + unsigned long anon_size; + unsigned long zram_compress_size; + unsigned long eswap_compress_size; + unsigned long zram_page_size; + unsigned long eswap_page_size; + + seq_printf(m, "%-8s %-8s %-8s %-8s %-8s %s \n", + "anon", "zram_c", "zram_p", "eswap_c", "eswap_p", + "memcg_n"); + while ((memcg = fetch_next_memcg(memcg))) { + if (!MEMCGRP_ITEM_DATA(memcg)) + continue; + + anon_size = memcg_anon_pages(memcg); + zram_compress_size = hybridswap_read_mcg_stats(memcg, + MCG_ZRAM_STORED_SZ); + eswap_compress_size = hybridswap_read_mcg_stats(memcg, + MCG_DISK_STORED_SZ); + zram_page_size = hybridswap_read_mcg_stats(memcg, + MCG_ZRAM_STORED_PG_SZ); + eswap_page_size = hybridswap_read_mcg_stats(memcg, + MCG_DISK_STORED_PG_SZ); + + anon_size *= PAGE_SIZE / SZ_1K; + zram_compress_size /= SZ_1K; + eswap_compress_size /= SZ_1K; + zram_page_size *= PAGE_SIZE / SZ_1K; + eswap_page_size *= PAGE_SIZE / SZ_1K; + + seq_printf(m, "%-8lu %-8lu %-8lu %-8lu %-8lu %s \n", + anon_size, zram_compress_size, zram_page_size, + eswap_compress_size, eswap_page_size, + MEMCGRP_ITEM(memcg, name)); + } + + return 0; +} + +static int memcg_swap_stat_show(struct seq_file *m, void *v) +{ + struct mem_cgroup *memcg = NULL; + unsigned long eswap_out_cnt; + unsigned long eswap_out_size; + unsigned long eswap_in_size; + unsigned long eswap_in_cnt; + unsigned long page_fault_cnt; + unsigned long cur_eswap_size; + unsigned long max_eswap_size; + unsigned long zram_compress_size, zram_page_size; + unsigned long eswap_compress_size, eswap_page_size; + + memcg = mem_cgroup_from_css(seq_css(m)); + + zram_compress_size = hybridswap_read_mcg_stats(memcg, MCG_ZRAM_STORED_SZ); + zram_page_size = hybridswap_read_mcg_stats(memcg, MCG_ZRAM_STORED_PG_SZ); + eswap_compress_size = hybridswap_read_mcg_stats(memcg, MCG_DISK_STORED_SZ); + eswap_page_size = hybridswap_read_mcg_stats(memcg, MCG_DISK_STORED_PG_SZ); + + eswap_out_cnt = hybridswap_read_mcg_stats(memcg, MCG_ESWAPOUT_CNT); + eswap_out_size = hybridswap_read_mcg_stats(memcg, MCG_ESWAPOUT_SZ); + eswap_in_size = hybridswap_read_mcg_stats(memcg, MCG_ESWAPIN_SZ); + eswap_in_cnt = hybridswap_read_mcg_stats(memcg, MCG_ESWAPIN_CNT); + page_fault_cnt = hybridswap_read_mcg_stats(memcg, MCG_DISK_FAULT_CNT); + cur_eswap_size = hybridswap_read_mcg_stats(memcg, MCG_DISK_SPACE); + max_eswap_size = hybridswap_read_mcg_stats(memcg, MCG_DISK_SPACE_PEAK); + + seq_printf(m, "%-32s %12lu KB\n", "zramCompressedSize:", + zram_compress_size / SZ_1K); + seq_printf(m, "%-32s %12lu KB\n", "zramOrignalSize:", + zram_page_size << (PAGE_SHIFT - 10)); + seq_printf(m, "%-32s %12lu KB\n", "eswapCompressedSize:", + eswap_compress_size / SZ_1K); + seq_printf(m, "%-32s %12lu KB\n", "eswapOrignalSize:", + eswap_page_size << (PAGE_SHIFT - 10)); + seq_printf(m, "%-32s %12lu \n", "eswapOutTotal:", eswap_out_cnt); + seq_printf(m, "%-32s %12lu KB\n", "eswapOutSize:", eswap_out_size / SZ_1K); + seq_printf(m, "%-32s %12lu\n", "eswapInTotal:", eswap_in_cnt); + seq_printf(m, "%-32s %12lu KB\n", "eswapInSize:", eswap_in_size / SZ_1K); + seq_printf(m, "%-32s %12lu\n", "pageInTotal:", page_fault_cnt); + seq_printf(m, "%-32s %12lu KB\n", "eswapSizeCur:", cur_eswap_size / SZ_1K); + seq_printf(m, "%-32s %12lu KB\n", "eswapSizeMax:", max_eswap_size / SZ_1K); + + return 0; +} + +static ssize_t mem_cgroup_name_write(struct kernfs_open_file *of, char *buf, + size_t nbytes, loff_t off) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(of_css(of)); + memcg_hybs_t *hybp = MEMCGRP_ITEM_DATA(memcg); + int len, w_len; + + if (!hybp) + return -EINVAL; + + buf = strstrip(buf); + len = strlen(buf) + 1; + if (len > MEM_CGROUP_NAME_MAX_LEN) + len = MEM_CGROUP_NAME_MAX_LEN; + + w_len = snprintf(hybp->name, len, "%s", buf); + if (w_len > len) + hybp->name[len - 1] = '\0'; + + return nbytes; +} + +static int mem_cgroup_name_show(struct seq_file *m, void *v) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(m)); + + if (!MEMCGRP_ITEM_DATA(memcg)) + return -EPERM; + + seq_printf(m, "%s\n", MEMCGRP_ITEM(memcg, name)); + + return 0; +} + +static int mem_cgroup_app_grade_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + struct mem_cgroup *memcg; + memcg_hybs_t *hybs; + + if (val > MAX_APP_GRADE || val < 0) + return -EINVAL; + + memcg = mem_cgroup_from_css(css); + hybs = MEMCGRP_ITEM_DATA(memcg); + if (!hybs) { + hybs = hybridswap_cache_alloc(memcg, false); + if (!hybs) + return -EINVAL; + } + + if (atomic64_read(&MEMCGRP_ITEM(memcg, app_grade)) != val) + atomic64_set(&MEMCGRP_ITEM(memcg, app_grade), val); + memcg_app_grade_update(memcg); + + return 0; +} + +static s64 mem_cgroup_app_grade_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(css); + + if (!MEMCGRP_ITEM_DATA(memcg)) + return -EPERM; + + return atomic64_read(&MEMCGRP_ITEM(memcg, app_grade)); +} + +int mem_cgroup_app_uid_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + struct mem_cgroup *memcg; + memcg_hybs_t *hybs; + + if (val < 0) + return -EINVAL; + + memcg = mem_cgroup_from_css(css); + hybs = MEMCGRP_ITEM_DATA(memcg); + if (!hybs) { + return -EINVAL; + } + + if (atomic64_read(&MEMCGRP_ITEM(memcg, app_uid)) != val) + atomic64_set(&MEMCGRP_ITEM(memcg, app_uid), val); + + return 0; +} + +static s64 mem_cgroup_app_uid_read(struct cgroup_subsys_state *css, struct cftype *cft) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(css); + + if (!MEMCGRP_ITEM_DATA(memcg)) + return -EPERM; + + return atomic64_read(&MEMCGRP_ITEM(memcg, app_uid)); +} + +static int mem_cgroup_ufs2zram_scale_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(css); + + if (!MEMCGRP_ITEM_DATA(memcg)) + return -EPERM; + + if (val > MAX_RATIO || val < MIN_RATIO) + return -EINVAL; + + atomic64_set(&MEMCGRP_ITEM(memcg, ufs2zram_scale), val); + + return 0; +} + +static s64 mem_cgroup_ufs2zram_scale_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(css); + + if (!MEMCGRP_ITEM_DATA(memcg)) + return -EPERM; + + return atomic64_read(&MEMCGRP_ITEM(memcg, ufs2zram_scale)); +} + +static int mem_cgroup_force_eswapin_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(css); + memcg_hybs_t *hybs; + unsigned long size = 0; + const unsigned int scale = 100; + + hybs = MEMCGRP_ITEM_DATA(memcg); + if (!hybs) + return -EPERM; + +#ifdef CONFIG_HYBRIDSWAP_CORE + size = atomic64_read(&hybs->hybridswap_stored_size); +#endif + size = atomic64_read(&hybs->ufs2zram_scale) * size / scale; + size = ESWAP_ALIGN_UP(size); + +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_batches(memcg, size, val ? true : false); +#endif + + return 0; +} + +static int mem_cgroup_force_eswapout_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_force_reclaim(mem_cgroup_from_css(css)); +#endif + return 0; +} + +struct mem_cgroup *fetch_next_memcg(struct mem_cgroup *prev) +{ + memcg_hybs_t *hybs = NULL; + struct mem_cgroup *memcg = NULL; + struct list_head *pos = NULL; + unsigned long flags; + bool prev_got = true; + + spin_lock_irqsave(&grade_list_lock, flags); +find_again: + if (unlikely(!prev)) + pos = &grade_head; + else + pos = &MEMCGRP_ITEM(prev, grade_node); + + if (list_empty(pos)) /* deleted node */ + goto unlock; + + if (pos->next == &grade_head) + goto unlock; + + hybs = list_entry(pos->next, struct mem_cgroup_hybridswap, grade_node); + memcg = hybs->memcg; + if (unlikely(!memcg)) + goto unlock; + + if (!css_tryget(&memcg->css)) { + if (prev && prev_got) + css_put(&prev->css); + prev = memcg; + prev_got = false; + goto find_again; + } + +unlock: + spin_unlock_irqrestore(&grade_list_lock, flags); + if (prev && prev_got) + css_put(&prev->css); + + return memcg; +} + +void fetch_next_memcg_break(struct mem_cgroup *memcg) +{ + if (memcg) + css_put(&memcg->css); +} + +static struct cftype mem_cgroup_hybridswap_legacy_files[] = { + { + .name = "force_shrink_anon", + .write = mem_cgroup_force_shrink_anon, + }, + { + .name = "total_info_per_app", + .flags = CFTYPE_ONLY_ON_ROOT, + .seq_show = memcg_total_info_per_app_show, + }, + { + .name = "swap_stat", + .seq_show = memcg_swap_stat_show, + }, + { + .name = "name", + .write = mem_cgroup_name_write, + .seq_show = mem_cgroup_name_show, + }, + { + .name = "app_score", + .write_s64 = mem_cgroup_app_grade_write, + .read_s64 = mem_cgroup_app_grade_read, + }, + { + .name = "app_uid", + .write_s64 = mem_cgroup_app_uid_write, + .read_s64 = mem_cgroup_app_uid_read, + }, + { + .name = "ub_ufs2zram_ratio", + .write_s64 = mem_cgroup_ufs2zram_scale_write, + .read_s64 = mem_cgroup_ufs2zram_scale_read, + }, + { + .name = "force_eswapin", + .write_s64 = mem_cgroup_force_eswapin_write, + }, + { + .name = "force_eswapout", + .write_s64 = mem_cgroup_force_eswapout_write, + }, +#ifdef CONFIG_HYBRIDSWAP_CORE + { + .name = "psi", + .flags = CFTYPE_ONLY_ON_ROOT, + .seq_show = hybridswap_psi_show, + }, + { + .name = "stored_wm_ratio", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = mem_cgroup_stored_wm_scale_write, + .read_s64 = mem_cgroup_stored_wm_scale_read, + }, +#endif + { }, /* terminate */ +}; + +static int hybridswap_enable(struct zram *zram) +{ + int ret = 0; + + if (hybridswap_enabled) { + hybp(HYB_WARN, "hybridswap_enabled is true\n"); + return ret; + } + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + ret = swapd_init(zram); + if (ret) + return ret; +#endif + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + ret = create_akcompressd_task(zram); + if (ret) + goto create_akcompressd_task_fail; +#endif + +#ifdef CONFIG_HYBRIDSWAP_CORE + ret = hybridswap_core_enable(); + if (ret) + goto hybridswap_core_enable_fail; +#endif + hybridswap_enabled = true; + + return 0; + +#ifdef CONFIG_HYBRIDSWAP_CORE +hybridswap_core_enable_fail: +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + destroy_akcompressd_task(zram); +create_akcompressd_task_fail: +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD + swapd_exit(); +#endif + return ret; +} + +static void hybridswap_disable(struct zram * zram) +{ + if (!hybridswap_enabled) { + hybp(HYB_WARN, "hybridswap_enabled is false\n"); + return; + } + +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_core_disable(); +#endif + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + destroy_akcompressd_task(zram); +#endif + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + swapd_exit(); +#endif + hybridswap_enabled = false; +} + +ssize_t hybridswap_enable_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int len = snprintf(buf, PAGE_SIZE, "hybridswap %s out_to_eswap %s swapd %s\n", + hybridswap_core_enabled() ? "enable" : "disable", + hybridswap_out_to_eswap_enable() ? "enable" : "disable", + hybridswap_swapd_enabled() ? "enable" : "disable"); + + return len; +} + +ssize_t hybridswap_enable_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned long val; + char *kbuf; + struct zram *zram; + + kbuf = strstrip((char *)buf); + ret = kstrtoul(kbuf, 0, &val); + if (unlikely(ret)) { + hybp(HYB_ERR, "val %s is invalid!\n", kbuf); + + return -EINVAL; + } + + mutex_lock(&hybridswap_enable_lock); + zram = dev_to_zram(dev); + if (val == 0) + hybridswap_disable(zram); + else + ret = hybridswap_enable(zram); + mutex_unlock(&hybridswap_enable_lock); + + if (ret == 0) + ret = len; + return ret; +} + +int __init hybridswap_pre_init(void) +{ + int ret; + + INIT_LIST_HEAD(&grade_head); + log_level = HYB_INFO; + + hybridswap_cache = kmem_cache_create("mem_cgroup_hybridswap", + sizeof(struct mem_cgroup_hybridswap), + 0, SLAB_PANIC, NULL); + if (!hybridswap_cache) { + hybp(HYB_ERR, "create hybridswap_cache failed\n"); + ret = -ENOMEM; + return ret; + } + + ret = cgroup_add_legacy_cftypes(&memory_cgrp_subsys, + mem_cgroup_hybridswap_legacy_files); + if (ret) { + hybp(HYB_INFO, "add mem_cgroup_hybridswap_legacy_files failed\n"); + goto error_out; + } + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + ret = cgroup_add_legacy_cftypes(&memory_cgrp_subsys, + mem_cgroup_swapd_legacy_files); + if (ret) { + hybp(HYB_INFO, "add mem_cgroup_swapd_legacy_files failed!\n"); + goto error_out; + } +#endif + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + akcompressd_pre_init(); +#endif + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + swapd_pre_init(); +#endif + ret = register_all_hooks(); + if (ret) + goto fail_out; + + hybp(HYB_INFO, "hybridswap inited success!\n"); + return 0; + +fail_out: +#ifdef CONFIG_HYBRIDSWAP_SWAPD + swapd_pre_deinit(); +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + akcompressd_pre_deinit(); +#endif +error_out: + if (hybridswap_cache) { + kmem_cache_destroy(hybridswap_cache); + hybridswap_cache = NULL; + } + return ret; +} + +void __exit hybridswap_exit(void) +{ + unregister_all_hook(); + +#ifdef CONFIG_HYBRIDSWAP_SWAPD + swapd_pre_deinit(); +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + akcompressd_pre_deinit(); +#endif + + if (hybridswap_cache) { + kmem_cache_destroy(hybridswap_cache); + hybridswap_cache = NULL; + } +} diff --git a/drivers/moto_swap/hybridswap/hybridswap_swapd.c b/drivers/moto_swap/hybridswap/hybridswap_swapd.c new file mode 100644 index 000000000000..efdee6a5367e --- /dev/null +++ b/drivers/moto_swap/hybridswap/hybridswap_swapd.c @@ -0,0 +1,1963 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (C) 2020-2022 Oplus. All rights reserved. + */ + +#define pr_fmt(fmt) "moto_swap: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) +#include +#endif +#include + +#ifdef CONFIG_ZRAM_5_4 +#include "../zram-5.4/zram_drv.h" +#include "../zram-5.4/zram_drv_internal.h" +#else +#include "../zram-5.10/zram_drv.h" +#include "../zram-5.10/zram_drv_internal.h" +#endif +#include "hybridswap_internal.h" + +#define MOTO_SWAP_VERSION 1 + +struct swapd_param { + unsigned int min_grade; + unsigned int max_grade; + unsigned int mem2zram_scale; + unsigned int zram2ufs_scale; + unsigned int pagefault_level; +}; + +struct hybridswapd_task { + wait_queue_head_t swapd_wait; + atomic_t swapd_wait_flag; + struct task_struct *swapd; + struct cpumask swapd_bind_cpumask; +}; +#define PGDAT_ITEM_DATA(pgdat) ((struct hybridswapd_task*)(pgdat)->android_oem_data1) +#define PGDAT_ITEM(pgdat, item) (PGDAT_ITEM_DATA(pgdat)->item) + +#define INFOS_PAGEFAULT_THRESHOLD 3600 +#define PAGEFAULT_SNAPSHOT_MIN_GAP 150 +#define NOTHING_IGNORE_VALUE 20 +#define MAX_SKIP_GAP 1000 +#define EMPTY_ROUND_CHECK_THRESHOLD 10 +#define ZRAM_WM_RATIO 75 +#define COMPRESS_RATIO 33 +#define SWAPD_MAX_LEVEL_NUM 4 +#define SWAPD_DEFAULT_BIND_CPUS "0-3" +#define MAX_RECLAIMIN_SZ (100llu << 20) +#define page_to_kb(nr) (nr << (PAGE_SHIFT - 10)) +#define SWAPD_SHRINK_WINDOW (HZ * 10) +#define SWAPD_SHRINK_SIZE_PER_WINDOW 1024 +#define PAGES_TO_MB(pages) ((pages) >> 8) +#define PAGES_PER_1MB (1 << 8) + +unsigned long long total_pagefault_percent; +unsigned long long swapd_skip_interval; +bool last_round_is_empty; +unsigned long last_swapd_time; +atomic64_t zram_wm_scale = ATOMIC_LONG_INIT(ZRAM_WM_RATIO); +atomic64_t compress_scale = ATOMIC_LONG_INIT(COMPRESS_RATIO); +atomic_t usable_mem = ATOMIC_INIT(0); +atomic_t min_mem_watermark = ATOMIC_INIT(0); +atomic_t high_mem_watermark = ATOMIC_INIT(0); +atomic_t max_reclaim_size = ATOMIC_INIT(100); +atomic64_t free_swap_level = ATOMIC64_INIT(0); +atomic64_t zram_crit_thres = ATOMIC_LONG_INIT(0); +atomic64_t cpuload_level = ATOMIC_LONG_INIT(0); +atomic64_t infos_pagefault_level = ATOMIC_LONG_INIT(INFOS_PAGEFAULT_THRESHOLD); +atomic64_t pagefault_refresh_min = ATOMIC_LONG_INIT(PAGEFAULT_SNAPSHOT_MIN_GAP); +atomic64_t nothing_ignore_skip_interval = ATOMIC_LONG_INIT(NOTHING_IGNORE_VALUE); +atomic64_t max_skip_interval = ATOMIC_LONG_INIT(MAX_SKIP_GAP); +atomic64_t nothing_ignore_check_level = ATOMIC_LONG_INIT(EMPTY_ROUND_CHECK_THRESHOLD); +static unsigned long reclaim_exceed_sleep_ms = 50; +static unsigned long all_totalreserve_pages; + +static wait_queue_head_t refresh_daemonwait; +static atomic_t refresh_daemonwait_flag; +static atomic_t refresh_daemoninit_flag = ATOMIC_LONG_INIT(0); +static struct task_struct *refresh_daemontask; +static DEFINE_MUTEX(lowmem_event_lock); +static pid_t swapid = -1; +static unsigned long long infos_last_anon_pagefault; +static unsigned long last_refresh_t; +static struct swapd_param zswap_param[SWAPD_MAX_LEVEL_NUM]; +static enum cpuhp_state swapd_online; +static struct zram *swapd_zram = NULL; +static u64 max_reclaimin_size = MAX_RECLAIMIN_SZ; +atomic_long_t page_fault_pause = ATOMIC_LONG_INIT(0); +atomic_long_t page_fault_pause_cnt = ATOMIC_LONG_INIT(0); +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) +static struct notifier_block fb_notif; +static atomic_t display_off = ATOMIC_LONG_INIT(0); +#endif +static unsigned long swapd_shrink_window = SWAPD_SHRINK_WINDOW; +static unsigned long swapd_shrink_limit_per_window = SWAPD_SHRINK_SIZE_PER_WINDOW; +static unsigned long swapd_last_window_start; +static unsigned long swapd_last_window_shrink; +static atomic_t swapd_pause = ATOMIC_INIT(0); +static atomic_t swapd_enabled = ATOMIC_INIT(0); +static unsigned long swapd_nap_jiffies = 1; + +extern unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, + unsigned long nr_pages, + gfp_t gfp_mask, + bool may_swap); +#ifdef CONFIG_OPLUS_JANK +extern u32 fetch_cpu_load(u32 win_cnt, struct cpumask *mask); +#endif + +inline u64 fetch_zram_wm_scale_value(void) +{ + return atomic64_read(&zram_wm_scale); +} + +inline u64 fetch_compress_scale_value(void) +{ + return atomic64_read(&compress_scale); +} + +inline unsigned int fetch_usable_mem_value(void) +{ + return atomic_read(&usable_mem); +} + +inline unsigned int fetch_min_mem_watermark_value(void) +{ + return atomic_read(&min_mem_watermark); +} + +inline unsigned int fetch_high_mem_watermark_value(void) +{ + return atomic_read(&high_mem_watermark); +} + +inline u64 fetch_swapd_max_reclaim_size(void) +{ + return atomic_read(&max_reclaim_size); +} + +inline u64 fetch_free_swap_level_value(void) +{ + return atomic64_read(&free_swap_level); +} + +inline unsigned long long fetch_infos_pagefault_level_value(void) +{ + return atomic64_read(&infos_pagefault_level); +} + +inline unsigned long fetch_pagefault_refresh_min_value(void) +{ + return atomic64_read(&pagefault_refresh_min); +} + +inline unsigned long long fetch_nothing_ignore_skip_interval_value(void) +{ + return atomic64_read(¬hing_ignore_skip_interval); +} + +inline unsigned long long fetch_max_skip_interval_value(void) +{ + return atomic64_read(&max_skip_interval); +} + +inline unsigned long long fetch_nothing_ignore_check_level_value(void) +{ + return atomic64_read(¬hing_ignore_check_level); +} + +inline u64 fetch_zram_critical_level_value(void) +{ + return atomic64_read(&zram_crit_thres); +} + +inline u64 fetch_cpuload_level_value(void) +{ + return atomic64_read(&cpuload_level); +} + +static ssize_t usable_mem_params_write(struct kernfs_open_file *of, + char *buf, size_t nbytes, loff_t off) +{ + unsigned int usable_mem_value; + unsigned int min_mem_watermark_value; + unsigned int high_mem_watermark_value; + u64 free_swap_level_value; + + buf = strstrip(buf); + + if (sscanf(buf, "%u %u %u %llu", + &usable_mem_value, + &min_mem_watermark_value, + &high_mem_watermark_value, + &free_swap_level_value) != 4) + return -EINVAL; + + atomic_set(&usable_mem, usable_mem_value); + atomic_set(&min_mem_watermark, min_mem_watermark_value); + atomic_set(&high_mem_watermark, high_mem_watermark_value); + atomic64_set(&free_swap_level, + (free_swap_level_value * (SZ_1M / PAGE_SIZE))); + + if (atomic_read(&min_mem_watermark) == 0) + atomic_set(&refresh_daemoninit_flag, 0); + else + atomic_set(&refresh_daemoninit_flag, 1); + + wake_all_swapd(); + + return nbytes; +} + +static int usable_mem_params_show(struct seq_file *m, void *v) +{ + seq_printf(m, "avail_buffers: %u\n", + atomic_read(&usable_mem)); + seq_printf(m, "min_avail_buffers: %u\n", + atomic_read(&min_mem_watermark)); + seq_printf(m, "high_avail_buffers: %u\n", + atomic_read(&high_mem_watermark)); + seq_printf(m, "free_swap_threshold: %llu\n", + (atomic64_read(&free_swap_level) * PAGE_SIZE / SZ_1M)); + + return 0; +} + +static ssize_t swapd_max_reclaim_size_write(struct kernfs_open_file *of, + char *buf, size_t nbytes, loff_t off) +{ + const unsigned int base = 10; + u32 max_reclaim_size_value; + int ret; + + buf = strstrip(buf); + ret = kstrtouint(buf, base, &max_reclaim_size_value); + if (ret) + return -EINVAL; + + atomic_set(&max_reclaim_size, max_reclaim_size_value); + + return nbytes; +} + +static int swapd_max_reclaim_size_show(struct seq_file *m, void *v) +{ + seq_printf(m, "swapd_max_reclaim_size: %u\n", + atomic_read(&max_reclaim_size)); + + return 0; +} + +static int infos_pagefault_level_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(&infos_pagefault_level, val); + + return 0; +} + +static s64 infos_pagefault_level_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(&infos_pagefault_level); +} + +static int nothing_ignore_skip_interval_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(¬hing_ignore_skip_interval, val); + + return 0; +} + +static s64 nothing_ignore_skip_interval_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(¬hing_ignore_skip_interval); +} + +static int max_skip_interval_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(&max_skip_interval, val); + + return 0; +} + +static s64 max_skip_interval_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(&max_skip_interval); +} + +static int nothing_ignore_check_level_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(¬hing_ignore_check_level, val); + + return 0; +} + +static s64 nothing_ignore_check_level_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(¬hing_ignore_check_level); +} + +static int pagefault_refresh_min_write( + struct cgroup_subsys_state *css, struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(&pagefault_refresh_min, val); + + return 0; +} + +static s64 pagefault_refresh_min_read( + struct cgroup_subsys_state *css, struct cftype *cft) +{ + return atomic64_read(&pagefault_refresh_min); +} + +static int zram_critical_thres_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(&zram_crit_thres, val << (20 - PAGE_SHIFT)); + + return 0; +} + +static s64 zram_critical_thres_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(&zram_crit_thres) >> (20 - PAGE_SHIFT); +} + +static s64 cpuload_level_read(struct cgroup_subsys_state *css, + struct cftype *cft) + +{ + return atomic64_read(&cpuload_level); +} + +static int cpuload_level_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + atomic64_set(&cpuload_level, val); + + return 0; +} + +static s64 swapid_read(struct cgroup_subsys_state *css, struct cftype *cft) +{ + return swapid; +} + +static void swapd_mcgs_setup_parse(int level_num) +{ + struct mem_cgroup *memcg = NULL; + memcg_hybs_t *hybs = NULL; + int i; + + while ((memcg = fetch_next_memcg(memcg))) { + hybs = MEMCGRP_ITEM_DATA(memcg); + + for (i = 0; i < level_num; ++i) { + if (atomic64_read(&hybs->app_grade) >= zswap_param[i].min_grade && + atomic64_read(&hybs->app_grade) <= zswap_param[i].max_grade) + break; + } + atomic_set(&hybs->mem2zram_scale, zswap_param[i].mem2zram_scale); + atomic_set(&hybs->zram2ufs_scale, zswap_param[i].zram2ufs_scale); + atomic_set(&hybs->pagefault_level, zswap_param[i].pagefault_level); + } +} + +static void update_swapd_memcg_hybs(memcg_hybs_t *hybs) +{ + int i; + + for (i = 0; i < SWAPD_MAX_LEVEL_NUM; ++i) { + if (!zswap_param[i].min_grade && !zswap_param[i].max_grade) + return; + + if (atomic64_read(&hybs->app_grade) >= zswap_param[i].min_grade && + atomic64_read(&hybs->app_grade) <= zswap_param[i].max_grade) + break; + } + + if (i == SWAPD_MAX_LEVEL_NUM) + return; + + atomic_set(&hybs->mem2zram_scale, zswap_param[i].mem2zram_scale); + atomic_set(&hybs->zram2ufs_scale, zswap_param[i].zram2ufs_scale); + atomic_set(&hybs->pagefault_level, zswap_param[i].pagefault_level); +} + +void update_swapd_mcg_setup(struct mem_cgroup *memcg) +{ + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(memcg); + + if (!hybs) + return; + + update_swapd_memcg_hybs(hybs); +} + +static int update_swapd_mcgs_setup(char *buf) +{ + const char delim[] = " "; + char *token = NULL; + int level_num; + int i; + + buf = strstrip(buf); + token = strsep(&buf, delim); + + if (!token) + return -EINVAL; + + if (kstrtoint(token, 0, &level_num)) + return -EINVAL; + + if (level_num > SWAPD_MAX_LEVEL_NUM || level_num < 0) + return -EINVAL; + + mutex_lock(&reclaim_para_lock); + for (i = 0; i < level_num; ++i) { + token = strsep(&buf, delim); + if (!token) + goto out; + + if (kstrtoint(token, 0, &zswap_param[i].min_grade) || + zswap_param[i].min_grade > MAX_APP_GRADE) + goto out; + + token = strsep(&buf, delim); + if (!token) + goto out; + + if (kstrtoint(token, 0, &zswap_param[i].max_grade) || + zswap_param[i].max_grade > MAX_APP_GRADE) + goto out; + + token = strsep(&buf, delim); + if (!token) + goto out; + + if (kstrtoint(token, 0, &zswap_param[i].mem2zram_scale) || + zswap_param[i].mem2zram_scale > MAX_RATIO) + goto out; + + token = strsep(&buf, delim); + if (!token) + goto out; + + if (kstrtoint(token, 0, &zswap_param[i].zram2ufs_scale) || + zswap_param[i].zram2ufs_scale > MAX_RATIO) + goto out; + + token = strsep(&buf, delim); + if (!token) + goto out; + + if (kstrtoint(token, 0, &zswap_param[i].pagefault_level)) + goto out; + } + + swapd_mcgs_setup_parse(level_num); + mutex_unlock(&reclaim_para_lock); + return 0; + +out: + mutex_unlock(&reclaim_para_lock); + return -EINVAL; +} + +static ssize_t swapd_mcgs_setup_write(struct kernfs_open_file *of, char *buf, + size_t nbytes, loff_t off) +{ + int ret = update_swapd_mcgs_setup(buf); + + if (ret) + return ret; + + return nbytes; +} + +static int swapd_mcgs_setup_show(struct seq_file *m, void *v) +{ + int i; + + for (i = 0; i < SWAPD_MAX_LEVEL_NUM; ++i) { + seq_printf(m, "level %d min score: %u\n", + i, zswap_param[i].min_grade); + seq_printf(m, "level %d max score: %u\n", + i, zswap_param[i].max_grade); + seq_printf(m, "level %d ub_mem2zram_ratio: %u\n", + i, zswap_param[i].mem2zram_scale); + seq_printf(m, "level %d ub_zram2ufs_ratio: %u\n", + i, zswap_param[i].zram2ufs_scale); + seq_printf(m, "memcg %d refault_threshold: %u\n", + i, zswap_param[i].pagefault_level); + } + + return 0; +} + +static ssize_t swapd_nap_jiffies_write(struct kernfs_open_file *of, char *buf, + size_t nbytes, loff_t off) +{ + unsigned long nap; + + buf = strstrip(buf); + if (!buf) + return -EINVAL; + + if (kstrtoul(buf, 0, &nap)) + return -EINVAL; + + swapd_nap_jiffies = nap; + return nbytes; +} + +static int swapd_nap_jiffies_show(struct seq_file *m, void *v) +{ + seq_printf(m, "%lu\n", swapd_nap_jiffies); + + return 0; +} + +static ssize_t swapd_single_mcg_setup_write(struct kernfs_open_file *of, + char *buf, size_t nbytes, loff_t off) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(of_css(of)); + unsigned int mem2zram_scale; + unsigned int zram2ufs_scale; + unsigned int pagefault_level; + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(memcg); + + if (!hybs) + return -EINVAL; + + buf = strstrip(buf); + + if (sscanf(buf, "%u %u %u", &mem2zram_scale, &zram2ufs_scale, + &pagefault_level) != 3) + return -EINVAL; + + if (mem2zram_scale > MAX_RATIO || zram2ufs_scale > MAX_RATIO) + return -EINVAL; + + atomic_set(&MEMCGRP_ITEM(memcg, mem2zram_scale), mem2zram_scale); + atomic_set(&MEMCGRP_ITEM(memcg, zram2ufs_scale), zram2ufs_scale); + atomic_set(&MEMCGRP_ITEM(memcg, pagefault_level), pagefault_level); + + return nbytes; +} + +static int swapd_single_mcg_setup_show(struct seq_file *m, void *v) +{ + struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(m)); + memcg_hybs_t *hybs = MEMCGRP_ITEM_DATA(memcg); + + if (!hybs) + return -EINVAL; + + seq_printf(m, "memcg score: %llu\n", + atomic64_read(&hybs->app_grade)); + seq_printf(m, "memcg ub_mem2zram_ratio: %u\n", + atomic_read(&hybs->mem2zram_scale)); + seq_printf(m, "memcg ub_zram2ufs_ratio: %u\n", + atomic_read(&hybs->zram2ufs_scale)); + seq_printf(m, "memcg refault_threshold: %u\n", + atomic_read(&hybs->pagefault_level)); + + return 0; +} + +static int mem_cgroup_zram_wm_scale_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val > MAX_RATIO || val < MIN_RATIO) + return -EINVAL; + + atomic64_set(&zram_wm_scale, val); + + return 0; +} + +static s64 mem_cgroup_zram_wm_scale_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(&zram_wm_scale); +} + +static int mem_cgroup_compress_scale_write(struct cgroup_subsys_state *css, + struct cftype *cft, s64 val) +{ + if (val > MAX_RATIO || val < MIN_RATIO) + return -EINVAL; + + atomic64_set(&compress_scale, val); + + return 0; +} + +static s64 mem_cgroup_compress_scale_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return atomic64_read(&compress_scale); +} + +static int memcg_active_app_info_list_show(struct seq_file *m, void *v) +{ + struct mem_cgroup *memcg = NULL; + unsigned long anon_size; + unsigned long zram_size; + unsigned long eswap_size; + + while ((memcg = fetch_next_memcg(memcg))) { + u64 grade; + + if (!MEMCGRP_ITEM_DATA(memcg)) + continue; + + grade = atomic64_read(&MEMCGRP_ITEM(memcg, app_grade)); + anon_size = memcg_anon_pages(memcg); + eswap_size = hybridswap_read_mcg_stats(memcg, + MCG_DISK_STORED_PG_SZ); + zram_size = hybridswap_read_mcg_stats(memcg, + MCG_ZRAM_STORED_PG_SZ); + + if (anon_size + zram_size + eswap_size == 0) + continue; + + if (!strlen(MEMCGRP_ITEM(memcg, name))) + continue; + + anon_size *= PAGE_SIZE / SZ_1K; + zram_size *= PAGE_SIZE / SZ_1K; + eswap_size *= PAGE_SIZE / SZ_1K; + + seq_printf(m, "%s %llu %lu %lu %lu %llu\n", + MEMCGRP_ITEM(memcg, name), grade, + anon_size, zram_size, eswap_size, + MEMCGRP_ITEM(memcg, reclaimed_pagefault)); + } + return 0; +} + +static unsigned long fetch_totalreserve_pages(void) +{ + int nid; + unsigned long val = 0; + + for_each_node_state(nid, N_MEMORY) { + pg_data_t *pgdat = NODE_DATA(nid); + + if (pgdat) + val += pgdat->totalreserve_pages; + } + + return val; +} + +struct pglist_data *first_online_pgdat(void) +{ + return NODE_DATA(first_online_node); +} + +struct pglist_data *next_online_pgdat(struct pglist_data *pgdat) +{ + int nid = next_online_node(pgdat->node_id); + + if (nid == MAX_NUMNODES) + return NULL; + return NODE_DATA(nid); +} + +struct zone *next_zone(struct zone *zone) +{ + pg_data_t *pgdat = zone->zone_pgdat; + + if (zone < pgdat->node_zones + MAX_NR_ZONES - 1) + zone++; + else { + pgdat = next_online_pgdat(pgdat); + if (pgdat) + zone = pgdat->node_zones; + else + zone = NULL; + } + return zone; +} + +unsigned int system_cur_usable_mem(void) +{ + unsigned long reclaimable; + long buffers; + unsigned long pagecache; + unsigned long wmark_low = 0; + struct zone *zone; + + buffers = global_zone_page_state(NR_FREE_PAGES) - all_totalreserve_pages; + + for_each_zone(zone) + wmark_low += low_wmark_pages(zone); + pagecache = global_node_page_state(NR_ACTIVE_FILE) + + global_node_page_state(NR_INACTIVE_FILE); + pagecache -= min(pagecache / 2, wmark_low); + buffers += pagecache; + +#if LINUX_VERSION_CODE < KERNEL_VERSION(5, 10, 0) + reclaimable = global_node_page_state(NR_SLAB_RECLAIMABLE) + + global_node_page_state(NR_KERNEL_MISC_RECLAIMABLE); +#else + reclaimable = global_node_page_state(NR_SLAB_RECLAIMABLE_B) + + global_node_page_state(NR_KERNEL_MISC_RECLAIMABLE); +#endif + buffers += reclaimable - min(reclaimable / 2, wmark_low); + + if (buffers < 0) + buffers = 0; + + return buffers >> 8; /* pages to MB */ +} + +static bool min_buffer_is_suitable(void) +{ + u32 curr_buffers = system_cur_usable_mem(); + + if (curr_buffers >= fetch_min_mem_watermark_value()) + return true; + + return false; +} + +bool high_buffer_is_suitable(void) +{ + u32 curr_buffers = system_cur_usable_mem(); + + if (curr_buffers >= fetch_high_mem_watermark_value()) + return true; + + return false; +} + +static void refresh_pagefaults(void) +{ + struct mem_cgroup *memcg = NULL; + + while ((memcg = fetch_next_memcg(memcg))) { + MEMCGRP_ITEM(memcg, reclaimed_pagefault) = + hybridswap_read_mcg_stats(memcg, MCG_ANON_FAULT_CNT); + } + + infos_last_anon_pagefault = hybridswap_fetch_zram_pagefault(); + last_refresh_t = jiffies; +} + +bool fetch_memcg_pagefault_status(struct mem_cgroup *memcg, + pg_data_t *pgdat) +{ + const unsigned int percent_constant = 100; + unsigned long long cur_anon_pagefault; + unsigned long anon_total; + unsigned int scale, thresh; + memcg_hybs_t *hybs; + + if (!memcg || !MEMCGRP_ITEM_DATA(memcg)) + return false; + + hybs = MEMCGRP_ITEM_DATA(memcg); + thresh = atomic_read(&hybs->pagefault_level); + if (thresh == 0) + return false; + + cur_anon_pagefault = hybridswap_read_mcg_stats(memcg, MCG_ANON_FAULT_CNT); + if (cur_anon_pagefault == hybs->reclaimed_pagefault) + return false; + + anon_total = memcg_anon_pages(memcg) + + hybridswap_read_mcg_stats(memcg, MCG_DISK_STORED_PG_SZ) + + hybridswap_read_mcg_stats(memcg, MCG_ZRAM_STORED_PG_SZ); + scale = (cur_anon_pagefault - hybs->reclaimed_pagefault) * + percent_constant / (anon_total + 1); + hybp(HYB_INFO, "CHECK MEMCG PAGEFAULT %s: anon pagefault %u%%(%u)\n", hybs->name, + scale, thresh); + + if (scale >= thresh) + return true; + + return false; +} + +static bool hybridswap_scale_ok(void) +{ + struct hybstatus *stat = NULL; + + stat = hybridswap_fetch_stat_obj(); + if (unlikely(!stat)) { + hybp(HYB_ERR, "can't fetch stat obj!\n"); + + return false; + } + + return (atomic64_read(&stat->zram_stored_pages) > + atomic64_read(&stat->stored_pages)); +} + +static bool fetch_infos_pagefault_status(void) +{ + const unsigned int percent_constant = 1000; + unsigned long long cur_anon_pagefault; + unsigned long long cur_time; + unsigned long long scale = 0; + + cur_anon_pagefault = hybridswap_fetch_zram_pagefault(); + cur_time = jiffies; + + if (cur_anon_pagefault == infos_last_anon_pagefault + || cur_time == last_refresh_t) + goto false_out; + + scale = (cur_anon_pagefault - infos_last_anon_pagefault) * + percent_constant / (jiffies_to_msecs(cur_time - + last_refresh_t) + 1); + + total_pagefault_percent = scale; + + if (scale > fetch_infos_pagefault_level_value()) + return true; + + hybp(HYB_INFO, "current %llu t %llu last %llu t %lu scale %llu pagefault_scale %llu\n", + cur_anon_pagefault, cur_time, + infos_last_anon_pagefault, last_refresh_t, + scale, (unsigned long long)infos_pagefault_level.counter); +false_out: + return false; +} + +static int reclaim_exceed_sleep_ms_write( + struct cgroup_subsys_state *css, struct cftype *cft, s64 val) +{ + if (val < 0) + return -EINVAL; + + reclaim_exceed_sleep_ms = val; + + return 0; +} + +static s64 reclaim_exceed_sleep_ms_read( + struct cgroup_subsys_state *css, struct cftype *cft) +{ + return reclaim_exceed_sleep_ms; +} + +static int max_reclaimin_size_mb_write( + struct cgroup_subsys_state *css, struct cftype *cft, u64 val) +{ + max_reclaimin_size = (val << 20); + + return 0; +} + +static u64 max_reclaimin_size_mb_read( + struct cgroup_subsys_state *css, struct cftype *cft) +{ + return max_reclaimin_size >> 20; +} + +static ssize_t swapd_shrink_parameter_write(struct kernfs_open_file *of, + char *buf, size_t nbytes, loff_t off) +{ + unsigned long window, limit; + + buf = strstrip(buf); + if (sscanf(buf, "%lu %lu", &window, &limit) != 2) + return -EINVAL; + + swapd_shrink_window = msecs_to_jiffies(window); + swapd_shrink_limit_per_window = limit; + + return nbytes; +} + +static int swapd_shrink_parameter_show(struct seq_file *m, void *v) +{ + seq_printf(m, "%-32s %lu(jiffies) %u(msec)\n", "swapd_shrink_window", + swapd_shrink_window, jiffies_to_msecs(swapd_shrink_window)); + seq_printf(m, "%-32s %lu MB\n", "swapd_shrink_limit_per_window", + swapd_shrink_limit_per_window); + seq_printf(m, "%-32s %u msec\n", "swapd_last_window", + jiffies_to_msecs(jiffies - swapd_last_window_start)); + seq_printf(m, "%-32s %lu MB\n", "swapd_last_window_shrink", + swapd_last_window_shrink); + + return 0; +} + +static int swapd_update_cpumask(struct task_struct *tsk, char *buf, + struct pglist_data *pgdat) +{ + int retval; + struct cpumask temp_mask; + const struct cpumask *cpumask = cpumask_of_node(pgdat->node_id); + struct hybridswapd_task* hyb_task = PGDAT_ITEM_DATA(pgdat); + + if (unlikely(!hyb_task)) { + hybp(HYB_ERR, "set task %s cpumask %s node %d failed, " + "hyb_task is NULL\n", tsk->comm, buf, pgdat->node_id); + return -EINVAL; + } + + cpumask_clear(&temp_mask); + retval = cpulist_parse(buf, &temp_mask); + if (retval < 0 || cpumask_empty(&temp_mask)) { + hybp(HYB_ERR, "%s are invalid, use default\n", buf); + goto use_default; + } + + if (!cpumask_subset(&temp_mask, cpu_present_mask)) { + hybp(HYB_ERR, "%s is not subset of cpu_present_mask, use default\n", + buf); + goto use_default; + } + + if (!cpumask_subset(&temp_mask, cpumask)) { + hybp(HYB_ERR, "%s is not subset of cpumask, use default\n", buf); + goto use_default; + } + + set_cpus_allowed_ptr(tsk, &temp_mask); + cpumask_copy(&hyb_task->swapd_bind_cpumask, &temp_mask); + return 0; + +use_default: + if (cpumask_empty(&hyb_task->swapd_bind_cpumask)) + set_cpus_allowed_ptr(tsk, cpumask); + return -EINVAL; +} + +static ssize_t swapd_bind_write(struct kernfs_open_file *of, char *buf, + size_t nbytes, loff_t off) +{ + int ret = 0, nid; + struct pglist_data *pgdat; + + buf = strstrip(buf); + for_each_node_state(nid, N_MEMORY) { + pgdat = NODE_DATA(nid); + if (!PGDAT_ITEM_DATA(pgdat)) + continue; + + if (PGDAT_ITEM(pgdat, swapd)) { + ret = swapd_update_cpumask(PGDAT_ITEM(pgdat, swapd), + buf, pgdat); + if (ret) + break; + } + } + + if (ret) + return ret; + + return nbytes; +} + +static int swapd_bind_read(struct seq_file *m, void *v) +{ + int nid; + struct pglist_data *pgdat; + struct hybridswapd_task* hyb_task; + + seq_printf(m, "%4s %s\n", "Node", "mask"); + for_each_node_state(nid, N_MEMORY) { + pgdat = NODE_DATA(nid); + hyb_task = PGDAT_ITEM_DATA(pgdat); + if (!hyb_task) + continue; + + if (!hyb_task->swapd) + continue; + seq_printf(m, "%4d %*pbl\n", nid, + cpumask_pr_args(&hyb_task->swapd_bind_cpumask)); + } + + return 0; +} + +static s64 mem_cgroup_version_read(struct cgroup_subsys_state *css, + struct cftype *cft) +{ + return MOTO_SWAP_VERSION; +} + +struct cftype mem_cgroup_swapd_legacy_files[] = { + { + .name = "moto_swap_version", + .flags = CFTYPE_ONLY_ON_ROOT, + .read_s64 = mem_cgroup_version_read, + }, + { + .name = "active_app_info_list", + .flags = CFTYPE_ONLY_ON_ROOT, + .seq_show = memcg_active_app_info_list_show, + }, + { + .name = "zram_wm_ratio", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = mem_cgroup_zram_wm_scale_write, + .read_s64 = mem_cgroup_zram_wm_scale_read, + }, + { + .name = "compress_ratio", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = mem_cgroup_compress_scale_write, + .read_s64 = mem_cgroup_compress_scale_read, + }, + { + .name = "swapd_pid", + .flags = CFTYPE_ONLY_ON_ROOT, + .read_s64 = swapid_read, + }, + { + .name = "avail_buffers", + .flags = CFTYPE_ONLY_ON_ROOT, + .write = usable_mem_params_write, + .seq_show = usable_mem_params_show, + }, + { + .name = "swapd_max_reclaim_size", + .flags = CFTYPE_ONLY_ON_ROOT, + .write = swapd_max_reclaim_size_write, + .seq_show = swapd_max_reclaim_size_show, + }, + { + .name = "area_anon_refault_threshold", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = infos_pagefault_level_write, + .read_s64 = infos_pagefault_level_read, + }, + { + .name = "empty_round_skip_interval", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = nothing_ignore_skip_interval_write, + .read_s64 = nothing_ignore_skip_interval_read, + }, + { + .name = "max_skip_interval", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = max_skip_interval_write, + .read_s64 = max_skip_interval_read, + }, + { + .name = "empty_round_check_threshold", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = nothing_ignore_check_level_write, + .read_s64 = nothing_ignore_check_level_read, + }, + { + .name = "anon_refault_snapshot_min_interval", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = pagefault_refresh_min_write, + .read_s64 = pagefault_refresh_min_read, + }, + { + .name = "swapd_mcgs_setup", + .flags = CFTYPE_ONLY_ON_ROOT, + .write = swapd_mcgs_setup_write, + .seq_show = swapd_mcgs_setup_show, + }, + { + .name = "swapd_single_mcg_setup", + .write = swapd_single_mcg_setup_write, + .seq_show = swapd_single_mcg_setup_show, + }, + { + .name = "zram_critical_threshold", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = zram_critical_thres_write, + .read_s64 = zram_critical_thres_read, + }, + { + .name = "cpuload_threshold", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = cpuload_level_write, + .read_s64 = cpuload_level_read, + }, + { + .name = "reclaim_exceed_sleep_ms", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_s64 = reclaim_exceed_sleep_ms_write, + .read_s64 = reclaim_exceed_sleep_ms_read, + }, + { + .name = "swapd_bind", + .flags = CFTYPE_ONLY_ON_ROOT, + .write = swapd_bind_write, + .seq_show = swapd_bind_read, + }, + { + .name = "max_reclaimin_size_mb", + .flags = CFTYPE_ONLY_ON_ROOT, + .write_u64 = max_reclaimin_size_mb_write, + .read_u64 = max_reclaimin_size_mb_read, + }, + { + .name = "swapd_shrink_parameter", + .flags = CFTYPE_ONLY_ON_ROOT, + .write = swapd_shrink_parameter_write, + .seq_show = swapd_shrink_parameter_show, + }, + { + .name = "swapd_nap_jiffies", + .flags = CFTYPE_ONLY_ON_ROOT, + .write = swapd_nap_jiffies_write, + .seq_show = swapd_nap_jiffies_show, + }, + { }, /* terminate */ +}; + +void wakeup_refresh_daemon(void) +{ + unsigned long curr_refresh = + jiffies_to_msecs(jiffies - last_refresh_t); + + if (curr_refresh >= + fetch_pagefault_refresh_min_value()) { + atomic_set(&refresh_daemonwait_flag, 1); + wake_up_interruptible(&refresh_daemonwait); + } +} + +static int refresh_daemon(void *p) +{ + int ret; + + while (!kthread_should_stop()) { + ret = wait_event_interruptible(refresh_daemonwait, + atomic_read(&refresh_daemonwait_flag)); + if (ret) + continue; + + if (unlikely(kthread_should_stop())) + break; + + atomic_set(&refresh_daemonwait_flag, 0); + + refresh_pagefaults(); + count_swapd_event(SWAPD_SNAPSHOT_TIMES); + } + + return 0; +} + +static int refresh_daemonrun(void) +{ + atomic_set(&refresh_daemonwait_flag, 0); + init_waitqueue_head(&refresh_daemonwait); + refresh_daemontask = kthread_run(refresh_daemon, NULL, "snapshotd"); + + if (IS_ERR(refresh_daemontask)) { + hybp(HYB_ERR, "Failed to start refresh_daemon\n"); + return PTR_ERR(refresh_daemontask); + } + + return 0; +} + +static void refresh_daemonexit(void) +{ + if (refresh_daemontask) { + atomic_set(&refresh_daemonwait_flag, 1); + kthread_stop(refresh_daemontask); + } + refresh_daemontask = NULL; +} + +unsigned long fetch_nr_zram_total(void) +{ + unsigned long nr_zram = 1; + + if (!swapd_zram) + return nr_zram; + + nr_zram = swapd_zram->disksize >> PAGE_SHIFT; +#if (defined CONFIG_ZRAM_WRITEBACK) || (defined CONFIG_HYBRIDSWAP_CORE) + nr_zram -= swapd_zram->increase_nr_pages; +#endif + return nr_zram ?: 1; +} + +u64 get_hybridswap_meminfo(const char *type) +{ + if (!type || !swapd_zram) + return 0; + + if (!strcmp(type, "same_pages")) + return (u64)atomic64_read(&swapd_zram->stats.same_pages); + if (!strcmp(type, "compr_data_size")) + return (u64)atomic64_read(&swapd_zram->stats.compr_data_size); + if (!strcmp(type, "pages_stored")) + return (u64)atomic64_read(&swapd_zram->stats.pages_stored); + return 0; +} +EXPORT_SYMBOL(get_hybridswap_meminfo); + +bool zram_watermark_ok(void) +{ + long long diff_buffers; + long long wm = 0; + long long cur_scale = 0; + unsigned long zram_used = hybridswap_fetch_zram_used_pages(); + const unsigned int percent_constant = 100; + + diff_buffers = fetch_high_mem_watermark_value() - + system_cur_usable_mem(); + diff_buffers *= SZ_1M / PAGE_SIZE; + diff_buffers *= fetch_compress_scale_value() / 10; + diff_buffers = diff_buffers * percent_constant / fetch_nr_zram_total(); + + cur_scale = zram_used * percent_constant / fetch_nr_zram_total(); + wm = min(fetch_zram_wm_scale_value(), fetch_zram_wm_scale_value()- diff_buffers); + + return cur_scale > wm; +} + +static inline bool zram_is_full(void) +{ + unsigned long nr_used = hybridswap_fetch_zram_used_pages(); + unsigned long nr_total = fetch_nr_zram_total(); + + return nr_used >= nr_total; +} + +bool free_zram_is_ok(void) +{ + unsigned long nr_used = hybridswap_fetch_zram_used_pages(); + unsigned long nr_total = fetch_nr_zram_total(); + unsigned long reserve = nr_total >> 6; + + return (nr_used < (nr_total - reserve)); +} + +bool free_swap_is_low(void) +{ + struct sysinfo info; + si_swapinfo(&info); + return (info.freeswap < fetch_free_swap_level_value()); +} + +static bool zram_need_swapout(void) +{ + bool zram_wm_ok = zram_watermark_ok(); + bool avail_buffer_wm_ok = !high_buffer_is_suitable(); + bool ufs_wm_ok = true; + +#ifdef CONFIG_HYBRIDSWAP_CORE + ufs_wm_ok = hybridswap_stored_wm_ok(); +#endif + + if (zram_wm_ok && avail_buffer_wm_ok && ufs_wm_ok) + return true; + + hybp(HYB_INFO, "zram_wm_ok %d avail_buffer_wm_ok %d ufs_wm_ok %d\n", + zram_wm_ok, avail_buffer_wm_ok, ufs_wm_ok); + + return false; +} + +bool zram_watermark_exceed(void) +{ + u64 nr_zram_used; + u64 nr_wm = fetch_zram_critical_level_value(); + + if (!nr_wm) + return false; + + nr_zram_used = hybridswap_fetch_zram_used_pages(); + + if (nr_zram_used > nr_wm) + return true; + + return false; +} + +#ifdef CONFIG_OPLUS_JANK +static bool is_cpu_busy(void) +{ + unsigned int cpuload = 0; + int i; + struct cpumask mask; + + cpumask_clear(&mask); + + for (i = 0; i < 6; i++) + cpumask_set_cpu(i, &mask); + + cpuload = fetch_cpu_load(1, &mask); + if (cpuload > fetch_cpuload_level_value()) { + hybp(HYB_INFO, "cpuload %d\n", cpuload); + return true; + } + + return false; +} +#endif + +static void wakeup_swapd(pg_data_t *pgdat) +{ + unsigned long curr_interval; + struct hybridswapd_task* hyb_task = PGDAT_ITEM_DATA(pgdat); + + if (!hyb_task || !hyb_task->swapd) + return; + + if (atomic_read(&swapd_pause)) { + count_swapd_event(SWAPD_MANUAL_PAUSE); + return; + } + +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) + if (atomic_read(&display_off)) + return; +#endif + + if (!waitqueue_active(&hyb_task->swapd_wait)) + return; + + if (atomic_read(&refresh_daemoninit_flag) == 1) + wakeup_refresh_daemon(); + + if (min_buffer_is_suitable()) { + count_swapd_event(SWAPD_OVER_MIN_BUFFER_SKIP_TIMES); + return; + } + + curr_interval = jiffies_to_msecs(jiffies - last_swapd_time); + if (curr_interval < swapd_skip_interval) { + count_swapd_event(SWAPD_EMPTY_ROUND_SKIP_TIMES); + return; + } + + atomic_set(&hyb_task->swapd_wait_flag, 1); + wake_up_interruptible(&hyb_task->swapd_wait); +} + +void wake_all_swapd(void) +{ + pg_data_t *pgdat = NULL; + int nid; + + for_each_online_node(nid) { + pgdat = NODE_DATA(nid); + wakeup_swapd(pgdat); + } +} + +static inline u64 __calc_nr_to_reclaim(void) +{ + u32 curr_buffers; + u64 high_buffers; + u64 max_reclaim_size_value; + u64 reclaim_size = 0; + + high_buffers = fetch_high_mem_watermark_value(); + curr_buffers = system_cur_usable_mem(); + max_reclaim_size_value = fetch_swapd_max_reclaim_size(); + if (curr_buffers < high_buffers) + reclaim_size = high_buffers - curr_buffers; + + reclaim_size = min(reclaim_size, max_reclaim_size_value); + + return reclaim_size * SZ_1M / PAGE_SIZE; +} + +static inline u64 calc_shrink_scale(pg_data_t *pgdat) +{ + struct mem_cgroup *memcg = NULL; + const u32 percent_constant = 100; + u64 total_can_reclaimed = 0; + + while ((memcg = fetch_next_memcg(memcg))) { + s64 nr_anon, nr_zram, nr_eswap, total, can_reclaimed, thresh; + memcg_hybs_t *hybs; + + hybs = MEMCGRP_ITEM_DATA(memcg); + thresh = atomic_read(&hybs->mem2zram_scale); + if (!thresh || fetch_memcg_pagefault_status(memcg, pgdat)) { + hybs->can_reclaimed = 0; + continue; + } + + nr_anon = memcg_anon_pages(memcg); + if (nr_anon == 0) { + hybs->can_reclaimed = 0; + continue; + } + + nr_zram = hybridswap_read_mcg_stats(memcg, MCG_ZRAM_STORED_PG_SZ); + nr_eswap = hybridswap_read_mcg_stats(memcg, MCG_DISK_STORED_PG_SZ); + total = nr_anon + nr_zram + nr_eswap; + + can_reclaimed = total * thresh / percent_constant; + if (can_reclaimed <= (nr_zram + nr_eswap)) + hybs->can_reclaimed = 0; + else + hybs->can_reclaimed = can_reclaimed - (nr_zram + nr_eswap); + hybp(HYB_INFO, "CHECK MEMCG %s: can_reclaim %luKB nr_anon %luKB zram %luKB eswap %luKB total %luKB reclaimed %lu%%(%lu)\n", + hybs->name, (unsigned long)page_to_kb(hybs->can_reclaimed), + (unsigned long)page_to_kb(nr_anon), (unsigned long)page_to_kb(nr_zram), + (unsigned long)page_to_kb(nr_eswap), (unsigned long)page_to_kb(total), + (unsigned long)((nr_zram + nr_eswap) * 100 / (total + 1)), + (unsigned long)thresh); + total_can_reclaimed += hybs->can_reclaimed; + } + + return total_can_reclaimed; +} + +static unsigned long swapd_shrink_anon(pg_data_t *pgdat, + unsigned long nr_to_reclaim) +{ + struct mem_cgroup *memcg = NULL; + unsigned long nr_reclaimed = 0; + unsigned long reclaim_memcg_cnt = 0; + u64 total_can_reclaimed = calc_shrink_scale(pgdat); + unsigned long start_js = jiffies; + bool exit = false; + unsigned long RECLAIM_PAGES_PER_CYCLE = PAGES_PER_1MB * 100; + unsigned long reclaim_pages_this_cycle = 0; + unsigned long reclaim_cycles = 0; + + if (unlikely(total_can_reclaimed == 0)) + goto out; + + if (total_can_reclaimed < nr_to_reclaim) + nr_to_reclaim = total_can_reclaimed; + + reclaim_cycles = (nr_to_reclaim / RECLAIM_PAGES_PER_CYCLE) + (nr_to_reclaim % RECLAIM_PAGES_PER_CYCLE > 0 ? 1 : 0); + + hybp(HYB_INFO, "SWAPD_SHRINK ANON + available %uMB(%lu) to_reclaim %luKB can_reclaimed %luKB reclaim_cycles %lu", + (unsigned int)system_cur_usable_mem(), (unsigned long)fetch_high_mem_watermark_value(), + page_to_kb(nr_to_reclaim), (unsigned long)page_to_kb(total_can_reclaimed), reclaim_cycles); + + while (reclaim_cycles) { + reclaim_pages_this_cycle = min(nr_to_reclaim - nr_reclaimed, RECLAIM_PAGES_PER_CYCLE); + if (reclaim_pages_this_cycle <= 0) break; + + while ((memcg = fetch_next_memcg(memcg))) { + unsigned long memcg_nr_reclaimed, memcg_to_reclaim; + memcg_hybs_t *hybs; + + if (high_buffer_is_suitable() || atomic_read(&swapd_pause)) { + fetch_next_memcg_break(memcg); + exit = true; + break; + } + + hybs = MEMCGRP_ITEM_DATA(memcg); + if (!hybs->can_reclaimed) + continue; + + memcg_to_reclaim = reclaim_pages_this_cycle * hybs->can_reclaimed / total_can_reclaimed; + if (memcg_to_reclaim > 0) { + memcg_nr_reclaimed = try_to_free_mem_cgroup_pages(memcg, + memcg_to_reclaim, GFP_KERNEL, true); + reclaim_memcg_cnt++; + hybs->can_reclaimed -= memcg_nr_reclaimed; + hybp(HYB_INFO, "SHRINK MEMCG %s: to_reclaim %lu reclaimed %lu\n", hybs->name, + memcg_to_reclaim, memcg_nr_reclaimed); + nr_reclaimed += memcg_nr_reclaimed; + } + if (nr_reclaimed >= nr_to_reclaim) { + fetch_next_memcg_break(memcg); + exit = true; + break; + } + + if (hybs->can_reclaimed < 0 || memcg_nr_reclaimed == 0) // no pages can be reclaimed, skip it. + hybs->can_reclaimed = 0; + + if (swapd_nap_jiffies && time_after_eq(jiffies, start_js + swapd_nap_jiffies)) { + set_current_state(TASK_INTERRUPTIBLE); + schedule_timeout((jiffies - start_js) * 2); + start_js = jiffies; + } + } + if (exit) + break; + + if (zram_is_full()) + break; + reclaim_cycles--; + } + +out: + hybp(HYB_INFO, "SWAPD_SHRINK ANON - available %uMB(%lu) reclaimed %luKB from memcg %lu\n", + (unsigned int)system_cur_usable_mem(), (unsigned long)fetch_high_mem_watermark_value(), + page_to_kb(nr_reclaimed), reclaim_memcg_cnt); + return nr_reclaimed; +} + +static void swapd_shrink_node(pg_data_t *pgdat) +{ + const unsigned int increase_rate = 2; + unsigned long nr_reclaimed = 0; + unsigned long nr_to_reclaim; + +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) + if (atomic_read(&display_off)) + return; +#endif + + if (zram_is_full()) + return; + + if (high_buffer_is_suitable()) + return; + +#ifdef CONFIG_OPLUS_JANK + if (is_cpu_busy()) { + count_swapd_event(SWAPD_CPU_BUSY_BREAK_TIMES); + return; + } +#endif + + if ((jiffies - swapd_last_window_start) < swapd_shrink_window) { + if (swapd_last_window_shrink >= swapd_shrink_limit_per_window) { + count_swapd_event(SWAPD_SKIP_SHRINK_OF_WINDOW); + hybp(HYB_INFO, "swapd_last_window_shrink %lu, skip shrink\n", + swapd_last_window_shrink); + set_current_state(TASK_INTERRUPTIBLE); + schedule_timeout(msecs_to_jiffies(reclaim_exceed_sleep_ms)); + return; + } + } else { + swapd_last_window_start = jiffies; + swapd_last_window_shrink = 0lu; + } + + nr_to_reclaim = __calc_nr_to_reclaim(); + if (!nr_to_reclaim) + return; + + count_swapd_event(SWAPD_SHRINK_ANON); + nr_reclaimed = swapd_shrink_anon(pgdat, nr_to_reclaim); + swapd_last_window_shrink += PAGES_TO_MB(nr_reclaimed); + + if (nr_reclaimed < fetch_nothing_ignore_check_level_value()) { + count_swapd_event(SWAPD_EMPTY_ROUND); + if (last_round_is_empty) + swapd_skip_interval = min(swapd_skip_interval * + increase_rate, + fetch_max_skip_interval_value()); + else + swapd_skip_interval = + fetch_nothing_ignore_skip_interval_value(); + last_round_is_empty = true; + hybp(HYB_INFO, "SWAPD_SHRINK_EMPTY_ROUND, reclaimed %lu KB, swapd_skip_interval %llu\n", + (unsigned long)nr_reclaimed * 4, swapd_skip_interval); + } else { + swapd_skip_interval = 0; + last_round_is_empty = false; + } +} + +static int swapd(void *p) +{ + pg_data_t *pgdat = (pg_data_t *)p; + struct task_struct *tsk = current; + struct hybridswapd_task* hyb_task = PGDAT_ITEM_DATA(pgdat); + static unsigned long last_reclaimin_jiffies = 0; + long page_fault_pause_value; + int display_un_blank = 1; + + swapid = tsk->pid; + + cpumask_clear(&hyb_task->swapd_bind_cpumask); + (void)swapd_update_cpumask(tsk, SWAPD_DEFAULT_BIND_CPUS, pgdat); + set_freezable(); + + swapd_last_window_start = jiffies - swapd_shrink_window; + while (!kthread_should_stop()) { + bool pagefault = false; + u64 available, wmhigh; + u64 to_swapout, swapped_size = 0; + + wait_event_freezable(hyb_task->swapd_wait, + atomic_read(&hyb_task->swapd_wait_flag)); + atomic_set(&hyb_task->swapd_wait_flag, 0); + if (unlikely(kthread_should_stop())) + break; + count_swapd_event(SWAPD_WAKEUP); + hybp(HYB_INFO, "SWAPD_WAKEUP"); + + if (fetch_infos_pagefault_status() && hybridswap_scale_ok()) { + pagefault = true; + count_swapd_event(SWAPD_REFAULT); + goto do_eswap; + } + + swapd_shrink_node(pgdat); + last_swapd_time = jiffies; +do_eswap: + page_fault_pause_value = atomic_long_read(&page_fault_pause); +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) + display_un_blank = !atomic_read(&display_off); +#endif + if (!hybridswap_reclaim_work_running() && display_un_blank && + (zram_need_swapout() || pagefault) && !page_fault_pause_value && + jiffies_to_msecs(jiffies - last_reclaimin_jiffies) >= 50) { + wmhigh = fetch_high_mem_watermark_value(); + available = system_cur_usable_mem(); + + if (available < wmhigh && !atomic_read(&swapd_pause)) { + to_swapout = (wmhigh - available) * SZ_1M; + to_swapout = min_t(u64, to_swapout, max_reclaimin_size); +#ifdef CONFIG_HYBRIDSWAP_CORE + hybp(HYB_INFO, "SWAPD_ESWAPOUT + available %uMB(%lu) to_swapout %luMB", + (unsigned int)available, (unsigned long)wmhigh, (unsigned long)(to_swapout / SZ_1M)); + swapped_size = hybridswap_out_to_eswap(to_swapout); + count_swapd_event(SWAPD_SWAPOUT); + last_reclaimin_jiffies = jiffies; + hybp(HYB_INFO, "SWAPD_ESWAPOUT - swapped out %luMB\n", (unsigned long)(swapped_size / SZ_1M)); +#endif + } else { + count_swapd_event(SWAPD_SKIP_SWAPOUT); + } + } + hybp(HYB_INFO, "SWAPD_SLEEP"); + } + + return 0; +} + +int swapd_run(int nid) +{ + pg_data_t *pgdat = NODE_DATA(nid); + struct sched_param param = { + .sched_priority = DEFAULT_PRIO, + }; + struct hybridswapd_task* hyb_task = PGDAT_ITEM_DATA(pgdat); + int ret; + + if (!hyb_task || hyb_task->swapd) + return 0; + + atomic_set(&hyb_task->swapd_wait_flag, 0); + hyb_task->swapd = kthread_create(swapd, pgdat, "mswapd:%d", nid); + if (IS_ERR(hyb_task->swapd)) { + hybp(HYB_ERR, "Failed to start swapd on node %d\n", nid); + ret = PTR_ERR(hyb_task->swapd); + hyb_task->swapd = NULL; + return ret; + } + + sched_setscheduler_nocheck(hyb_task->swapd, SCHED_NORMAL, ¶m); + set_user_nice(hyb_task->swapd, PRIO_TO_NICE(param.sched_priority)); + wake_up_process(hyb_task->swapd); + + return 0; +} + +void swapd_stop(int nid) +{ + struct pglist_data *pgdata = NODE_DATA(nid); + struct task_struct *swapd; + struct hybridswapd_task* hyb_task; + + if (unlikely(!PGDAT_ITEM_DATA(pgdata))) { + hybp(HYB_ERR, "nid %d pgdata %p PGDAT_ITEM_DATA is NULL\n", + nid, pgdata); + return; + } + + hyb_task = PGDAT_ITEM_DATA(pgdata); + swapd = hyb_task->swapd; + if (swapd) { + atomic_set(&hyb_task->swapd_wait_flag, 1); + kthread_stop(swapd); + hyb_task->swapd = NULL; + } + + swapid = -1; +} + +static int mem_hotplug_swapd_notifier(struct notifier_block *nb, + unsigned long action, void *data) +{ + struct memory_notify *arg = (struct memory_notify*)data; + int nid = arg->status_change_nid; + + if (action == MEM_ONLINE) + swapd_run(nid); + else if (action == MEM_OFFLINE) + swapd_stop(nid); + + return NOTIFY_OK; +} + +static struct notifier_block swapd_notifier_nb = { + .notifier_call = mem_hotplug_swapd_notifier, +}; + +static int swapd_cpu_online(unsigned int cpu) +{ + int nid; + + for_each_node_state(nid, N_MEMORY) { + pg_data_t *pgdat = NODE_DATA(nid); + struct hybridswapd_task* hyb_task; + struct cpumask *mask; + + hyb_task = PGDAT_ITEM_DATA(pgdat); + mask = &hyb_task->swapd_bind_cpumask; + + if (cpumask_any_and(cpu_online_mask, mask) < nr_cpu_ids) + set_cpus_allowed_ptr(PGDAT_ITEM(pgdat, swapd), mask); + } + return 0; +} + +void alloc_pages_slowpath_hook(void *data, gfp_t gfp_flags, + unsigned int order, unsigned long delta) +{ + if (gfp_flags & __GFP_KSWAPD_RECLAIM) + wake_all_swapd(); +} + +void rmqueue_hook(void *data, struct zone *preferred_zone, + struct zone *zone, unsigned int order, gfp_t gfp_flags, + unsigned int alloc_flags, int migratetype) +{ + if (gfp_flags & __GFP_KSWAPD_RECLAIM) + wake_all_swapd(); +} + +static int create_swapd_thread(struct zram *zram) +{ + int nid; + int ret; + struct pglist_data *pgdat; + struct hybridswapd_task *tsk_info; + + for_each_node(nid) { + pgdat = NODE_DATA(nid); + if (!PGDAT_ITEM_DATA(pgdat)) { + tsk_info = kzalloc(sizeof(struct hybridswapd_task), + GFP_KERNEL); + if (!tsk_info) { + hybp(HYB_ERR, "kmalloc tsk_info failed node %d\n", nid); + goto error_out; + } + + pgdat->android_oem_data1 = (u64)tsk_info; + } + + init_waitqueue_head(&PGDAT_ITEM(pgdat, swapd_wait)); + } + + for_each_node_state(nid, N_MEMORY) { + if (swapd_run(nid)) + goto error_out; + } + + ret = cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN, + "mm/swapd:online", swapd_cpu_online, NULL); + if (ret < 0) { + hybp(HYB_ERR, "swapd: failed to register hotplug callbacks.\n"); + goto error_out; + } + swapd_online = ret; + + return 0; + +error_out: + for_each_node(nid) { + pgdat = NODE_DATA(node); + + if (!PGDAT_ITEM_DATA(pgdat)) + continue; + + if (PGDAT_ITEM(pgdat, swapd)) { + kthread_stop(PGDAT_ITEM(pgdat, swapd)); + PGDAT_ITEM(pgdat, swapd) = NULL; + } + + kfree((void*)PGDAT_ITEM_DATA(pgdat)); + pgdat->android_oem_data1 = 0; + } + + return -ENOMEM; +} + +static void destroy_swapd_thread(void) +{ + int nid; + struct pglist_data *pgdat; + + cpuhp_remove_state_nocalls(swapd_online); + for_each_node(nid) { + pgdat = NODE_DATA(node); + if (!PGDAT_ITEM_DATA(pgdat)) + continue; + + swapd_stop(nid); + kfree((void*)PGDAT_ITEM_DATA(pgdat)); + pgdat->android_oem_data1 = 0; + } +} + +ssize_t hybridswap_swapd_pause_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + char *type_buf = NULL; + bool val; + + type_buf = strstrip((char *)buf); + if (kstrtobool(type_buf, &val)) + return -EINVAL; + atomic_set(&swapd_pause, val); + + return len; +} + +ssize_t hybridswap_swapd_pause_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + ssize_t size = 0; + + size += scnprintf(buf + size, PAGE_SIZE - size, + "%d\n", atomic_read(&swapd_pause)); + + return size; +} + +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) +static int bright_fb_notifier_callback(struct notifier_block *self, + unsigned long event, void *data) +{ + struct msm_drm_notifier *evdata = data; + int *blank; + + if (evdata && evdata->data) { + blank = evdata->data; + + if (*blank == MSM_DRM_BLANK_POWERDOWN) + atomic_set(&display_off, 1); + else if (*blank == MSM_DRM_BLANK_UNBLANK) + atomic_set(&display_off, 0); + } + + return NOTIFY_OK; +} +#endif + +void __init swapd_pre_init(void) +{ + all_totalreserve_pages = fetch_totalreserve_pages(); +} + +void swapd_pre_deinit(void) +{ + all_totalreserve_pages = 0; +} + +int swapd_init(struct zram *zram) +{ + int ret; + + ret = register_memory_notifier(&swapd_notifier_nb); + if (ret) { + hybp(HYB_ERR, "register_memory_notifier failed, ret = %d\n", ret); + return ret; + } + +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) + fb_notif.notifier_call = bright_fb_notifier_callback; + ret = msm_drm_register_client(&fb_notif); + if (ret) { + hybp(HYB_ERR, "msm_drm_register_client failed, ret=%d\n", ret); + goto msm_drm_register_fail; + } +#endif + + ret = refresh_daemonrun(); + if (ret) { + hybp(HYB_ERR, "refresh_daemonrun failed, ret=%d\n", ret); + goto refresh_daemonfail; + } + + ret = create_swapd_thread(zram); + if (ret) { + hybp(HYB_ERR, "create_swapd_thread failed, ret=%d\n", ret); + goto create_swapd_fail; + } + + swapd_zram = zram; + atomic_set(&swapd_enabled, 1); + return 0; + +create_swapd_fail: + refresh_daemonexit(); +refresh_daemonfail: +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) + msm_drm_unregister_client(&fb_notif); +msm_drm_register_fail: +#endif + unregister_memory_notifier(&swapd_notifier_nb); + return ret; +} + +void swapd_exit(void) +{ + destroy_swapd_thread(); + refresh_daemonexit(); +#if IS_ENABLED(CONFIG_DRM_MSM) || IS_ENABLED(CONFIG_DRM_OPLUS_NOTIFY) + msm_drm_unregister_client(&fb_notif); +#endif + unregister_memory_notifier(&swapd_notifier_nb); + atomic_set(&swapd_enabled, 0); +} + +bool hybridswap_swapd_enabled(void) +{ + return !!atomic_read(&swapd_enabled); +} diff --git a/drivers/moto_swap/zram-5.10/zcomp.c b/drivers/moto_swap/zram-5.10/zcomp.c new file mode 100644 index 000000000000..33e3b76c4fa9 --- /dev/null +++ b/drivers/moto_swap/zram-5.10/zcomp.c @@ -0,0 +1,232 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +/* + * Copyright (C) 2014 Sergey Senozhatsky. + */ + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "zcomp.h" + +static const char * const backends[] = { + "lzo", + "lzo-rle", +#if IS_ENABLED(CONFIG_CRYPTO_LZ4) + "lz4", +#endif +#if IS_ENABLED(CONFIG_CRYPTO_LZ4HC) + "lz4hc", +#endif +#if IS_ENABLED(CONFIG_CRYPTO_842) + "842", +#endif +#if IS_ENABLED(CONFIG_CRYPTO_ZSTD) + "zstd", +#endif +}; + +static void zcomp_strm_free(struct zcomp_strm *zstrm) +{ + if (!IS_ERR_OR_NULL(zstrm->tfm)) + crypto_free_comp(zstrm->tfm); + free_pages((unsigned long)zstrm->buffer, 1); + zstrm->tfm = NULL; + zstrm->buffer = NULL; +} + +/* + * Initialize zcomp_strm structure with ->tfm initialized by backend, and + * ->buffer. Return a negative value on error. + */ +static int zcomp_strm_init(struct zcomp_strm *zstrm, struct zcomp *comp) +{ + zstrm->tfm = crypto_alloc_comp(comp->name, 0, 0); + /* + * allocate 2 pages. 1 for compressed data, plus 1 extra for the + * case when compressed size is larger than the original one + */ + zstrm->buffer = (void *)__get_free_pages(GFP_KERNEL | __GFP_ZERO, 1); + if (IS_ERR_OR_NULL(zstrm->tfm) || !zstrm->buffer) { + zcomp_strm_free(zstrm); + return -ENOMEM; + } + return 0; +} + +bool zcomp_available_algorithm(const char *comp) +{ + int i; + + i = sysfs_match_string(backends, comp); + if (i >= 0) + return true; + + /* + * Crypto does not ignore a trailing new line symbol, + * so make sure you don't supply a string containing + * one. + * This also means that we permit zcomp initialisation + * with any compressing algorithm known to crypto api. + */ + return crypto_has_comp(comp, 0, 0) == 1; +} + +/* show available compressors */ +ssize_t zcomp_available_show(const char *comp, char *buf) +{ + bool known_algorithm = false; + ssize_t sz = 0; + int i; + + for (i = 0; i < ARRAY_SIZE(backends); i++) { + if (!strcmp(comp, backends[i])) { + known_algorithm = true; + sz += scnprintf(buf + sz, PAGE_SIZE - sz - 2, + "[%s] ", backends[i]); + } else { + sz += scnprintf(buf + sz, PAGE_SIZE - sz - 2, + "%s ", backends[i]); + } + } + + /* + * Out-of-tree module known to crypto api or a missing + * entry in `backends'. + */ + if (!known_algorithm && crypto_has_comp(comp, 0, 0) == 1) + sz += scnprintf(buf + sz, PAGE_SIZE - sz - 2, + "[%s] ", comp); + + sz += scnprintf(buf + sz, PAGE_SIZE - sz, "\n"); + return sz; +} + +struct zcomp_strm *zcomp_stream_get(struct zcomp *comp) +{ + local_lock(&comp->stream->lock); + return this_cpu_ptr(comp->stream); +} + +void zcomp_stream_put(struct zcomp *comp) +{ + local_unlock(&comp->stream->lock); +} + +int zcomp_compress(struct zcomp_strm *zstrm, + const void *src, unsigned int *dst_len) +{ + /* + * Our dst memory (zstrm->buffer) is always `2 * PAGE_SIZE' sized + * because sometimes we can endup having a bigger compressed data + * due to various reasons: for example compression algorithms tend + * to add some padding to the compressed buffer. Speaking of padding, + * comp algorithm `842' pads the compressed length to multiple of 8 + * and returns -ENOSP when the dst memory is not big enough, which + * is not something that ZRAM wants to see. We can handle the + * `compressed_size > PAGE_SIZE' case easily in ZRAM, but when we + * receive -ERRNO from the compressing backend we can't help it + * anymore. To make `842' happy we need to tell the exact size of + * the dst buffer, zram_drv will take care of the fact that + * compressed buffer is too big. + */ + *dst_len = PAGE_SIZE * 2; + + return crypto_comp_compress(zstrm->tfm, + src, PAGE_SIZE, + zstrm->buffer, dst_len); +} + +int zcomp_decompress(struct zcomp_strm *zstrm, + const void *src, unsigned int src_len, void *dst) +{ + unsigned int dst_len = PAGE_SIZE; + + return crypto_comp_decompress(zstrm->tfm, + src, src_len, + dst, &dst_len); +} + +int zcomp_cpu_up_prepare(unsigned int cpu, struct hlist_node *node) +{ + struct zcomp *comp = hlist_entry(node, struct zcomp, node); + struct zcomp_strm *zstrm; + int ret; + + zstrm = per_cpu_ptr(comp->stream, cpu); + local_lock_init(&zstrm->lock); + + ret = zcomp_strm_init(zstrm, comp); + if (ret) + pr_err("Can't allocate a compression stream\n"); + return ret; +} + +int zcomp_cpu_dead(unsigned int cpu, struct hlist_node *node) +{ + struct zcomp *comp = hlist_entry(node, struct zcomp, node); + struct zcomp_strm *zstrm; + + zstrm = per_cpu_ptr(comp->stream, cpu); + zcomp_strm_free(zstrm); + return 0; +} + +static int zcomp_init(struct zcomp *comp) +{ + int ret; + + comp->stream = alloc_percpu(struct zcomp_strm); + if (!comp->stream) + return -ENOMEM; + + ret = cpuhp_state_add_instance(CPUHP_ZCOMP_PREPARE, &comp->node); + if (ret < 0) + goto cleanup; + return 0; + +cleanup: + free_percpu(comp->stream); + return ret; +} + +void zcomp_destroy(struct zcomp *comp) +{ + cpuhp_state_remove_instance(CPUHP_ZCOMP_PREPARE, &comp->node); + free_percpu(comp->stream); + kfree(comp); +} + +/* + * search available compressors for requested algorithm. + * allocate new zcomp and initialize it. return compressing + * backend pointer or ERR_PTR if things went bad. ERR_PTR(-EINVAL) + * if requested algorithm is not supported, ERR_PTR(-ENOMEM) in + * case of allocation error, or any other error potentially + * returned by zcomp_init(). + */ +struct zcomp *zcomp_create(const char *compress) +{ + struct zcomp *comp; + int error; + + if (!zcomp_available_algorithm(compress)) + return ERR_PTR(-EINVAL); + + comp = kzalloc(sizeof(struct zcomp), GFP_KERNEL); + if (!comp) + return ERR_PTR(-ENOMEM); + + comp->name = compress; + error = zcomp_init(comp); + if (error) { + kfree(comp); + return ERR_PTR(error); + } + return comp; +} diff --git a/drivers/moto_swap/zram-5.10/zcomp.h b/drivers/moto_swap/zram-5.10/zcomp.h new file mode 100644 index 000000000000..40f6420f4b2e --- /dev/null +++ b/drivers/moto_swap/zram-5.10/zcomp.h @@ -0,0 +1,43 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * Copyright (C) 2014 Sergey Senozhatsky. + */ + +#ifndef _ZCOMP_H_ +#define _ZCOMP_H_ +#include + +struct zcomp_strm { + /* The members ->buffer and ->tfm are protected by ->lock. */ + local_lock_t lock; + /* compression/decompression buffer */ + void *buffer; + struct crypto_comp *tfm; +}; + +/* dynamic per-device compression frontend */ +struct zcomp { + struct zcomp_strm __percpu *stream; + const char *name; + struct hlist_node node; +}; + +int zcomp_cpu_up_prepare(unsigned int cpu, struct hlist_node *node); +int zcomp_cpu_dead(unsigned int cpu, struct hlist_node *node); +ssize_t zcomp_available_show(const char *comp, char *buf); +bool zcomp_available_algorithm(const char *comp); + +struct zcomp *zcomp_create(const char *comp); +void zcomp_destroy(struct zcomp *comp); + +struct zcomp_strm *zcomp_stream_get(struct zcomp *comp); +void zcomp_stream_put(struct zcomp *comp); + +int zcomp_compress(struct zcomp_strm *zstrm, + const void *src, unsigned int *dst_len); + +int zcomp_decompress(struct zcomp_strm *zstrm, + const void *src, unsigned int src_len, void *dst); + +bool zcomp_set_max_streams(struct zcomp *comp, int num_strm); +#endif /* _ZCOMP_H_ */ diff --git a/drivers/moto_swap/zram-5.10/zram_drv.c b/drivers/moto_swap/zram-5.10/zram_drv.c new file mode 100644 index 000000000000..56112f2d2e13 --- /dev/null +++ b/drivers/moto_swap/zram-5.10/zram_drv.c @@ -0,0 +1,2339 @@ +/* + * Compressed RAM block device + * + * Copyright (C) 2008, 2009, 2010 Nitin Gupta + * 2012, 2013 Minchan Kim + * + * This code is released using a dual license strategy: BSD/GPL + * You can choose the licence that better fits your requirements. + * + * Released under the terms of 3-clause BSD License + * Released under the terms of GNU General Public License Version 2.0 + * + */ + +#define KMSG_COMPONENT "zram" +#define pr_fmt(fmt) KMSG_COMPONENT ": " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "zram_drv.h" +#include "zram_drv_internal.h" +#ifdef CONFIG_HYBRIDSWAP +#include "../hybridswap/hybridswap.h" +#endif + +static DEFINE_IDR(zram_index_idr); +/* idr index must be protected */ +static DEFINE_MUTEX(zram_index_mutex); + +static int zram_major; +static const char *default_compressor = "lzo-rle"; + +/* Module params (documentation at end) */ +static unsigned int num_devices = 1; +/* + * Pages that compress to sizes equals or greater than this are stored + * uncompressed in memory. + */ +static size_t huge_class_size; + +static const struct block_device_operations zram_devops; +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static const struct block_device_operations zram_wb_devops; +#endif + +static void zram_free_page(struct zram *zram, size_t index); +static int zram_bvec_read(struct zram *zram, struct bio_vec *bvec, + u32 index, int offset, struct bio *bio); + + +static int zram_slot_trylock(struct zram *zram, u32 index) +{ + return bit_spin_trylock(ZRAM_LOCK, &zram->table[index].flags); +} + +static unsigned long zram_get_element(struct zram *zram, u32 index) +{ + return zram->table[index].element; +} + +static inline bool zram_allocated(struct zram *zram, u32 index) +{ + return zram_get_obj_size(zram, index) || + zram_test_flag(zram, index, ZRAM_SAME) || + zram_test_flag(zram, index, ZRAM_WB); +} + +#if PAGE_SIZE != 4096 +static inline bool is_partial_io(struct bio_vec *bvec) +{ + return bvec->bv_len != PAGE_SIZE; +} +#else +static inline bool is_partial_io(struct bio_vec *bvec) +{ + return false; +} +#endif + +/* + * Check if request is within bounds and aligned on zram logical blocks. + */ +static inline bool valid_io_request(struct zram *zram, + sector_t start, unsigned int size) +{ + u64 end, bound; + + /* unaligned request */ + if (unlikely(start & (ZRAM_SECTOR_PER_LOGICAL_BLOCK - 1))) + return false; + if (unlikely(size & (ZRAM_LOGICAL_BLOCK_SIZE - 1))) + return false; + + end = start + (size >> SECTOR_SHIFT); + bound = zram->disksize >> SECTOR_SHIFT; + /* out of range range */ + if (unlikely(start >= bound || end > bound || start > end)) + return false; + + /* I/O request is valid */ + return true; +} + +static void update_position(u32 *index, int *offset, struct bio_vec *bvec) +{ + *index += (*offset + bvec->bv_len) / PAGE_SIZE; + *offset = (*offset + bvec->bv_len) % PAGE_SIZE; +} + +static inline void update_used_max(struct zram *zram, + const unsigned long pages) +{ + unsigned long old_max, cur_max; + + old_max = atomic_long_read(&zram->stats.max_used_pages); + + do { + cur_max = old_max; + if (pages > cur_max) + old_max = atomic_long_cmpxchg( + &zram->stats.max_used_pages, cur_max, pages); + } while (old_max != cur_max); +} + +static inline void zram_fill_page(void *ptr, unsigned long len, + unsigned long value) +{ + WARN_ON_ONCE(!IS_ALIGNED(len, sizeof(unsigned long))); + memset_l(ptr, value, len / sizeof(unsigned long)); +} + +static bool page_same_filled(void *ptr, unsigned long *element) +{ + unsigned long *page; + unsigned long val; + unsigned int pos, last_pos = PAGE_SIZE / sizeof(*page) - 1; + + page = (unsigned long *)ptr; + val = page[0]; + + if (val != page[last_pos]) + return false; + + for (pos = 1; pos < last_pos; pos++) { + if (val != page[pos]) + return false; + } + + *element = val; + + return true; +} + +static ssize_t initstate_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + u32 val; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + val = init_done(zram); + up_read(&zram->init_lock); + + return scnprintf(buf, PAGE_SIZE, "%u\n", val); +} + +static ssize_t disksize_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + + return scnprintf(buf, PAGE_SIZE, "%llu\n", zram->disksize); +} + +static ssize_t mem_limit_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + u64 limit; + char *tmp; + struct zram *zram = dev_to_zram(dev); + + limit = memparse(buf, &tmp); + if (buf == tmp) /* no chars parsed, invalid input */ + return -EINVAL; + + down_write(&zram->init_lock); + zram->limit_pages = PAGE_ALIGN(limit) >> PAGE_SHIFT; + up_write(&zram->init_lock); + + return len; +} + +static ssize_t mem_used_max_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int err; + unsigned long val; + struct zram *zram = dev_to_zram(dev); + + err = kstrtoul(buf, 10, &val); + if (err || val != 0) + return -EINVAL; + + down_read(&zram->init_lock); + if (init_done(zram)) { + atomic_long_set(&zram->stats.max_used_pages, + zs_get_total_pages(zram->mem_pool)); + } + up_read(&zram->init_lock); + + return len; +} + +static ssize_t idle_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; + int index; + + if (!sysfs_streq(buf, "all")) + return -EINVAL; + + down_read(&zram->init_lock); + if (!init_done(zram)) { + up_read(&zram->init_lock); + return -EINVAL; + } + + for (index = 0; index < nr_pages; index++) { + /* + * Do not mark ZRAM_UNDER_WB slot as ZRAM_IDLE to close race. + * See the comment in writeback_store. + */ + zram_slot_lock(zram, index); + if (zram_allocated(zram, index) && + !zram_test_flag(zram, index, ZRAM_UNDER_WB)) + zram_set_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + } + + up_read(&zram->init_lock); + + return len; +} + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static ssize_t writeback_limit_enable_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + u64 val; + ssize_t ret = -EINVAL; + + if (kstrtoull(buf, 10, &val)) + return ret; + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + zram->wb_limit_enable = val; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + ret = len; + + return ret; +} + +static ssize_t writeback_limit_enable_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + bool val; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + val = zram->wb_limit_enable; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + + return scnprintf(buf, PAGE_SIZE, "%d\n", val); +} + +static ssize_t writeback_limit_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + u64 val; + ssize_t ret = -EINVAL; + + if (kstrtoull(buf, 10, &val)) + return ret; + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + zram->bd_wb_limit = val; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + ret = len; + + return ret; +} + +static ssize_t writeback_limit_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + u64 val; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + val = zram->bd_wb_limit; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + + return scnprintf(buf, PAGE_SIZE, "%llu\n", val); +} + +static void reset_bdev(struct zram *zram) +{ + struct block_device *bdev; + + if (!zram->backing_dev) + return; + + bdev = zram->bdev; + if (zram->old_block_size) + set_blocksize(bdev, zram->old_block_size); + blkdev_put(bdev, FMODE_READ|FMODE_WRITE|FMODE_EXCL); + /* hope filp_close flush all of IO */ + filp_close(zram->backing_dev, NULL); + zram->backing_dev = NULL; + zram->old_block_size = 0; + zram->bdev = NULL; + zram->disk->fops = &zram_devops; + kvfree(zram->bitmap); + zram->bitmap = NULL; +} + +static ssize_t backing_dev_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct file *file; + struct zram *zram = dev_to_zram(dev); + char *p; + ssize_t ret; + + down_read(&zram->init_lock); + file = zram->backing_dev; + if (!file) { + memcpy(buf, "none\n", 5); + up_read(&zram->init_lock); + return 5; + } + + p = file_path(file, buf, PAGE_SIZE - 1); + if (IS_ERR(p)) { + ret = PTR_ERR(p); + goto out; + } + + ret = strlen(p); + memmove(buf, p, ret); + buf[ret++] = '\n'; +out: + up_read(&zram->init_lock); + return ret; +} + +static ssize_t backing_dev_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + char *file_name; + size_t sz; + struct file *backing_dev = NULL; + struct inode *inode; + struct address_space *mapping; + unsigned int bitmap_sz, old_block_size = 0; + unsigned long nr_pages, *bitmap = NULL; + struct block_device *bdev = NULL; + int err; + struct zram *zram = dev_to_zram(dev); + + file_name = kmalloc(PATH_MAX, GFP_KERNEL); + if (!file_name) + return -ENOMEM; + + down_write(&zram->init_lock); + if (init_done(zram)) { + pr_info("Can't setup backing device for initialized device\n"); + err = -EBUSY; + goto out; + } + + strlcpy(file_name, buf, PATH_MAX); + /* ignore trailing newline */ + sz = strlen(file_name); + if (sz > 0 && file_name[sz - 1] == '\n') + file_name[sz - 1] = 0x00; + + backing_dev = filp_open_block(file_name, O_RDWR|O_LARGEFILE, 0); + if (IS_ERR(backing_dev)) { + err = PTR_ERR(backing_dev); + backing_dev = NULL; + goto out; + } + + mapping = backing_dev->f_mapping; + inode = mapping->host; + + /* Support only block device in this moment */ + if (!S_ISBLK(inode->i_mode)) { + err = -ENOTBLK; + goto out; + } + + bdev = blkdev_get_by_dev(inode->i_rdev, + FMODE_READ | FMODE_WRITE | FMODE_EXCL, zram); + if (IS_ERR(bdev)) { + err = PTR_ERR(bdev); + bdev = NULL; + goto out; + } + + nr_pages = i_size_read(inode) >> PAGE_SHIFT; + bitmap_sz = BITS_TO_LONGS(nr_pages) * sizeof(long); + bitmap = kvzalloc(bitmap_sz, GFP_KERNEL); + if (!bitmap) { + err = -ENOMEM; + goto out; + } + + old_block_size = block_size(bdev); + err = set_blocksize(bdev, PAGE_SIZE); + if (err) + goto out; + + reset_bdev(zram); + + zram->old_block_size = old_block_size; + zram->bdev = bdev; + zram->backing_dev = backing_dev; + zram->bitmap = bitmap; + zram->nr_pages = nr_pages; + /* + * With writeback feature, zram does asynchronous IO so it's no longer + * synchronous device so let's remove synchronous io flag. Othewise, + * upper layer(e.g., swap) could wait IO completion rather than + * (submit and return), which will cause system sluggish. + * Furthermore, when the IO function returns(e.g., swap_readpage), + * upper layer expects IO was done so it could deallocate the page + * freely but in fact, IO is going on so finally could cause + * use-after-free when the IO is really done. + */ + zram->disk->fops = &zram_wb_devops; + up_write(&zram->init_lock); + + pr_info("setup backing device %s\n", file_name); + kfree(file_name); + + return len; +out: + if (bitmap) + kvfree(bitmap); + + if (bdev) + blkdev_put(bdev, FMODE_READ | FMODE_WRITE | FMODE_EXCL); + + if (backing_dev) + filp_close(backing_dev, NULL); + + up_write(&zram->init_lock); + + kfree(file_name); + + return err; +} + +static unsigned long alloc_block_bdev(struct zram *zram) +{ + unsigned long blk_index = 1; +retry: + /* skip 0 bit to confuse zram.handle = 0 */ + blk_index = find_next_zero_bit(zram->bitmap, zram->nr_pages, blk_index); + if (blk_index == zram->nr_pages) + return 0; + + if (test_and_set_bit(blk_index, zram->bitmap)) + goto retry; + + atomic64_inc(&zram->stats.bd_count); + return blk_index; +} + +static void free_block_bdev(struct zram *zram, unsigned long blk_index) +{ + int was_set; + + was_set = test_and_clear_bit(blk_index, zram->bitmap); + WARN_ON_ONCE(!was_set); + atomic64_dec(&zram->stats.bd_count); +} + +static void zram_page_end_io(struct bio *bio) +{ + struct page *page = bio_first_page_all(bio); + + page_endio(page, op_is_write(bio_op(bio)), + blk_status_to_errno(bio->bi_status)); + bio_put(bio); +} + +/* + * Returns 1 if the submission is successful. + */ +static int read_from_bdev_async(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *parent) +{ + struct bio *bio; + + bio = bio_alloc(GFP_ATOMIC, 1); + if (!bio) + return -ENOMEM; + + bio->bi_iter.bi_sector = entry * (PAGE_SIZE >> 9); + bio_set_dev(bio, zram->bdev); + if (!bio_add_page(bio, bvec->bv_page, bvec->bv_len, bvec->bv_offset)) { + bio_put(bio); + return -EIO; + } + + if (!parent) { + bio->bi_opf = REQ_OP_READ; + bio->bi_end_io = zram_page_end_io; + } else { + bio->bi_opf = parent->bi_opf; + bio_chain(bio, parent); + } + + submit_bio(bio); + return 1; +} + +#define PAGE_WB_SIG "page_index=" + +#define PAGE_WRITEBACK 0 +#define HUGE_WRITEBACK 1 +#define IDLE_WRITEBACK 2 + + +static ssize_t writeback_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; + unsigned long index = 0; + struct bio bio; + struct bio_vec bio_vec; + struct page *page; + ssize_t ret = len; + int mode, err; + unsigned long blk_index = 0; + + if (sysfs_streq(buf, "idle")) + mode = IDLE_WRITEBACK; + else if (sysfs_streq(buf, "huge")) + mode = HUGE_WRITEBACK; + else { + if (strncmp(buf, PAGE_WB_SIG, sizeof(PAGE_WB_SIG) - 1)) + return -EINVAL; + + if (kstrtol(buf + sizeof(PAGE_WB_SIG) - 1, 10, &index) || + index >= nr_pages) + return -EINVAL; + + nr_pages = 1; + mode = PAGE_WRITEBACK; + } + + down_read(&zram->init_lock); + if (!init_done(zram)) { + ret = -EINVAL; + goto release_init_lock; + } + + if (!zram->backing_dev) { + ret = -ENODEV; + goto release_init_lock; + } + + page = alloc_page(GFP_KERNEL); + if (!page) { + ret = -ENOMEM; + goto release_init_lock; + } + + for (; nr_pages != 0; index++, nr_pages--) { + struct bio_vec bvec; + + bvec.bv_page = page; + bvec.bv_len = PAGE_SIZE; + bvec.bv_offset = 0; + + spin_lock(&zram->wb_limit_lock); + if (zram->wb_limit_enable && !zram->bd_wb_limit) { + spin_unlock(&zram->wb_limit_lock); + ret = -EIO; + break; + } + spin_unlock(&zram->wb_limit_lock); + + if (!blk_index) { + blk_index = alloc_block_bdev(zram); + if (!blk_index) { + ret = -ENOSPC; + break; + } + } + + zram_slot_lock(zram, index); + if (!zram_allocated(zram, index)) + goto next; + + if (zram_test_flag(zram, index, ZRAM_WB) || + zram_test_flag(zram, index, ZRAM_SAME) || + zram_test_flag(zram, index, ZRAM_UNDER_WB)) + goto next; + + if (mode == IDLE_WRITEBACK && + !zram_test_flag(zram, index, ZRAM_IDLE)) + goto next; + if (mode == HUGE_WRITEBACK && + !zram_test_flag(zram, index, ZRAM_HUGE)) + goto next; + /* + * Clearing ZRAM_UNDER_WB is duty of caller. + * IOW, zram_free_page never clear it. + */ + zram_set_flag(zram, index, ZRAM_UNDER_WB); + /* Need for hugepage writeback racing */ + zram_set_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + if (zram_bvec_read(zram, &bvec, index, 0, NULL)) { + zram_slot_lock(zram, index); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_clear_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + continue; + } + + bio_init(&bio, &bio_vec, 1); + bio_set_dev(&bio, zram->bdev); + bio.bi_iter.bi_sector = blk_index * (PAGE_SIZE >> 9); + bio.bi_opf = REQ_OP_WRITE | REQ_SYNC; + + bio_add_page(&bio, bvec.bv_page, bvec.bv_len, + bvec.bv_offset); + /* + * XXX: A single page IO would be inefficient for write + * but it would be not bad as starter. + */ + err = submit_bio_wait(&bio); + if (err) { + zram_slot_lock(zram, index); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_clear_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + /* + * Return last IO error unless every IO were + * not suceeded. + */ + ret = err; + continue; + } + + atomic64_inc(&zram->stats.bd_writes); + /* + * We released zram_slot_lock so need to check if the slot was + * changed. If there is freeing for the slot, we can catch it + * easily by zram_allocated. + * A subtle case is the slot is freed/reallocated/marked as + * ZRAM_IDLE again. To close the race, idle_store doesn't + * mark ZRAM_IDLE once it found the slot was ZRAM_UNDER_WB. + * Thus, we could close the race by checking ZRAM_IDLE bit. + */ + zram_slot_lock(zram, index); + if (!zram_allocated(zram, index) || + !zram_test_flag(zram, index, ZRAM_IDLE)) { + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_clear_flag(zram, index, ZRAM_IDLE); + goto next; + } + + zram_free_page(zram, index); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_set_flag(zram, index, ZRAM_WB); + zram_set_element(zram, index, blk_index); + blk_index = 0; + atomic64_inc(&zram->stats.pages_stored); + spin_lock(&zram->wb_limit_lock); + if (zram->wb_limit_enable && zram->bd_wb_limit > 0) + zram->bd_wb_limit -= 1UL << (PAGE_SHIFT - 12); + spin_unlock(&zram->wb_limit_lock); +next: + zram_slot_unlock(zram, index); + } + + if (blk_index) + free_block_bdev(zram, blk_index); + __free_page(page); +release_init_lock: + up_read(&zram->init_lock); + + return ret; +} + +struct zram_work { + struct work_struct work; + struct zram *zram; + unsigned long entry; + struct bio *bio; + struct bio_vec bvec; +}; + +#if PAGE_SIZE != 4096 +static void zram_sync_read(struct work_struct *work) +{ + struct zram_work *zw = container_of(work, struct zram_work, work); + struct zram *zram = zw->zram; + unsigned long entry = zw->entry; + struct bio *bio = zw->bio; + + read_from_bdev_async(zram, &zw->bvec, entry, bio); +} + +/* + * Block layer want one ->submit_bio to be active at a time, so if we use + * chained IO with parent IO in same context, it's a deadlock. To avoid that, + * use a worker thread context. + */ +static int read_from_bdev_sync(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *bio) +{ + struct zram_work work; + + work.bvec = *bvec; + work.zram = zram; + work.entry = entry; + work.bio = bio; + + INIT_WORK_ONSTACK(&work.work, zram_sync_read); + queue_work(system_unbound_wq, &work.work); + flush_work(&work.work); + destroy_work_on_stack(&work.work); + + return 1; +} +#else +static int read_from_bdev_sync(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *bio) +{ + WARN_ON(1); + return -EIO; +} +#endif + +static int read_from_bdev(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *parent, bool sync) +{ + atomic64_inc(&zram->stats.bd_reads); + if (sync) + return read_from_bdev_sync(zram, bvec, entry, parent); + else + return read_from_bdev_async(zram, bvec, entry, parent); +} +#else +static inline void reset_bdev(struct zram *zram) {}; +static int read_from_bdev(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *parent, bool sync) +{ + return -EIO; +} + +static void free_block_bdev(struct zram *zram, unsigned long blk_index) {}; +#endif + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + +static struct dentry *zram_debugfs_root; + +static void zram_debugfs_create(void) +{ + zram_debugfs_root = debugfs_create_dir("zram", NULL); +} + +static void zram_debugfs_destroy(void) +{ + debugfs_remove_recursive(zram_debugfs_root); +} + +static void zram_accessed(struct zram *zram, u32 index) +{ + zram_clear_flag(zram, index, ZRAM_IDLE); + zram->table[index].ac_time = ktime_get_boottime(); +} + +static ssize_t read_block_state(struct file *file, char __user *buf, + size_t count, loff_t *ppos) +{ + char *kbuf; + ssize_t index, written = 0; + struct zram *zram = file->private_data; + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; + struct timespec64 ts; + + kbuf = kvmalloc(count, GFP_KERNEL); + if (!kbuf) + return -ENOMEM; + + down_read(&zram->init_lock); + if (!init_done(zram)) { + up_read(&zram->init_lock); + kvfree(kbuf); + return -EINVAL; + } + + for (index = *ppos; index < nr_pages; index++) { + int copied; + + zram_slot_lock(zram, index); + if (!zram_allocated(zram, index)) + goto next; + + ts = ktime_to_timespec64(zram->table[index].ac_time); + copied = snprintf(kbuf + written, count, + "%12zd %12lld.%06lu %c%c%c%c\n", + index, (s64)ts.tv_sec, + ts.tv_nsec / NSEC_PER_USEC, + zram_test_flag(zram, index, ZRAM_SAME) ? 's' : '.', + zram_test_flag(zram, index, ZRAM_WB) ? 'w' : '.', + zram_test_flag(zram, index, ZRAM_HUGE) ? 'h' : '.', + zram_test_flag(zram, index, ZRAM_IDLE) ? 'i' : '.'); + + if (count < copied) { + zram_slot_unlock(zram, index); + break; + } + written += copied; + count -= copied; +next: + zram_slot_unlock(zram, index); + *ppos += 1; + } + + up_read(&zram->init_lock); + if (copy_to_user(buf, kbuf, written)) + written = -EFAULT; + kvfree(kbuf); + + return written; +} + +static const struct file_operations proc_zram_block_state_op = { + .open = simple_open, + .read = read_block_state, + .llseek = default_llseek, +}; + +static void zram_debugfs_register(struct zram *zram) +{ + if (!zram_debugfs_root) + return; + + zram->debugfs_dir = debugfs_create_dir(zram->disk->disk_name, + zram_debugfs_root); + debugfs_create_file("block_state", 0400, zram->debugfs_dir, + zram, &proc_zram_block_state_op); +} + +static void zram_debugfs_unregister(struct zram *zram) +{ + debugfs_remove_recursive(zram->debugfs_dir); +} +#else +static void zram_debugfs_create(void) {}; +static void zram_debugfs_destroy(void) {}; +static void zram_accessed(struct zram *zram, u32 index) +{ + zram_clear_flag(zram, index, ZRAM_IDLE); +}; +static void zram_debugfs_register(struct zram *zram) {}; +static void zram_debugfs_unregister(struct zram *zram) {}; +#endif + +/* + * We switched to per-cpu streams and this attr is not needed anymore. + * However, we will keep it around for some time, because: + * a) we may revert per-cpu streams in the future + * b) it's visible to user space and we need to follow our 2 years + * retirement rule; but we already have a number of 'soon to be + * altered' attrs, so max_comp_streams need to wait for the next + * layoff cycle. + */ +static ssize_t max_comp_streams_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return scnprintf(buf, PAGE_SIZE, "%d\n", num_online_cpus()); +} + +static ssize_t max_comp_streams_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + return len; +} + +static ssize_t comp_algorithm_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + size_t sz; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + sz = zcomp_available_show(zram->compressor, buf); + up_read(&zram->init_lock); + + return sz; +} + +static ssize_t comp_algorithm_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + char compressor[ARRAY_SIZE(zram->compressor)]; + size_t sz; + + strlcpy(compressor, buf, sizeof(compressor)); + /* ignore trailing newline */ + sz = strlen(compressor); + if (sz > 0 && compressor[sz - 1] == '\n') + compressor[sz - 1] = 0x00; + + if (!zcomp_available_algorithm(compressor)) + return -EINVAL; + + down_write(&zram->init_lock); + if (init_done(zram)) { + up_write(&zram->init_lock); + pr_info("Can't change algorithm for initialized device\n"); + return -EBUSY; + } + + strcpy(zram->compressor, compressor); + up_write(&zram->init_lock); + return len; +} + +static ssize_t compact_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + if (!init_done(zram)) { + up_read(&zram->init_lock); + return -EINVAL; + } + + zs_compact(zram->mem_pool); + up_read(&zram->init_lock); + + return len; +} + +static ssize_t io_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + ssize_t ret; + + down_read(&zram->init_lock); + ret = scnprintf(buf, PAGE_SIZE, + "%8llu %8llu %8llu %8llu\n", + (u64)atomic64_read(&zram->stats.failed_reads), + (u64)atomic64_read(&zram->stats.failed_writes), + (u64)atomic64_read(&zram->stats.invalid_io), + (u64)atomic64_read(&zram->stats.notify_free)); + up_read(&zram->init_lock); + + return ret; +} + +static ssize_t mm_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + struct zs_pool_stats pool_stats; + u64 orig_size, mem_used = 0; + long max_used; + ssize_t ret; + + memset(&pool_stats, 0x00, sizeof(struct zs_pool_stats)); + + down_read(&zram->init_lock); + if (init_done(zram)) { + mem_used = zs_get_total_pages(zram->mem_pool); + zs_pool_stats(zram->mem_pool, &pool_stats); + } + + orig_size = atomic64_read(&zram->stats.pages_stored); + max_used = atomic_long_read(&zram->stats.max_used_pages); + + ret = scnprintf(buf, PAGE_SIZE, + "%8llu %8llu %8llu %8lu %8ld %8llu %8lu %8llu\n", + orig_size << PAGE_SHIFT, + (u64)atomic64_read(&zram->stats.compr_data_size), + mem_used << PAGE_SHIFT, + zram->limit_pages << PAGE_SHIFT, + max_used << PAGE_SHIFT, + (u64)atomic64_read(&zram->stats.same_pages), + atomic_long_read(&pool_stats.pages_compacted), + (u64)atomic64_read(&zram->stats.huge_pages)); + up_read(&zram->init_lock); + + return ret; +} + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +#define FOUR_K(x) ((x) * (1 << (PAGE_SHIFT - 12))) +static ssize_t bd_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + ssize_t ret; + + down_read(&zram->init_lock); + ret = scnprintf(buf, PAGE_SIZE, + "%8llu %8llu %8llu\n", + FOUR_K((u64)atomic64_read(&zram->stats.bd_count)), + FOUR_K((u64)atomic64_read(&zram->stats.bd_reads)), + FOUR_K((u64)atomic64_read(&zram->stats.bd_writes))); + up_read(&zram->init_lock); + + return ret; +} +#endif + +static ssize_t debug_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int version = 1; + struct zram *zram = dev_to_zram(dev); + ssize_t ret; + + down_read(&zram->init_lock); + ret = scnprintf(buf, PAGE_SIZE, + "version: %d\n%8llu %8llu\n", + version, + (u64)atomic64_read(&zram->stats.writestall), + (u64)atomic64_read(&zram->stats.miss_free)); + up_read(&zram->init_lock); + + return ret; +} + +static DEVICE_ATTR_RO(io_stat); +static DEVICE_ATTR_RO(mm_stat); +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static DEVICE_ATTR_RO(bd_stat); +#endif +static DEVICE_ATTR_RO(debug_stat); + +static void zram_meta_free(struct zram *zram, u64 disksize) +{ + size_t num_pages = disksize >> PAGE_SHIFT; + size_t index; + + /* Free all pages that are still in this zram device */ + for (index = 0; index < num_pages; index++) + zram_free_page(zram, index); + + zs_destroy_pool(zram->mem_pool); + vfree(zram->table); +} + +static bool zram_meta_alloc(struct zram *zram, u64 disksize) +{ + size_t num_pages; + + num_pages = disksize >> PAGE_SHIFT; + zram->table = vzalloc(array_size(num_pages, sizeof(*zram->table))); + if (!zram->table) + return false; + + zram->mem_pool = zs_create_pool(zram->disk->disk_name); + if (!zram->mem_pool) { + vfree(zram->table); + return false; + } + + if (!huge_class_size) + huge_class_size = zs_huge_class_size(zram->mem_pool); + return true; +} + +/* + * To protect concurrent access to the same index entry, + * caller should hold this table index entry's bit_spinlock to + * indicate this index entry is accessing. + */ +static void zram_free_page(struct zram *zram, size_t index) +{ + unsigned long handle; + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + zram->table[index].ac_time = 0; +#endif + if (zram_test_flag(zram, index, ZRAM_IDLE)) + zram_clear_flag(zram, index, ZRAM_IDLE); + + if (zram_test_flag(zram, index, ZRAM_HUGE)) { + zram_clear_flag(zram, index, ZRAM_HUGE); + atomic64_dec(&zram->stats.huge_pages); + } + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (zram_test_flag(zram, index, ZRAM_CACHED)) { + struct page *page = (struct page *)zram_get_page(zram, index); + + del_page_from_cache(page); + page->mem_cgroup = NULL; + put_free_page(page); + zram_clear_flag(zram, index, ZRAM_CACHED); + goto out; + } + + if (zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + zram_clear_flag(zram, index, ZRAM_CACHED_COMPRESS); + goto out; + } +#endif + +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_untrack(zram, index); +#endif + + if (zram_test_flag(zram, index, ZRAM_WB)) { + zram_clear_flag(zram, index, ZRAM_WB); + free_block_bdev(zram, zram_get_element(zram, index)); + atomic64_dec(&zram->stats.pages_stored); + goto out; + } + + /* + * No memory is allocated for same element filled pages. + * Simply clear same page flag. + */ + if (zram_test_flag(zram, index, ZRAM_SAME)) { + zram_clear_flag(zram, index, ZRAM_SAME); + atomic64_dec(&zram->stats.same_pages); + atomic64_dec(&zram->stats.pages_stored); + goto out; + } + + handle = zram_get_handle(zram, index); + if (!handle) + return; + + zs_free(zram->mem_pool, handle); + + atomic64_sub(zram_get_obj_size(zram, index), + &zram->stats.compr_data_size); + atomic64_dec(&zram->stats.pages_stored); + +out: + zram_set_handle(zram, index, 0); + zram_set_obj_size(zram, index, 0); + WARN_ON_ONCE(zram->table[index].flags & + ~(1UL << ZRAM_LOCK | 1UL << ZRAM_UNDER_WB)); +} + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +void update_zram_index(struct zram *zram, u32 index, unsigned long page) +{ + zram_slot_lock(zram, index); + put_anon_pages((struct page*)page); + + zram_free_page(zram, index); + zram_set_flag(zram, index, ZRAM_CACHED); + zram_set_page(zram, index, page); + zram_set_obj_size(zram, index, PAGE_SIZE); + zram_slot_unlock(zram, index); +} + +int async_compress_page(struct zram *zram, struct page* page) +{ + int ret = 0; + unsigned long alloced_pages; + unsigned long handle = 0; + unsigned int comp_len = 0; + void *src, *dst; + struct zcomp_strm *zstrm; + int index = get_zram_index(page); + +compress_again: + zram_slot_lock(zram, index); + if (!zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + zram_slot_unlock(zram, index); + return 0; + } + zram_slot_unlock(zram, index); + + zstrm = zcomp_stream_get(zram->comp); + src = kmap_atomic(page); + ret = zcomp_compress(zstrm, src, &comp_len); + kunmap_atomic(src); + + if (unlikely(ret)) { + zcomp_stream_put(zram->comp); + pr_err("Compression failed! err=%d\n", ret); + zs_free(zram->mem_pool, handle); + return ret; + } + + if (comp_len >= huge_class_size) + comp_len = PAGE_SIZE; + + if (!handle) + handle = zs_malloc(zram->mem_pool, comp_len, + __GFP_KSWAPD_RECLAIM | + __GFP_NOWARN | + __GFP_HIGHMEM | + __GFP_MOVABLE | + __GFP_CMA); + if (!handle) { + zcomp_stream_put(zram->comp); + atomic64_inc(&zram->stats.writestall); + handle = zs_malloc(zram->mem_pool, comp_len, + GFP_NOIO | __GFP_HIGHMEM | + __GFP_MOVABLE | __GFP_CMA); + if (handle) + goto compress_again; + return -ENOMEM; + } + + alloced_pages = zs_get_total_pages(zram->mem_pool); + update_used_max(zram, alloced_pages); + + if (zram->limit_pages && alloced_pages > zram->limit_pages) { + zcomp_stream_put(zram->comp); + zs_free(zram->mem_pool, handle); + return -ENOMEM; + } + + dst = zs_map_object(zram->mem_pool, handle, ZS_MM_WO); + + src = zstrm->buffer; + if (comp_len == PAGE_SIZE) + src = kmap_atomic(page); + memcpy(dst, src, comp_len); + if (comp_len == PAGE_SIZE) + kunmap_atomic(src); + + zcomp_stream_put(zram->comp); + zs_unmap_object(zram->mem_pool, handle); + atomic64_add(comp_len, &zram->stats.compr_data_size); + + /* + * Free memory associated with this sector + * before overwriting unused sectors. + */ + zram_slot_lock(zram, index); + if (!zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + atomic64_sub(comp_len, &zram->stats.compr_data_size); + zs_free(zram->mem_pool, handle); + zram_slot_unlock(zram, index); + return 0; + } + zram_free_page(zram, index); + + if (comp_len == PAGE_SIZE) { + zram_set_flag(zram, index, ZRAM_HUGE); + atomic64_inc(&zram->stats.huge_pages); + } + + zram_set_handle(zram, index, handle); + zram_set_obj_size(zram, index, comp_len); +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_record(zram, index, page->mem_cgroup); +#endif + zram_slot_unlock(zram, index); + + /* Update stats */ + atomic64_inc(&zram->stats.pages_stored); + + return ret; +} +#endif + +static int __zram_bvec_read(struct zram *zram, struct page *page, u32 index, + struct bio *bio, bool partial_io) +{ + struct zcomp_strm *zstrm; + unsigned long handle; + unsigned int size; + void *src, *dst; + int ret; + + zram_slot_lock(zram, index); + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (akcompress_cache_page_fault(zram, page, index)) + return 0; +#endif + +#ifdef CONFIG_HYBRIDSWAP_CORE + if (likely(!bio)) { + ret = hybridswap_page_fault(zram, index); + if (unlikely(ret)) { + pr_err("search in hybridswap failed! err=%d, page=%u\n", + ret, index); + zram_slot_unlock(zram, index); + return ret; + } + } +#endif + + if (zram_test_flag(zram, index, ZRAM_WB)) { + struct bio_vec bvec; + + zram_slot_unlock(zram, index); + + bvec.bv_page = page; + bvec.bv_len = PAGE_SIZE; + bvec.bv_offset = 0; + return read_from_bdev(zram, &bvec, + zram_get_element(zram, index), + bio, partial_io); + } + + handle = zram_get_handle(zram, index); + if (!handle || zram_test_flag(zram, index, ZRAM_SAME)) { + unsigned long value; + void *mem; + + value = handle ? zram_get_element(zram, index) : 0; + mem = kmap_atomic(page); + zram_fill_page(mem, PAGE_SIZE, value); + kunmap_atomic(mem); + zram_slot_unlock(zram, index); + return 0; + } + + size = zram_get_obj_size(zram, index); + + if (size != PAGE_SIZE) + zstrm = zcomp_stream_get(zram->comp); + + src = zs_map_object(zram->mem_pool, handle, ZS_MM_RO); + if (size == PAGE_SIZE) { + dst = kmap_atomic(page); + memcpy(dst, src, PAGE_SIZE); + kunmap_atomic(dst); + ret = 0; + } else { + dst = kmap_atomic(page); + ret = zcomp_decompress(zstrm, src, size, dst); + kunmap_atomic(dst); + zcomp_stream_put(zram->comp); + } + zs_unmap_object(zram->mem_pool, handle); + zram_slot_unlock(zram, index); + + /* Should NEVER happen. Return bio error if it does. */ + if (WARN_ON(ret)) + pr_err("Decompression failed! err=%d, page=%u\n", ret, index); + + return ret; +} + +static int zram_bvec_read(struct zram *zram, struct bio_vec *bvec, + u32 index, int offset, struct bio *bio) +{ + int ret; + struct page *page; + + page = bvec->bv_page; + if (is_partial_io(bvec)) { + /* Use a temporary buffer to decompress the page */ + page = alloc_page(GFP_NOIO|__GFP_HIGHMEM); + if (!page) + return -ENOMEM; + } + + ret = __zram_bvec_read(zram, page, index, bio, is_partial_io(bvec)); + if (unlikely(ret)) + goto out; + + if (is_partial_io(bvec)) { + void *dst = kmap_atomic(bvec->bv_page); + void *src = kmap_atomic(page); + + memcpy(dst + bvec->bv_offset, src + offset, bvec->bv_len); + kunmap_atomic(src); + kunmap_atomic(dst); + } +out: + if (is_partial_io(bvec)) + __free_page(page); + + return ret; +} + +static int __zram_bvec_write(struct zram *zram, struct bio_vec *bvec, + u32 index, struct bio *bio) +{ + int ret = 0; + unsigned long alloced_pages; + unsigned long handle = 0; + unsigned int comp_len = 0; + void *src, *dst, *mem; + struct zcomp_strm *zstrm; + struct page *page = bvec->bv_page; + unsigned long element = 0; + enum zram_pageflags flags = 0; + + mem = kmap_atomic(page); + if (page_same_filled(mem, &element)) { + kunmap_atomic(mem); + /* Free memory associated with this sector now. */ + flags = ZRAM_SAME; + atomic64_inc(&zram->stats.same_pages); + goto out; + } + kunmap_atomic(mem); + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if ((current_is_kswapd() || current_is_mswapd(current)) && + add_anon_page2cache(zram, index, page)) { + return 0; + } +#endif + +compress_again: + zstrm = zcomp_stream_get(zram->comp); + src = kmap_atomic(page); + ret = zcomp_compress(zstrm, src, &comp_len); + kunmap_atomic(src); + + if (unlikely(ret)) { + zcomp_stream_put(zram->comp); + pr_err("Compression failed! err=%d\n", ret); + zs_free(zram->mem_pool, handle); + return ret; + } + + if (comp_len >= huge_class_size) + comp_len = PAGE_SIZE; + /* + * handle allocation has 2 paths: + * a) fast path is executed with preemption disabled (for + * per-cpu streams) and has __GFP_DIRECT_RECLAIM bit clear, + * since we can't sleep; + * b) slow path enables preemption and attempts to allocate + * the page with __GFP_DIRECT_RECLAIM bit set. we have to + * put per-cpu compression stream and, thus, to re-do + * the compression once handle is allocated. + * + * if we have a 'non-null' handle here then we are coming + * from the slow path and handle has already been allocated. + */ + if (!handle) + handle = zs_malloc(zram->mem_pool, comp_len, + __GFP_KSWAPD_RECLAIM | + __GFP_NOWARN | + __GFP_HIGHMEM | + __GFP_MOVABLE | + __GFP_CMA); + if (!handle) { + zcomp_stream_put(zram->comp); + atomic64_inc(&zram->stats.writestall); + handle = zs_malloc(zram->mem_pool, comp_len, + GFP_NOIO | __GFP_HIGHMEM | + __GFP_MOVABLE | __GFP_CMA); + if (handle) + goto compress_again; + return -ENOMEM; + } + + alloced_pages = zs_get_total_pages(zram->mem_pool); + update_used_max(zram, alloced_pages); + + if (zram->limit_pages && alloced_pages > zram->limit_pages) { + zcomp_stream_put(zram->comp); + zs_free(zram->mem_pool, handle); + return -ENOMEM; + } + + dst = zs_map_object(zram->mem_pool, handle, ZS_MM_WO); + + src = zstrm->buffer; + if (comp_len == PAGE_SIZE) + src = kmap_atomic(page); + memcpy(dst, src, comp_len); + if (comp_len == PAGE_SIZE) + kunmap_atomic(src); + + zcomp_stream_put(zram->comp); + zs_unmap_object(zram->mem_pool, handle); + atomic64_add(comp_len, &zram->stats.compr_data_size); +out: + /* + * Free memory associated with this sector + * before overwriting unused sectors. + */ + zram_slot_lock(zram, index); + zram_free_page(zram, index); + + if (comp_len == PAGE_SIZE) { + zram_set_flag(zram, index, ZRAM_HUGE); + atomic64_inc(&zram->stats.huge_pages); + } + + if (flags) { + zram_set_flag(zram, index, flags); + zram_set_element(zram, index, element); + } else { + zram_set_handle(zram, index, handle); + zram_set_obj_size(zram, index, comp_len); + } + +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_record(zram, index, page->mem_cgroup); +#endif + zram_slot_unlock(zram, index); + + /* Update stats */ + atomic64_inc(&zram->stats.pages_stored); + return ret; +} + +static int zram_bvec_write(struct zram *zram, struct bio_vec *bvec, + u32 index, int offset, struct bio *bio) +{ + int ret; + struct page *page = NULL; + void *src; + struct bio_vec vec; + + vec = *bvec; + if (is_partial_io(bvec)) { + void *dst; + /* + * This is a partial IO. We need to read the full page + * before to write the changes. + */ + page = alloc_page(GFP_NOIO|__GFP_HIGHMEM); + if (!page) + return -ENOMEM; + + ret = __zram_bvec_read(zram, page, index, bio, true); + if (ret) + goto out; + + src = kmap_atomic(bvec->bv_page); + dst = kmap_atomic(page); + memcpy(dst + offset, src + bvec->bv_offset, bvec->bv_len); + kunmap_atomic(dst); + kunmap_atomic(src); + + vec.bv_page = page; + vec.bv_len = PAGE_SIZE; + vec.bv_offset = 0; + } + + ret = __zram_bvec_write(zram, &vec, index, bio); +out: + if (is_partial_io(bvec)) + __free_page(page); + return ret; +} + +/* + * zram_bio_discard - handler on discard request + * @index: physical block index in PAGE_SIZE units + * @offset: byte offset within physical block + */ +static void zram_bio_discard(struct zram *zram, u32 index, + int offset, struct bio *bio) +{ + size_t n = bio->bi_iter.bi_size; + + /* + * zram manages data in physical block size units. Because logical block + * size isn't identical with physical block size on some arch, we + * could get a discard request pointing to a specific offset within a + * certain physical block. Although we can handle this request by + * reading that physiclal block and decompressing and partially zeroing + * and re-compressing and then re-storing it, this isn't reasonable + * because our intent with a discard request is to save memory. So + * skipping this logical block is appropriate here. + */ + if (offset) { + if (n <= (PAGE_SIZE - offset)) + return; + + n -= (PAGE_SIZE - offset); + index++; + } + + while (n >= PAGE_SIZE) { + zram_slot_lock(zram, index); + zram_free_page(zram, index); + zram_slot_unlock(zram, index); + atomic64_inc(&zram->stats.notify_free); + index++; + n -= PAGE_SIZE; + } +} + +/* + * Returns errno if it has some problem. Otherwise return 0 or 1. + * Returns 0 if IO request was done synchronously + * Returns 1 if IO request was successfully submitted. + */ +static int zram_bvec_rw(struct zram *zram, struct bio_vec *bvec, u32 index, + int offset, unsigned int op, struct bio *bio) +{ + int ret; + + if (!op_is_write(op)) { + atomic64_inc(&zram->stats.num_reads); + ret = zram_bvec_read(zram, bvec, index, offset, bio); + flush_dcache_page(bvec->bv_page); + } else { + atomic64_inc(&zram->stats.num_writes); + ret = zram_bvec_write(zram, bvec, index, offset, bio); + } + + zram_slot_lock(zram, index); + zram_accessed(zram, index); + zram_slot_unlock(zram, index); + + if (unlikely(ret < 0)) { + if (!op_is_write(op)) + atomic64_inc(&zram->stats.failed_reads); + else + atomic64_inc(&zram->stats.failed_writes); + } + + return ret; +} + +static void __zram_make_request(struct zram *zram, struct bio *bio) +{ + int offset; + u32 index; + struct bio_vec bvec; + struct bvec_iter iter; + unsigned long start_time; + + index = bio->bi_iter.bi_sector >> SECTORS_PER_PAGE_SHIFT; + offset = (bio->bi_iter.bi_sector & + (SECTORS_PER_PAGE - 1)) << SECTOR_SHIFT; + + switch (bio_op(bio)) { + case REQ_OP_DISCARD: + case REQ_OP_WRITE_ZEROES: + zram_bio_discard(zram, index, offset, bio); + bio_endio(bio); + return; + default: + break; + } + + start_time = bio_start_io_acct(bio); + bio_for_each_segment(bvec, bio, iter) { + struct bio_vec bv = bvec; + unsigned int unwritten = bvec.bv_len; + + do { + bv.bv_len = min_t(unsigned int, PAGE_SIZE - offset, + unwritten); + if (zram_bvec_rw(zram, &bv, index, offset, + bio_op(bio), bio) < 0) { + bio->bi_status = BLK_STS_IOERR; + break; + } + + bv.bv_offset += bv.bv_len; + unwritten -= bv.bv_len; + + update_position(&index, &offset, &bv); + } while (unwritten); + } + bio_end_io_acct(bio, start_time); + bio_endio(bio); +} + +/* + * Handler function for all zram I/O requests. + */ +static blk_qc_t zram_submit_bio(struct bio *bio) +{ + struct zram *zram = bio->bi_disk->private_data; + + if (!valid_io_request(zram, bio->bi_iter.bi_sector, + bio->bi_iter.bi_size)) { + atomic64_inc(&zram->stats.invalid_io); + goto error; + } + + __zram_make_request(zram, bio); + return BLK_QC_T_NONE; + +error: + bio_io_error(bio); + return BLK_QC_T_NONE; +} + +static void zram_slot_free_notify(struct block_device *bdev, + unsigned long index) +{ + struct zram *zram; + + zram = bdev->bd_disk->private_data; + + atomic64_inc(&zram->stats.notify_free); + if (!zram_slot_trylock(zram, index)) { + atomic64_inc(&zram->stats.miss_free); + return; + } + +#ifdef CONFIG_HYBRIDSWAP_CORE + if (!hybridswap_delete(zram, index)) { + zram_slot_unlock(zram, index); + atomic64_inc(&zram->stats.miss_free); + return; + } +#endif + zram_free_page(zram, index); + zram_slot_unlock(zram, index); +} + +static int zram_rw_page(struct block_device *bdev, sector_t sector, + struct page *page, unsigned int op) +{ + int offset, ret; + u32 index; + struct zram *zram; + struct bio_vec bv; + unsigned long start_time; + + if (PageTransHuge(page)) + return -ENOTSUPP; + zram = bdev->bd_disk->private_data; + + if (!valid_io_request(zram, sector, PAGE_SIZE)) { + atomic64_inc(&zram->stats.invalid_io); + ret = -EINVAL; + goto out; + } + + index = sector >> SECTORS_PER_PAGE_SHIFT; + offset = (sector & (SECTORS_PER_PAGE - 1)) << SECTOR_SHIFT; + + bv.bv_page = page; + bv.bv_len = PAGE_SIZE; + bv.bv_offset = 0; + + start_time = disk_start_io_acct(bdev->bd_disk, SECTORS_PER_PAGE, op); + ret = zram_bvec_rw(zram, &bv, index, offset, op, NULL); + disk_end_io_acct(bdev->bd_disk, op, start_time); +out: + /* + * If I/O fails, just return error(ie, non-zero) without + * calling page_endio. + * It causes resubmit the I/O with bio request by upper functions + * of rw_page(e.g., swap_readpage, __swap_writepage) and + * bio->bi_end_io does things to handle the error + * (e.g., SetPageError, set_page_dirty and extra works). + */ + if (unlikely(ret < 0)) + return ret; + + switch (ret) { + case 0: + page_endio(page, op_is_write(op), 0); + break; + case 1: + ret = 0; + break; + default: + WARN_ON(1); + } + return ret; +} + +static void zram_reset_device(struct zram *zram) +{ + struct zcomp *comp; + u64 disksize; + + down_write(&zram->init_lock); + + zram->limit_pages = 0; + + if (!init_done(zram)) { + up_write(&zram->init_lock); + return; + } + + comp = zram->comp; + disksize = zram->disksize; + zram->disksize = 0; + + set_capacity(zram->disk, 0); + part_stat_set_all(&zram->disk->part0, 0); + + up_write(&zram->init_lock); + /* I/O operation under all of CPU are done so let's free */ + zram_meta_free(zram, disksize); + memset(&zram->stats, 0, sizeof(zram->stats)); + zcomp_destroy(comp); + reset_bdev(zram); +} + +static ssize_t disksize_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + u64 disksize; + struct zcomp *comp; + struct zram *zram = dev_to_zram(dev); + int err; + + disksize = memparse(buf, NULL); + if (!disksize) + return -EINVAL; + + down_write(&zram->init_lock); + if (init_done(zram)) { + pr_info("Cannot change disksize for initialized device\n"); + err = -EBUSY; + goto out_unlock; + } + + disksize = PAGE_ALIGN(disksize); + if (!zram_meta_alloc(zram, disksize)) { + err = -ENOMEM; + goto out_unlock; + } + + comp = zcomp_create(zram->compressor); + if (IS_ERR(comp)) { + pr_err("Cannot initialise %s compressing backend\n", + zram->compressor); + err = PTR_ERR(comp); + goto out_free_meta; + } + + zram->comp = comp; + zram->disksize = disksize; + set_capacity(zram->disk, zram->disksize >> SECTOR_SHIFT); + + revalidate_disk_size(zram->disk, true); + up_write(&zram->init_lock); + + return len; + +out_free_meta: + zram_meta_free(zram, disksize); +out_unlock: + up_write(&zram->init_lock); + return err; +} + +static ssize_t reset_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned short do_reset; + struct zram *zram; + struct block_device *bdev; + + ret = kstrtou16(buf, 10, &do_reset); + if (ret) + return ret; + + if (!do_reset) + return -EINVAL; + + zram = dev_to_zram(dev); + bdev = bdget_disk(zram->disk, 0); + if (!bdev) + return -ENOMEM; + + mutex_lock(&bdev->bd_mutex); + /* Do not reset an active device or claimed device */ + if (bdev->bd_openers || zram->claim) { + mutex_unlock(&bdev->bd_mutex); + bdput(bdev); + return -EBUSY; + } + + /* From now on, anyone can't open /dev/zram[0-9] */ + zram->claim = true; + mutex_unlock(&bdev->bd_mutex); + + /* Make sure all the pending I/O are finished */ + fsync_bdev(bdev); + zram_reset_device(zram); + revalidate_disk_size(zram->disk, true); + bdput(bdev); + + mutex_lock(&bdev->bd_mutex); + zram->claim = false; + mutex_unlock(&bdev->bd_mutex); + + return len; +} + +static int zram_open(struct block_device *bdev, fmode_t mode) +{ + int ret = 0; + struct zram *zram; + + WARN_ON(!mutex_is_locked(&bdev->bd_mutex)); + + zram = bdev->bd_disk->private_data; + /* zram was claimed to reset so open request fails */ + if (zram->claim) + ret = -EBUSY; + + return ret; +} + +static const struct block_device_operations zram_devops = { + .open = zram_open, + .submit_bio = zram_submit_bio, + .swap_slot_free_notify = zram_slot_free_notify, + .rw_page = zram_rw_page, + .owner = THIS_MODULE +}; + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static const struct block_device_operations zram_wb_devops = { + .open = zram_open, + .submit_bio = zram_submit_bio, + .swap_slot_free_notify = zram_slot_free_notify, + .owner = THIS_MODULE +}; +#endif + +static DEVICE_ATTR_WO(compact); +static DEVICE_ATTR_RW(disksize); +static DEVICE_ATTR_RO(initstate); +static DEVICE_ATTR_WO(reset); +static DEVICE_ATTR_WO(mem_limit); +static DEVICE_ATTR_WO(mem_used_max); +static DEVICE_ATTR_WO(idle); +static DEVICE_ATTR_RW(max_comp_streams); +static DEVICE_ATTR_RW(comp_algorithm); +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static DEVICE_ATTR_RW(backing_dev); +static DEVICE_ATTR_WO(writeback); +static DEVICE_ATTR_RW(writeback_limit); +static DEVICE_ATTR_RW(writeback_limit_enable); +#endif +#ifdef CONFIG_HYBRIDSWAP +static DEVICE_ATTR_RO(hybridswap_vmstat); +static DEVICE_ATTR_RW(hybridswap_loglevel); +static DEVICE_ATTR_RW(hybridswap_enable); +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD +static DEVICE_ATTR_RW(hybridswap_swapd_pause); +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE +static DEVICE_ATTR_RW(hybridswap_core_enable); +static DEVICE_ATTR_RW(hybridswap_loop_device); +static DEVICE_ATTR_RW(hybridswap_dev_life); +static DEVICE_ATTR_RW(hybridswap_quota_day); +static DEVICE_ATTR_RO(hybridswap_report); +static DEVICE_ATTR_RO(hybridswap_stat_snap); +static DEVICE_ATTR_RO(hybridswap_meminfo); +static DEVICE_ATTR_RW(hybridswap_zram_increase); +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +static DEVICE_ATTR_RW(hybridswap_akcompress); +#endif + +static struct attribute *zram_disk_attrs[] = { + &dev_attr_disksize.attr, + &dev_attr_initstate.attr, + &dev_attr_reset.attr, + &dev_attr_compact.attr, + &dev_attr_mem_limit.attr, + &dev_attr_mem_used_max.attr, + &dev_attr_idle.attr, + &dev_attr_max_comp_streams.attr, + &dev_attr_comp_algorithm.attr, +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + &dev_attr_backing_dev.attr, + &dev_attr_writeback.attr, + &dev_attr_writeback_limit.attr, + &dev_attr_writeback_limit_enable.attr, +#endif + &dev_attr_io_stat.attr, + &dev_attr_mm_stat.attr, +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + &dev_attr_bd_stat.attr, +#endif + &dev_attr_debug_stat.attr, +#ifdef CONFIG_HYBRIDSWAP + &dev_attr_hybridswap_vmstat.attr, + &dev_attr_hybridswap_loglevel.attr, + &dev_attr_hybridswap_enable.attr, +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD + &dev_attr_hybridswap_swapd_pause.attr, +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + &dev_attr_hybridswap_core_enable.attr, + &dev_attr_hybridswap_report.attr, + &dev_attr_hybridswap_meminfo.attr, + &dev_attr_hybridswap_stat_snap.attr, + &dev_attr_hybridswap_loop_device.attr, + &dev_attr_hybridswap_dev_life.attr, + &dev_attr_hybridswap_quota_day.attr, + &dev_attr_hybridswap_zram_increase.attr, +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + &dev_attr_hybridswap_akcompress.attr, +#endif + NULL, +}; + +static const struct attribute_group zram_disk_attr_group = { + .attrs = zram_disk_attrs, +}; + +static const struct attribute_group *zram_disk_attr_groups[] = { + &zram_disk_attr_group, + NULL, +}; + +/* + * Allocate and initialize new zram device. the function returns + * '>= 0' device_id upon success, and negative value otherwise. + */ +static int zram_add(void) +{ + struct zram *zram; + struct request_queue *queue; + int ret, device_id; + + zram = kzalloc(sizeof(struct zram), GFP_KERNEL); + if (!zram) + return -ENOMEM; + + ret = idr_alloc(&zram_index_idr, zram, 0, 0, GFP_KERNEL); + if (ret < 0) + goto out_free_dev; + device_id = ret; + + init_rwsem(&zram->init_lock); +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + spin_lock_init(&zram->wb_limit_lock); +#endif + queue = blk_alloc_queue(NUMA_NO_NODE); + if (!queue) { + pr_err("Error allocating disk queue for device %d\n", + device_id); + ret = -ENOMEM; + goto out_free_idr; + } + + /* gendisk structure */ + zram->disk = alloc_disk(1); + if (!zram->disk) { + pr_err("Error allocating disk structure for device %d\n", + device_id); + ret = -ENOMEM; + goto out_free_queue; + } + + zram->disk->major = zram_major; + zram->disk->first_minor = device_id; + zram->disk->fops = &zram_devops; + zram->disk->queue = queue; + zram->disk->private_data = zram; + snprintf(zram->disk->disk_name, 16, "zram%d", device_id); + + /* Actual capacity set using syfs (/sys/block/zram/disksize */ + set_capacity(zram->disk, 0); + /* zram devices sort of resembles non-rotational disks */ + blk_queue_flag_set(QUEUE_FLAG_NONROT, zram->disk->queue); + blk_queue_flag_clear(QUEUE_FLAG_ADD_RANDOM, zram->disk->queue); + + /* + * To ensure that we always get PAGE_SIZE aligned + * and n*PAGE_SIZED sized I/O requests. + */ + blk_queue_physical_block_size(zram->disk->queue, PAGE_SIZE); + blk_queue_logical_block_size(zram->disk->queue, + ZRAM_LOGICAL_BLOCK_SIZE); + blk_queue_io_min(zram->disk->queue, PAGE_SIZE); + blk_queue_io_opt(zram->disk->queue, PAGE_SIZE); + zram->disk->queue->limits.discard_granularity = PAGE_SIZE; + blk_queue_max_discard_sectors(zram->disk->queue, UINT_MAX); + blk_queue_flag_set(QUEUE_FLAG_DISCARD, zram->disk->queue); + + /* + * zram_bio_discard() will clear all logical blocks if logical block + * size is identical with physical block size(PAGE_SIZE). But if it is + * different, we will skip discarding some parts of logical blocks in + * the part of the request range which isn't aligned to physical block + * size. So we can't ensure that all discarded logical blocks are + * zeroed. + */ + if (ZRAM_LOGICAL_BLOCK_SIZE == PAGE_SIZE) + blk_queue_max_write_zeroes_sectors(zram->disk->queue, UINT_MAX); + + blk_queue_flag_set(QUEUE_FLAG_STABLE_WRITES, zram->disk->queue); + device_add_disk(NULL, zram->disk, zram_disk_attr_groups); + + strlcpy(zram->compressor, default_compressor, sizeof(zram->compressor)); + + zram_debugfs_register(zram); + pr_info("Added device: %s\n", zram->disk->disk_name); + return device_id; + +out_free_queue: + blk_cleanup_queue(queue); +out_free_idr: + idr_remove(&zram_index_idr, device_id); +out_free_dev: + kfree(zram); + return ret; +} + +static int zram_remove(struct zram *zram) +{ + struct block_device *bdev; + + bdev = bdget_disk(zram->disk, 0); + if (!bdev) + return -ENOMEM; + + mutex_lock(&bdev->bd_mutex); + if (bdev->bd_openers || zram->claim) { + mutex_unlock(&bdev->bd_mutex); + bdput(bdev); + return -EBUSY; + } + + zram->claim = true; + mutex_unlock(&bdev->bd_mutex); + + zram_debugfs_unregister(zram); + + /* Make sure all the pending I/O are finished */ + fsync_bdev(bdev); + zram_reset_device(zram); + bdput(bdev); + + pr_info("Removed device: %s\n", zram->disk->disk_name); + + del_gendisk(zram->disk); + blk_cleanup_queue(zram->disk->queue); + put_disk(zram->disk); + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + destroy_akcompressd_task(zram); +#endif + kfree(zram); + return 0; +} + +/* zram-control sysfs attributes */ + +/* + * NOTE: hot_add attribute is not the usual read-only sysfs attribute. In a + * sense that reading from this file does alter the state of your system -- it + * creates a new un-initialized zram device and returns back this device's + * device_id (or an error code if it fails to create a new device). + */ +static ssize_t hot_add_show(struct class *class, + struct class_attribute *attr, + char *buf) +{ + int ret; + + mutex_lock(&zram_index_mutex); + ret = zram_add(); + mutex_unlock(&zram_index_mutex); + + if (ret < 0) + return ret; + return scnprintf(buf, PAGE_SIZE, "%d\n", ret); +} +static struct class_attribute class_attr_hot_add = + __ATTR(hot_add, 0400, hot_add_show, NULL); + +static ssize_t hot_remove_store(struct class *class, + struct class_attribute *attr, + const char *buf, + size_t count) +{ + struct zram *zram; + int ret, dev_id; + + /* dev_id is gendisk->first_minor, which is `int' */ + ret = kstrtoint(buf, 10, &dev_id); + if (ret) + return ret; + if (dev_id < 0) + return -EINVAL; + + mutex_lock(&zram_index_mutex); + + zram = idr_find(&zram_index_idr, dev_id); + if (zram) { + ret = zram_remove(zram); + if (!ret) + idr_remove(&zram_index_idr, dev_id); + } else { + ret = -ENODEV; + } + + mutex_unlock(&zram_index_mutex); + return ret ? ret : count; +} +static CLASS_ATTR_WO(hot_remove); + +static struct attribute *zram_control_class_attrs[] = { + &class_attr_hot_add.attr, + &class_attr_hot_remove.attr, + NULL, +}; +ATTRIBUTE_GROUPS(zram_control_class); + +static struct class zram_control_class = { + .name = "zram-control", + .owner = THIS_MODULE, + .class_groups = zram_control_class_groups, +}; + +static int zram_remove_cb(int id, void *ptr, void *data) +{ + zram_remove(ptr); + return 0; +} + +static void destroy_devices(void) +{ + class_unregister(&zram_control_class); + idr_for_each(&zram_index_idr, &zram_remove_cb, NULL); + zram_debugfs_destroy(); + idr_destroy(&zram_index_idr); + unregister_blkdev(zram_major, "zram"); + cpuhp_remove_multi_state(CPUHP_ZCOMP_PREPARE); +} + +static int __init zram_init(void) +{ + int ret; + + ret = cpuhp_setup_state_multi(CPUHP_ZCOMP_PREPARE, "block/zram:prepare", + zcomp_cpu_up_prepare, zcomp_cpu_dead); + if (ret < 0) + return ret; + + ret = class_register(&zram_control_class); + if (ret) { + pr_err("Unable to register zram-control class\n"); + cpuhp_remove_multi_state(CPUHP_ZCOMP_PREPARE); + return ret; + } + + zram_debugfs_create(); + zram_major = register_blkdev(0, "zram"); + if (zram_major <= 0) { + pr_err("Unable to get major number\n"); + class_unregister(&zram_control_class); + cpuhp_remove_multi_state(CPUHP_ZCOMP_PREPARE); + return -EBUSY; + } + + while (num_devices != 0) { + mutex_lock(&zram_index_mutex); + ret = zram_add(); + mutex_unlock(&zram_index_mutex); + if (ret < 0) + goto out_error; + num_devices--; + } + +#ifdef CONFIG_HYBRIDSWAP + ret = hybridswap_pre_init(); + if (ret) + goto out_error; +#endif + return 0; + +out_error: + destroy_devices(); + return ret; +} + +static void __exit zram_exit(void) +{ + destroy_devices(); +} + +module_init(zram_init); +module_exit(zram_exit); + +module_param(num_devices, uint, 0); +MODULE_PARM_DESC(num_devices, "Number of pre-created zram devices"); + +MODULE_LICENSE("Dual BSD/GPL"); +MODULE_AUTHOR("Nitin Gupta "); +MODULE_DESCRIPTION("Compressed RAM Block Device"); diff --git a/drivers/moto_swap/zram-5.10/zram_drv.h b/drivers/moto_swap/zram-5.10/zram_drv.h new file mode 100644 index 000000000000..7e8e5ab8c148 --- /dev/null +++ b/drivers/moto_swap/zram-5.10/zram_drv.h @@ -0,0 +1,150 @@ +/* + * Compressed RAM block device + * + * Copyright (C) 2008, 2009, 2010 Nitin Gupta + * 2012, 2013 Minchan Kim + * + * This code is released using a dual license strategy: BSD/GPL + * You can choose the licence that better fits your requirements. + * + * Released under the terms of 3-clause BSD License + * Released under the terms of GNU General Public License Version 2.0 + * + */ + +#ifndef _ZRAM_DRV_H_ +#define _ZRAM_DRV_H_ + +#include +#include +#include + +#include "zcomp.h" + +#define SECTORS_PER_PAGE_SHIFT (PAGE_SHIFT - SECTOR_SHIFT) +#define SECTORS_PER_PAGE (1 << SECTORS_PER_PAGE_SHIFT) +#define ZRAM_LOGICAL_BLOCK_SHIFT 12 +#define ZRAM_LOGICAL_BLOCK_SIZE (1 << ZRAM_LOGICAL_BLOCK_SHIFT) +#define ZRAM_SECTOR_PER_LOGICAL_BLOCK \ + (1 << (ZRAM_LOGICAL_BLOCK_SHIFT - SECTOR_SHIFT)) + + +/* + * The lower ZRAM_FLAG_SHIFT bits of table.flags is for + * object size (excluding header), the higher bits is for + * zram_pageflags. + * + * zram is mainly used for memory efficiency so we want to keep memory + * footprint small so we can squeeze size and flags into a field. + * The lower ZRAM_FLAG_SHIFT bits is for object size (excluding header), + * the higher bits is for zram_pageflags. + */ +#define ZRAM_FLAG_SHIFT 24 + +/* Flags for zram pages (table[page_no].flags) */ +enum zram_pageflags { + /* zram slot is locked */ + ZRAM_LOCK = ZRAM_FLAG_SHIFT, + ZRAM_SAME, /* Page consists the same element */ + ZRAM_WB, /* page is stored on backing_device */ + ZRAM_UNDER_WB, /* page is under writeback */ + ZRAM_HUGE, /* Incompressible page */ + ZRAM_IDLE, /* not accessed page since last idle marking */ +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + ZRAM_CACHED, /* page is cached in async compress cache buffer */ + ZRAM_CACHED_COMPRESS, /* page is under async compress */ +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + ZRAM_BATCHING_OUT, + ZRAM_FROM_HYBRIDSWAP, + ZRAM_MCGID_CLEAR, + ZRAM_IN_BD, /* zram stored in back device */ +#endif + __NR_ZRAM_PAGEFLAGS, +}; + +/*-- Data structures */ + +/* Allocated for each disk page */ +struct zram_table_entry { + union { + unsigned long handle; + unsigned long element; +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + unsigned long page; +#endif + }; + unsigned long flags; +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + ktime_t ac_time; +#endif +}; + +struct zram_stats { + atomic64_t compr_data_size; /* compressed size of pages stored */ + atomic64_t num_reads; /* failed + successful */ + atomic64_t num_writes; /* --do-- */ + atomic64_t failed_reads; /* can happen when memory is too low */ + atomic64_t failed_writes; /* can happen when memory is too low */ + atomic64_t invalid_io; /* non-page-aligned I/O requests */ + atomic64_t notify_free; /* no. of swap slot free notifications */ + atomic64_t same_pages; /* no. of same element filled pages */ + atomic64_t huge_pages; /* no. of huge pages */ + atomic64_t pages_stored; /* no. of pages currently stored */ + atomic_long_t max_used_pages; /* no. of maximum pages stored */ + atomic64_t writestall; /* no. of write slow paths */ + atomic64_t miss_free; /* no. of missed free */ +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + atomic64_t bd_count; /* no. of pages in backing device */ + atomic64_t bd_reads; /* no. of reads from backing device */ + atomic64_t bd_writes; /* no. of writes from backing device */ +#endif +}; + +struct zram { + struct zram_table_entry *table; + struct zs_pool *mem_pool; + struct zcomp *comp; + struct gendisk *disk; + /* Prevent concurrent execution of device init */ + struct rw_semaphore init_lock; + /* + * the number of pages zram can consume for storing compressed data + */ + unsigned long limit_pages; + + struct zram_stats stats; + /* + * This is the limit on amount of *uncompressed* worth of data + * we can store in a disk. + */ + u64 disksize; /* bytes */ + char compressor[CRYPTO_MAX_ALG_NAME]; + /* + * zram is claimed so open request will be failed + */ + bool claim; /* Protected by bdev->bd_mutex */ + struct file *backing_dev; +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + spinlock_t wb_limit_lock; + bool wb_limit_enable; + u64 bd_wb_limit; + struct block_device *bdev; + unsigned int old_block_size; + unsigned long *bitmap; + unsigned long nr_pages; +#endif +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + struct dentry *debugfs_dir; +#endif +#if (defined CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK) || (defined CONFIG_HYBRIDSWAP_CORE) + struct block_device *bdev; + unsigned int old_block_size; + unsigned long nr_pages; + unsigned long increase_nr_pages; +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + struct hyb_info *infos; +#endif +}; +#endif diff --git a/drivers/moto_swap/zram-5.10/zram_drv_internal.h b/drivers/moto_swap/zram-5.10/zram_drv_internal.h new file mode 100644 index 000000000000..3c102cf38773 --- /dev/null +++ b/drivers/moto_swap/zram-5.10/zram_drv_internal.h @@ -0,0 +1,39 @@ +#ifndef _ZRAM_DRV_INTERNAL_H_ +#define _ZRAM_DRV_INTERNAL_H_ +#ifdef BIT +#undef BIT +#define BIT(nr) (1lu << (nr)) +#endif + +#define zram_slot_lock(zram, index) (bit_spin_lock(ZRAM_LOCK, &zram->table[index].flags)) + +#define zram_slot_unlock(zram, index) (bit_spin_unlock(ZRAM_LOCK, &zram->table[index].flags)) + +#define init_done(zram) (zram->disksize) + +#define dev_to_zram(dev) ((struct zram *)dev_to_disk(dev)->private_data) + +#define zram_get_handle(zram, index) (zram->table[index].handle) + +#define zram_set_handle(zram, index, handle_val) (zram->table[index].handle = handle_val) + +#define zram_test_flag(zram, index, flag) (zram->table[index].flags & BIT(flag)) + +#define zram_set_flag(zram, index, flag) (zram->table[index].flags |= BIT(flag)) + +#define zram_clear_flag(zram, index, flag) (zram->table[index].flags &= ~BIT(flag)) + +#define zram_set_element(zram, index, element) (zram->table[index].element = element) + +#define zram_get_obj_size(zram, index) (zram->table[index].flags & (BIT(ZRAM_FLAG_SHIFT) - 1)) + +#define zram_set_obj_size(zram, index, size) do {\ + unsigned long flags = zram->table[index].flags >> ZRAM_FLAG_SHIFT; \ + zram->table[index].flags = (flags << ZRAM_FLAG_SHIFT) | size; \ +} while(0) + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +extern int async_compress_page(struct zram *zram, struct page* page); +extern void update_zram_index(struct zram *zram, u32 index, unsigned long page); +#endif +#endif diff --git a/drivers/moto_swap/zram-5.4/zcomp.c b/drivers/moto_swap/zram-5.4/zcomp.c new file mode 100644 index 000000000000..1a8564a79d8d --- /dev/null +++ b/drivers/moto_swap/zram-5.4/zcomp.c @@ -0,0 +1,239 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +/* + * Copyright (C) 2014 Sergey Senozhatsky. + */ + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "zcomp.h" + +static const char * const backends[] = { + "lzo", + "lzo-rle", +#if IS_ENABLED(CONFIG_CRYPTO_LZ4) + "lz4", +#endif +#if IS_ENABLED(CONFIG_CRYPTO_LZ4HC) + "lz4hc", +#endif +#if IS_ENABLED(CONFIG_CRYPTO_842) + "842", +#endif +#if IS_ENABLED(CONFIG_CRYPTO_ZSTD) + "zstd", +#endif + NULL +}; + +static void zcomp_strm_free(struct zcomp_strm *zstrm) +{ + if (!IS_ERR_OR_NULL(zstrm->tfm)) + crypto_free_comp(zstrm->tfm); + free_pages((unsigned long)zstrm->buffer, 1); + kfree(zstrm); +} + +/* + * allocate new zcomp_strm structure with ->tfm initialized by + * backend, return NULL on error + */ +static struct zcomp_strm *zcomp_strm_alloc(struct zcomp *comp) +{ + struct zcomp_strm *zstrm = kmalloc(sizeof(*zstrm), GFP_KERNEL); + if (!zstrm) + return NULL; + + zstrm->tfm = crypto_alloc_comp(comp->name, 0, 0); + /* + * allocate 2 pages. 1 for compressed data, plus 1 extra for the + * case when compressed size is larger than the original one + */ + zstrm->buffer = (void *)__get_free_pages(GFP_KERNEL | __GFP_ZERO, 1); + if (IS_ERR_OR_NULL(zstrm->tfm) || !zstrm->buffer) { + zcomp_strm_free(zstrm); + zstrm = NULL; + } + return zstrm; +} + +bool zcomp_available_algorithm(const char *comp) +{ + int i; + + i = __sysfs_match_string(backends, -1, comp); + if (i >= 0) + return true; + + /* + * Crypto does not ignore a trailing new line symbol, + * so make sure you don't supply a string containing + * one. + * This also means that we permit zcomp initialisation + * with any compressing algorithm known to crypto api. + */ + return crypto_has_comp(comp, 0, 0) == 1; +} + +/* show available compressors */ +ssize_t zcomp_available_show(const char *comp, char *buf) +{ + bool known_algorithm = false; + ssize_t sz = 0; + int i = 0; + + for (; backends[i]; i++) { + if (!strcmp(comp, backends[i])) { + known_algorithm = true; + sz += scnprintf(buf + sz, PAGE_SIZE - sz - 2, + "[%s] ", backends[i]); + } else { + sz += scnprintf(buf + sz, PAGE_SIZE - sz - 2, + "%s ", backends[i]); + } + } + + /* + * Out-of-tree module known to crypto api or a missing + * entry in `backends'. + */ + if (!known_algorithm && crypto_has_comp(comp, 0, 0) == 1) + sz += scnprintf(buf + sz, PAGE_SIZE - sz - 2, + "[%s] ", comp); + + sz += scnprintf(buf + sz, PAGE_SIZE - sz, "\n"); + return sz; +} + +struct zcomp_strm *zcomp_stream_get(struct zcomp *comp) +{ + return *get_cpu_ptr(comp->stream); +} + +void zcomp_stream_put(struct zcomp *comp) +{ + put_cpu_ptr(comp->stream); +} + +int zcomp_compress(struct zcomp_strm *zstrm, + const void *src, unsigned int *dst_len) +{ + /* + * Our dst memory (zstrm->buffer) is always `2 * PAGE_SIZE' sized + * because sometimes we can endup having a bigger compressed data + * due to various reasons: for example compression algorithms tend + * to add some padding to the compressed buffer. Speaking of padding, + * comp algorithm `842' pads the compressed length to multiple of 8 + * and returns -ENOSP when the dst memory is not big enough, which + * is not something that ZRAM wants to see. We can handle the + * `compressed_size > PAGE_SIZE' case easily in ZRAM, but when we + * receive -ERRNO from the compressing backend we can't help it + * anymore. To make `842' happy we need to tell the exact size of + * the dst buffer, zram_drv will take care of the fact that + * compressed buffer is too big. + */ + *dst_len = PAGE_SIZE * 2; + + return crypto_comp_compress(zstrm->tfm, + src, PAGE_SIZE, + zstrm->buffer, dst_len); +} + +int zcomp_decompress(struct zcomp_strm *zstrm, + const void *src, unsigned int src_len, void *dst) +{ + unsigned int dst_len = PAGE_SIZE; + + return crypto_comp_decompress(zstrm->tfm, + src, src_len, + dst, &dst_len); +} + +int zcomp_cpu_up_prepare(unsigned int cpu, struct hlist_node *node) +{ + struct zcomp *comp = hlist_entry(node, struct zcomp, node); + struct zcomp_strm *zstrm; + + if (WARN_ON(*per_cpu_ptr(comp->stream, cpu))) + return 0; + + zstrm = zcomp_strm_alloc(comp); + if (IS_ERR_OR_NULL(zstrm)) { + pr_err("Can't allocate a compression stream\n"); + return -ENOMEM; + } + *per_cpu_ptr(comp->stream, cpu) = zstrm; + return 0; +} + +int zcomp_cpu_dead(unsigned int cpu, struct hlist_node *node) +{ + struct zcomp *comp = hlist_entry(node, struct zcomp, node); + struct zcomp_strm *zstrm; + + zstrm = *per_cpu_ptr(comp->stream, cpu); + if (!IS_ERR_OR_NULL(zstrm)) + zcomp_strm_free(zstrm); + *per_cpu_ptr(comp->stream, cpu) = NULL; + return 0; +} + +static int zcomp_init(struct zcomp *comp) +{ + int ret; + + comp->stream = alloc_percpu(struct zcomp_strm *); + if (!comp->stream) + return -ENOMEM; + + ret = cpuhp_state_add_instance(CPUHP_ZCOMP_PREPARE, &comp->node); + if (ret < 0) + goto cleanup; + return 0; + +cleanup: + free_percpu(comp->stream); + return ret; +} + +void zcomp_destroy(struct zcomp *comp) +{ + cpuhp_state_remove_instance(CPUHP_ZCOMP_PREPARE, &comp->node); + free_percpu(comp->stream); + kfree(comp); +} + +/* + * search available compressors for requested algorithm. + * allocate new zcomp and initialize it. return compressing + * backend pointer or ERR_PTR if things went bad. ERR_PTR(-EINVAL) + * if requested algorithm is not supported, ERR_PTR(-ENOMEM) in + * case of allocation error, or any other error potentially + * returned by zcomp_init(). + */ +struct zcomp *zcomp_create(const char *compress) +{ + struct zcomp *comp; + int error; + + if (!zcomp_available_algorithm(compress)) + return ERR_PTR(-EINVAL); + + comp = kzalloc(sizeof(struct zcomp), GFP_KERNEL); + if (!comp) + return ERR_PTR(-ENOMEM); + + comp->name = compress; + error = zcomp_init(comp); + if (error) { + kfree(comp); + return ERR_PTR(error); + } + return comp; +} diff --git a/drivers/moto_swap/zram-5.4/zcomp.h b/drivers/moto_swap/zram-5.4/zcomp.h new file mode 100644 index 000000000000..1806475b919d --- /dev/null +++ b/drivers/moto_swap/zram-5.4/zcomp.h @@ -0,0 +1,40 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * Copyright (C) 2014 Sergey Senozhatsky. + */ + +#ifndef _ZCOMP_H_ +#define _ZCOMP_H_ + +struct zcomp_strm { + /* compression/decompression buffer */ + void *buffer; + struct crypto_comp *tfm; +}; + +/* dynamic per-device compression frontend */ +struct zcomp { + struct zcomp_strm * __percpu *stream; + const char *name; + struct hlist_node node; +}; + +int zcomp_cpu_up_prepare(unsigned int cpu, struct hlist_node *node); +int zcomp_cpu_dead(unsigned int cpu, struct hlist_node *node); +ssize_t zcomp_available_show(const char *comp, char *buf); +bool zcomp_available_algorithm(const char *comp); + +struct zcomp *zcomp_create(const char *comp); +void zcomp_destroy(struct zcomp *comp); + +struct zcomp_strm *zcomp_stream_get(struct zcomp *comp); +void zcomp_stream_put(struct zcomp *comp); + +int zcomp_compress(struct zcomp_strm *zstrm, + const void *src, unsigned int *dst_len); + +int zcomp_decompress(struct zcomp_strm *zstrm, + const void *src, unsigned int src_len, void *dst); + +bool zcomp_set_max_streams(struct zcomp *comp, int num_strm); +#endif /* _ZCOMP_H_ */ diff --git a/drivers/moto_swap/zram-5.4/zram_drv.c b/drivers/moto_swap/zram-5.4/zram_drv.c new file mode 100644 index 000000000000..7abdf6e8823b --- /dev/null +++ b/drivers/moto_swap/zram-5.4/zram_drv.c @@ -0,0 +1,2333 @@ +/* + * Compressed RAM block device + * + * Copyright (C) 2008, 2009, 2010 Nitin Gupta + * 2012, 2013 Minchan Kim + * + * This code is released using a dual license strategy: BSD/GPL + * You can choose the licence that better fits your requirements. + * + * Released under the terms of 3-clause BSD License + * Released under the terms of GNU General Public License Version 2.0 + * + */ + +#define KMSG_COMPONENT "zram" +#define pr_fmt(fmt) KMSG_COMPONENT ": " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "zram_drv.h" +#include "zram_drv_internal.h" +#ifdef CONFIG_HYBRIDSWAP +#include "../hybridswap/hybridswap.h" +#endif + +static DEFINE_IDR(zram_index_idr); +/* idr index must be protected */ +static DEFINE_MUTEX(zram_index_mutex); + +static int zram_major; +static const char *default_compressor = "lzo-rle"; + +/* Module params (documentation at end) */ +static unsigned int num_devices = 1; +/* + * Pages that compress to sizes equals or greater than this are stored + * uncompressed in memory. + */ +static size_t huge_class_size; + +static const struct block_device_operations zram_devops; +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static const struct block_device_operations zram_wb_devops; +#endif + +static void zram_free_page(struct zram *zram, size_t index); +static int zram_bvec_read(struct zram *zram, struct bio_vec *bvec, + u32 index, int offset, struct bio *bio); + + +static int zram_slot_trylock(struct zram *zram, u32 index) +{ + return bit_spin_trylock(ZRAM_LOCK, &zram->table[index].flags); +} + +static unsigned long zram_get_element(struct zram *zram, u32 index) +{ + return zram->table[index].element; +} + +static inline bool zram_allocated(struct zram *zram, u32 index) +{ + return zram_get_obj_size(zram, index) || + zram_test_flag(zram, index, ZRAM_SAME) || + zram_test_flag(zram, index, ZRAM_WB); +} + +#if PAGE_SIZE != 4096 +static inline bool is_partial_io(struct bio_vec *bvec) +{ + return bvec->bv_len != PAGE_SIZE; +} +#else +static inline bool is_partial_io(struct bio_vec *bvec) +{ + return false; +} +#endif + +/* + * Check if request is within bounds and aligned on zram logical blocks. + */ +static inline bool valid_io_request(struct zram *zram, + sector_t start, unsigned int size) +{ + u64 end, bound; + + /* unaligned request */ + if (unlikely(start & (ZRAM_SECTOR_PER_LOGICAL_BLOCK - 1))) + return false; + if (unlikely(size & (ZRAM_LOGICAL_BLOCK_SIZE - 1))) + return false; + + end = start + (size >> SECTOR_SHIFT); + bound = zram->disksize >> SECTOR_SHIFT; + /* out of range range */ + if (unlikely(start >= bound || end > bound || start > end)) + return false; + + /* I/O request is valid */ + return true; +} + +static void update_position(u32 *index, int *offset, struct bio_vec *bvec) +{ + *index += (*offset + bvec->bv_len) / PAGE_SIZE; + *offset = (*offset + bvec->bv_len) % PAGE_SIZE; +} + +static inline void update_used_max(struct zram *zram, + const unsigned long pages) +{ + unsigned long old_max, cur_max; + + old_max = atomic_long_read(&zram->stats.max_used_pages); + + do { + cur_max = old_max; + if (pages > cur_max) + old_max = atomic_long_cmpxchg( + &zram->stats.max_used_pages, cur_max, pages); + } while (old_max != cur_max); +} + +static inline void zram_fill_page(void *ptr, unsigned long len, + unsigned long value) +{ + WARN_ON_ONCE(!IS_ALIGNED(len, sizeof(unsigned long))); + memset_l(ptr, value, len / sizeof(unsigned long)); +} + +static bool page_same_filled(void *ptr, unsigned long *element) +{ + unsigned int pos; + unsigned long *page; + unsigned long val; + + page = (unsigned long *)ptr; + val = page[0]; + + for (pos = 1; pos < PAGE_SIZE / sizeof(*page); pos++) { + if (val != page[pos]) + return false; + } + + *element = val; + + return true; +} + +static ssize_t initstate_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + u32 val; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + val = init_done(zram); + up_read(&zram->init_lock); + + return scnprintf(buf, PAGE_SIZE, "%u\n", val); +} + +static ssize_t disksize_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + + return scnprintf(buf, PAGE_SIZE, "%llu\n", zram->disksize); +} + +static ssize_t mem_limit_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + u64 limit; + char *tmp; + struct zram *zram = dev_to_zram(dev); + + limit = memparse(buf, &tmp); + if (buf == tmp) /* no chars parsed, invalid input */ + return -EINVAL; + + down_write(&zram->init_lock); + zram->limit_pages = PAGE_ALIGN(limit) >> PAGE_SHIFT; + up_write(&zram->init_lock); + + return len; +} + +static ssize_t mem_used_max_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int err; + unsigned long val; + struct zram *zram = dev_to_zram(dev); + + err = kstrtoul(buf, 10, &val); + if (err || val != 0) + return -EINVAL; + + down_read(&zram->init_lock); + if (init_done(zram)) { + atomic_long_set(&zram->stats.max_used_pages, + zs_get_total_pages(zram->mem_pool)); + } + up_read(&zram->init_lock); + + return len; +} + +static ssize_t idle_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; + int index; + + if (!sysfs_streq(buf, "all")) + return -EINVAL; + + down_read(&zram->init_lock); + if (!init_done(zram)) { + up_read(&zram->init_lock); + return -EINVAL; + } + + for (index = 0; index < nr_pages; index++) { + /* + * Do not mark ZRAM_UNDER_WB slot as ZRAM_IDLE to close race. + * See the comment in writeback_store. + */ + zram_slot_lock(zram, index); + if (zram_allocated(zram, index) && + !zram_test_flag(zram, index, ZRAM_UNDER_WB)) + zram_set_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + } + + up_read(&zram->init_lock); + + return len; +} + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static ssize_t writeback_limit_enable_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + u64 val; + ssize_t ret = -EINVAL; + + if (kstrtoull(buf, 10, &val)) + return ret; + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + zram->wb_limit_enable = val; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + ret = len; + + return ret; +} + +static ssize_t writeback_limit_enable_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + bool val; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + val = zram->wb_limit_enable; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + + return scnprintf(buf, PAGE_SIZE, "%d\n", val); +} + +static ssize_t writeback_limit_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + u64 val; + ssize_t ret = -EINVAL; + + if (kstrtoull(buf, 10, &val)) + return ret; + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + zram->bd_wb_limit = val; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + ret = len; + + return ret; +} + +static ssize_t writeback_limit_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + u64 val; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + spin_lock(&zram->wb_limit_lock); + val = zram->bd_wb_limit; + spin_unlock(&zram->wb_limit_lock); + up_read(&zram->init_lock); + + return scnprintf(buf, PAGE_SIZE, "%llu\n", val); +} + +static void reset_bdev(struct zram *zram) +{ + struct block_device *bdev; + + if (!zram->backing_dev) + return; + + bdev = zram->bdev; + if (zram->old_block_size) + set_blocksize(bdev, zram->old_block_size); + blkdev_put(bdev, FMODE_READ|FMODE_WRITE|FMODE_EXCL); + /* hope filp_close flush all of IO */ + filp_close(zram->backing_dev, NULL); + zram->backing_dev = NULL; + zram->old_block_size = 0; + zram->bdev = NULL; + zram->disk->queue->backing_dev_info->capabilities |= + BDI_CAP_SYNCHRONOUS_IO; + kvfree(zram->bitmap); + zram->bitmap = NULL; +} + +static ssize_t backing_dev_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct file *file; + struct zram *zram = dev_to_zram(dev); + char *p; + ssize_t ret; + + down_read(&zram->init_lock); + file = zram->backing_dev; + if (!file) { + memcpy(buf, "none\n", 5); + up_read(&zram->init_lock); + return 5; + } + + p = file_path(file, buf, PAGE_SIZE - 1); + if (IS_ERR(p)) { + ret = PTR_ERR(p); + goto out; + } + + ret = strlen(p); + memmove(buf, p, ret); + buf[ret++] = '\n'; +out: + up_read(&zram->init_lock); + return ret; +} + +static ssize_t backing_dev_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + char *file_name; + size_t sz; + struct file *backing_dev = NULL; + struct inode *inode; + struct address_space *mapping; + unsigned int bitmap_sz, old_block_size = 0; + unsigned long nr_pages, *bitmap = NULL; + struct block_device *bdev = NULL; + int err; + struct zram *zram = dev_to_zram(dev); + + file_name = kmalloc(PATH_MAX, GFP_KERNEL); + if (!file_name) + return -ENOMEM; + + down_write(&zram->init_lock); + if (init_done(zram)) { + pr_info("Can't setup backing device for initialized device\n"); + err = -EBUSY; + goto out; + } + + strlcpy(file_name, buf, PATH_MAX); + /* ignore trailing newline */ + sz = strlen(file_name); + if (sz > 0 && file_name[sz - 1] == '\n') + file_name[sz - 1] = 0x00; + + backing_dev = filp_open(file_name, O_RDWR|O_LARGEFILE, 0); + if (IS_ERR(backing_dev)) { + err = PTR_ERR(backing_dev); + backing_dev = NULL; + goto out; + } + + mapping = backing_dev->f_mapping; + inode = mapping->host; + + /* Support only block device in this moment */ + if (!S_ISBLK(inode->i_mode)) { + err = -ENOTBLK; + goto out; + } + + bdev = bdgrab(I_BDEV(inode)); + err = blkdev_get(bdev, FMODE_READ | FMODE_WRITE | FMODE_EXCL, zram); + if (err < 0) { + bdev = NULL; + goto out; + } + + nr_pages = i_size_read(inode) >> PAGE_SHIFT; + bitmap_sz = BITS_TO_LONGS(nr_pages) * sizeof(long); + bitmap = kvzalloc(bitmap_sz, GFP_KERNEL); + if (!bitmap) { + err = -ENOMEM; + goto out; + } + + old_block_size = block_size(bdev); + err = set_blocksize(bdev, PAGE_SIZE); + if (err) + goto out; + + reset_bdev(zram); + + zram->old_block_size = old_block_size; + zram->bdev = bdev; + zram->backing_dev = backing_dev; + zram->bitmap = bitmap; + zram->nr_pages = nr_pages; + /* + * With writeback feature, zram does asynchronous IO so it's no longer + * synchronous device so let's remove synchronous io flag. Othewise, + * upper layer(e.g., swap) could wait IO completion rather than + * (submit and return), which will cause system sluggish. + * Furthermore, when the IO function returns(e.g., swap_readpage), + * upper layer expects IO was done so it could deallocate the page + * freely but in fact, IO is going on so finally could cause + * use-after-free when the IO is really done. + */ + zram->disk->queue->backing_dev_info->capabilities &= + ~BDI_CAP_SYNCHRONOUS_IO; + up_write(&zram->init_lock); + + pr_info("setup backing device %s\n", file_name); + kfree(file_name); + + return len; +out: + if (bitmap) + kvfree(bitmap); + + if (bdev) + blkdev_put(bdev, FMODE_READ | FMODE_WRITE | FMODE_EXCL); + + if (backing_dev) + filp_close(backing_dev, NULL); + + up_write(&zram->init_lock); + + kfree(file_name); + + return err; +} + +static unsigned long alloc_block_bdev(struct zram *zram) +{ + unsigned long blk_idx = 1; +retry: + /* skip 0 bit to confuse zram.handle = 0 */ + blk_idx = find_next_zero_bit(zram->bitmap, zram->nr_pages, blk_idx); + if (blk_idx == zram->nr_pages) + return 0; + + if (test_and_set_bit(blk_idx, zram->bitmap)) + goto retry; + + atomic64_inc(&zram->stats.bd_count); + return blk_idx; +} + +static void free_block_bdev(struct zram *zram, unsigned long blk_idx) +{ + int was_set; + + was_set = test_and_clear_bit(blk_idx, zram->bitmap); + WARN_ON_ONCE(!was_set); + atomic64_dec(&zram->stats.bd_count); +} + +static void zram_page_end_io(struct bio *bio) +{ + struct page *page = bio_first_page_all(bio); + + page_endio(page, op_is_write(bio_op(bio)), + blk_status_to_errno(bio->bi_status)); + bio_put(bio); +} + +/* + * Returns 1 if the submission is successful. + */ +static int read_from_bdev_async(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *parent) +{ + struct bio *bio; + + bio = bio_alloc(GFP_ATOMIC, 1); + if (!bio) + return -ENOMEM; + + bio->bi_iter.bi_sector = entry * (PAGE_SIZE >> 9); + bio_set_dev(bio, zram->bdev); + if (!bio_add_page(bio, bvec->bv_page, bvec->bv_len, bvec->bv_offset)) { + bio_put(bio); + return -EIO; + } + + if (!parent) { + bio->bi_opf = REQ_OP_READ; + bio->bi_end_io = zram_page_end_io; + } else { + bio->bi_opf = parent->bi_opf; + bio_chain(bio, parent); + } + + submit_bio(bio); + return 1; +} + +#define HUGE_WRITEBACK 1 +#define IDLE_WRITEBACK 2 + +static ssize_t writeback_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; + unsigned long index; + struct bio bio; + struct bio_vec bio_vec; + struct page *page; + ssize_t ret = len; + int mode, err; + unsigned long blk_idx = 0; + + if (sysfs_streq(buf, "idle")) + mode = IDLE_WRITEBACK; + else if (sysfs_streq(buf, "huge")) + mode = HUGE_WRITEBACK; + else + return -EINVAL; + + down_read(&zram->init_lock); + if (!init_done(zram)) { + ret = -EINVAL; + goto release_init_lock; + } + + if (!zram->backing_dev) { + ret = -ENODEV; + goto release_init_lock; + } + + page = alloc_page(GFP_KERNEL); + if (!page) { + ret = -ENOMEM; + goto release_init_lock; + } + + for (index = 0; index < nr_pages; index++) { + struct bio_vec bvec; + + bvec.bv_page = page; + bvec.bv_len = PAGE_SIZE; + bvec.bv_offset = 0; + + spin_lock(&zram->wb_limit_lock); + if (zram->wb_limit_enable && !zram->bd_wb_limit) { + spin_unlock(&zram->wb_limit_lock); + ret = -EIO; + break; + } + spin_unlock(&zram->wb_limit_lock); + + if (!blk_idx) { + blk_idx = alloc_block_bdev(zram); + if (!blk_idx) { + ret = -ENOSPC; + break; + } + } + + zram_slot_lock(zram, index); + if (!zram_allocated(zram, index)) + goto next; + + if (zram_test_flag(zram, index, ZRAM_WB) || + zram_test_flag(zram, index, ZRAM_SAME) || + zram_test_flag(zram, index, ZRAM_UNDER_WB)) + goto next; + + if (mode == IDLE_WRITEBACK && + !zram_test_flag(zram, index, ZRAM_IDLE)) + goto next; + if (mode == HUGE_WRITEBACK && + !zram_test_flag(zram, index, ZRAM_HUGE)) + goto next; + /* + * Clearing ZRAM_UNDER_WB is duty of caller. + * IOW, zram_free_page never clear it. + */ + zram_set_flag(zram, index, ZRAM_UNDER_WB); + /* Need for hugepage writeback racing */ + zram_set_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + if (zram_bvec_read(zram, &bvec, index, 0, NULL)) { + zram_slot_lock(zram, index); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_clear_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + continue; + } + + bio_init(&bio, &bio_vec, 1); + bio_set_dev(&bio, zram->bdev); + bio.bi_iter.bi_sector = blk_idx * (PAGE_SIZE >> 9); + bio.bi_opf = REQ_OP_WRITE | REQ_SYNC; + + bio_add_page(&bio, bvec.bv_page, bvec.bv_len, + bvec.bv_offset); + /* + * XXX: A single page IO would be inefficient for write + * but it would be not bad as starter. + */ + err = submit_bio_wait(&bio); + if (err) { + zram_slot_lock(zram, index); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_clear_flag(zram, index, ZRAM_IDLE); + zram_slot_unlock(zram, index); + /* + * Return last IO error unless every IO were + * not suceeded. + */ + ret = err; + continue; + } + + atomic64_inc(&zram->stats.bd_writes); + /* + * We released zram_slot_lock so need to check if the slot was + * changed. If there is freeing for the slot, we can catch it + * easily by zram_allocated. + * A subtle case is the slot is freed/reallocated/marked as + * ZRAM_IDLE again. To close the race, idle_store doesn't + * mark ZRAM_IDLE once it found the slot was ZRAM_UNDER_WB. + * Thus, we could close the race by checking ZRAM_IDLE bit. + */ + zram_slot_lock(zram, index); + if (!zram_allocated(zram, index) || + !zram_test_flag(zram, index, ZRAM_IDLE)) { + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_clear_flag(zram, index, ZRAM_IDLE); + goto next; + } + + zram_free_page(zram, index); + zram_clear_flag(zram, index, ZRAM_UNDER_WB); + zram_set_flag(zram, index, ZRAM_WB); + zram_set_element(zram, index, blk_idx); + blk_idx = 0; + atomic64_inc(&zram->stats.pages_stored); + spin_lock(&zram->wb_limit_lock); + if (zram->wb_limit_enable && zram->bd_wb_limit > 0) + zram->bd_wb_limit -= 1UL << (PAGE_SHIFT - 12); + spin_unlock(&zram->wb_limit_lock); +next: + zram_slot_unlock(zram, index); + } + + if (blk_idx) + free_block_bdev(zram, blk_idx); + __free_page(page); +release_init_lock: + up_read(&zram->init_lock); + + return ret; +} + +struct zram_work { + struct work_struct work; + struct zram *zram; + unsigned long entry; + struct bio *bio; + struct bio_vec bvec; +}; + +#if PAGE_SIZE != 4096 +static void zram_sync_read(struct work_struct *work) +{ + struct zram_work *zw = container_of(work, struct zram_work, work); + struct zram *zram = zw->zram; + unsigned long entry = zw->entry; + struct bio *bio = zw->bio; + + read_from_bdev_async(zram, &zw->bvec, entry, bio); +} + +/* + * Block layer want one ->make_request_fn to be active at a time + * so if we use chained IO with parent IO in same context, + * it's a deadlock. To avoid, it, it uses worker thread context. + */ +static int read_from_bdev_sync(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *bio) +{ + struct zram_work work; + + work.bvec = *bvec; + work.zram = zram; + work.entry = entry; + work.bio = bio; + + INIT_WORK_ONSTACK(&work.work, zram_sync_read); + queue_work(system_unbound_wq, &work.work); + flush_work(&work.work); + destroy_work_on_stack(&work.work); + + return 1; +} +#else +static int read_from_bdev_sync(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *bio) +{ + WARN_ON(1); + return -EIO; +} +#endif + +static int read_from_bdev(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *parent, bool sync) +{ + atomic64_inc(&zram->stats.bd_reads); + if (sync) + return read_from_bdev_sync(zram, bvec, entry, parent); + else + return read_from_bdev_async(zram, bvec, entry, parent); +} +#else +static inline void reset_bdev(struct zram *zram) {}; +static int read_from_bdev(struct zram *zram, struct bio_vec *bvec, + unsigned long entry, struct bio *parent, bool sync) +{ + return -EIO; +} + +static void free_block_bdev(struct zram *zram, unsigned long blk_idx) {}; +#endif + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + +static struct dentry *zram_debugfs_root; + +static void zram_debugfs_create(void) +{ + zram_debugfs_root = debugfs_create_dir("zram", NULL); +} + +static void zram_debugfs_destroy(void) +{ + debugfs_remove_recursive(zram_debugfs_root); +} + +static void zram_accessed(struct zram *zram, u32 index) +{ + zram_clear_flag(zram, index, ZRAM_IDLE); + zram->table[index].ac_time = ktime_get_boottime(); +} + +static ssize_t read_block_state(struct file *file, char __user *buf, + size_t count, loff_t *ppos) +{ + char *kbuf; + ssize_t index, written = 0; + struct zram *zram = file->private_data; + unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; + struct timespec64 ts; + + kbuf = kvmalloc(count, GFP_KERNEL); + if (!kbuf) + return -ENOMEM; + + down_read(&zram->init_lock); + if (!init_done(zram)) { + up_read(&zram->init_lock); + kvfree(kbuf); + return -EINVAL; + } + + for (index = *ppos; index < nr_pages; index++) { + int copied; + + zram_slot_lock(zram, index); + if (!zram_allocated(zram, index)) + goto next; + + ts = ktime_to_timespec64(zram->table[index].ac_time); + copied = snprintf(kbuf + written, count, + "%12zd %12lld.%06lu %c%c%c%c\n", + index, (s64)ts.tv_sec, + ts.tv_nsec / NSEC_PER_USEC, + zram_test_flag(zram, index, ZRAM_SAME) ? 's' : '.', + zram_test_flag(zram, index, ZRAM_WB) ? 'w' : '.', + zram_test_flag(zram, index, ZRAM_HUGE) ? 'h' : '.', + zram_test_flag(zram, index, ZRAM_IDLE) ? 'i' : '.'); + + if (count <= copied) { + zram_slot_unlock(zram, index); + break; + } + written += copied; + count -= copied; +next: + zram_slot_unlock(zram, index); + *ppos += 1; + } + + up_read(&zram->init_lock); + if (copy_to_user(buf, kbuf, written)) + written = -EFAULT; + kvfree(kbuf); + + return written; +} + +static const struct file_operations proc_zram_block_state_op = { + .open = simple_open, + .read = read_block_state, + .llseek = default_llseek, +}; + +static void zram_debugfs_register(struct zram *zram) +{ + if (!zram_debugfs_root) + return; + + zram->debugfs_dir = debugfs_create_dir(zram->disk->disk_name, + zram_debugfs_root); + debugfs_create_file("block_state", 0400, zram->debugfs_dir, + zram, &proc_zram_block_state_op); +} + +static void zram_debugfs_unregister(struct zram *zram) +{ + debugfs_remove_recursive(zram->debugfs_dir); +} +#else +static void zram_debugfs_create(void) {}; +static void zram_debugfs_destroy(void) {}; +static void zram_accessed(struct zram *zram, u32 index) +{ + zram_clear_flag(zram, index, ZRAM_IDLE); +}; +static void zram_debugfs_register(struct zram *zram) {}; +static void zram_debugfs_unregister(struct zram *zram) {}; +#endif + +/* + * We switched to per-cpu streams and this attr is not needed anymore. + * However, we will keep it around for some time, because: + * a) we may revert per-cpu streams in the future + * b) it's visible to user space and we need to follow our 2 years + * retirement rule; but we already have a number of 'soon to be + * altered' attrs, so max_comp_streams need to wait for the next + * layoff cycle. + */ +static ssize_t max_comp_streams_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + return scnprintf(buf, PAGE_SIZE, "%d\n", num_online_cpus()); +} + +static ssize_t max_comp_streams_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + return len; +} + +static ssize_t comp_algorithm_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + size_t sz; + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + sz = zcomp_available_show(zram->compressor, buf); + up_read(&zram->init_lock); + + return sz; +} + +static ssize_t comp_algorithm_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + char compressor[ARRAY_SIZE(zram->compressor)]; + size_t sz; + + strlcpy(compressor, buf, sizeof(compressor)); + /* ignore trailing newline */ + sz = strlen(compressor); + if (sz > 0 && compressor[sz - 1] == '\n') + compressor[sz - 1] = 0x00; + + if (!zcomp_available_algorithm(compressor)) + return -EINVAL; + + down_write(&zram->init_lock); + if (init_done(zram)) { + up_write(&zram->init_lock); + pr_info("Can't change algorithm for initialized device\n"); + return -EBUSY; + } + + strcpy(zram->compressor, compressor); + up_write(&zram->init_lock); + return len; +} + +static ssize_t compact_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + struct zram *zram = dev_to_zram(dev); + + down_read(&zram->init_lock); + if (!init_done(zram)) { + up_read(&zram->init_lock); + return -EINVAL; + } + + zs_compact(zram->mem_pool); + up_read(&zram->init_lock); + + return len; +} + +static ssize_t io_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + ssize_t ret; + + down_read(&zram->init_lock); + ret = scnprintf(buf, PAGE_SIZE, + "%8llu %8llu %8llu %8llu\n", + (u64)atomic64_read(&zram->stats.failed_reads), + (u64)atomic64_read(&zram->stats.failed_writes), + (u64)atomic64_read(&zram->stats.invalid_io), + (u64)atomic64_read(&zram->stats.notify_free)); + up_read(&zram->init_lock); + + return ret; +} + +static ssize_t mm_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + struct zs_pool_stats pool_stats; + u64 orig_size, mem_used = 0; + long max_used; + ssize_t ret; + + memset(&pool_stats, 0x00, sizeof(struct zs_pool_stats)); + + down_read(&zram->init_lock); + if (init_done(zram)) { + mem_used = zs_get_total_pages(zram->mem_pool); + zs_pool_stats(zram->mem_pool, &pool_stats); + } + + orig_size = atomic64_read(&zram->stats.pages_stored); + max_used = atomic_long_read(&zram->stats.max_used_pages); + + ret = scnprintf(buf, PAGE_SIZE, + "%8llu %8llu %8llu %8lu %8ld %8llu %8lu %8llu\n", + orig_size << PAGE_SHIFT, + (u64)atomic64_read(&zram->stats.compr_data_size), + mem_used << PAGE_SHIFT, + zram->limit_pages << PAGE_SHIFT, + max_used << PAGE_SHIFT, + (u64)atomic64_read(&zram->stats.same_pages), + atomic_long_read(&pool_stats.pages_compacted), + (u64)atomic64_read(&zram->stats.huge_pages)); + up_read(&zram->init_lock); + + return ret; +} + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +#define FOUR_K(x) ((x) * (1 << (PAGE_SHIFT - 12))) +static ssize_t bd_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct zram *zram = dev_to_zram(dev); + ssize_t ret; + + down_read(&zram->init_lock); + ret = scnprintf(buf, PAGE_SIZE, + "%8llu %8llu %8llu\n", + FOUR_K((u64)atomic64_read(&zram->stats.bd_count)), + FOUR_K((u64)atomic64_read(&zram->stats.bd_reads)), + FOUR_K((u64)atomic64_read(&zram->stats.bd_writes))); + up_read(&zram->init_lock); + + return ret; +} +#endif + +static ssize_t debug_stat_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + int version = 1; + struct zram *zram = dev_to_zram(dev); + ssize_t ret; + + down_read(&zram->init_lock); + ret = scnprintf(buf, PAGE_SIZE, + "version: %d\n%8llu %8llu\n", + version, + (u64)atomic64_read(&zram->stats.writestall), + (u64)atomic64_read(&zram->stats.miss_free)); + up_read(&zram->init_lock); + + return ret; +} + +static DEVICE_ATTR_RO(io_stat); +static DEVICE_ATTR_RO(mm_stat); +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static DEVICE_ATTR_RO(bd_stat); +#endif +static DEVICE_ATTR_RO(debug_stat); + +static void zram_meta_free(struct zram *zram, u64 disksize) +{ + size_t num_pages = disksize >> PAGE_SHIFT; + size_t index; + + /* Free all pages that are still in this zram device */ + for (index = 0; index < num_pages; index++) + zram_free_page(zram, index); + + zs_destroy_pool(zram->mem_pool); + vfree(zram->table); +} + +static bool zram_meta_alloc(struct zram *zram, u64 disksize) +{ + size_t num_pages; + + num_pages = disksize >> PAGE_SHIFT; + zram->table = vzalloc(array_size(num_pages, sizeof(*zram->table))); + if (!zram->table) + return false; + + zram->mem_pool = zs_create_pool(zram->disk->disk_name); + if (!zram->mem_pool) { + vfree(zram->table); + return false; + } + + if (!huge_class_size) + huge_class_size = zs_huge_class_size(zram->mem_pool); + return true; +} + +/* + * To protect concurrent access to the same index entry, + * caller should hold this table index entry's bit_spinlock to + * indicate this index entry is accessing. + */ +static void zram_free_page(struct zram *zram, size_t index) +{ + unsigned long handle; + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + zram->table[index].ac_time = 0; +#endif + if (zram_test_flag(zram, index, ZRAM_IDLE)) + zram_clear_flag(zram, index, ZRAM_IDLE); + + if (zram_test_flag(zram, index, ZRAM_HUGE)) { + zram_clear_flag(zram, index, ZRAM_HUGE); + atomic64_dec(&zram->stats.huge_pages); + } + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (zram_test_flag(zram, index, ZRAM_CACHED)) { + struct page *page = (struct page *)zram_get_page(zram, index); + + del_page_from_cache(page); + page->mem_cgroup = NULL; + put_free_page(page); + zram_clear_flag(zram, index, ZRAM_CACHED); + goto out; + } + + if (zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + zram_clear_flag(zram, index, ZRAM_CACHED_COMPRESS); + goto out; + } +#endif + +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_untrack(zram, index); +#endif + + if (zram_test_flag(zram, index, ZRAM_WB)) { + zram_clear_flag(zram, index, ZRAM_WB); + free_block_bdev(zram, zram_get_element(zram, index)); + atomic64_dec(&zram->stats.pages_stored); + goto out; + } + + /* + * No memory is allocated for same element filled pages. + * Simply clear same page flag. + */ + if (zram_test_flag(zram, index, ZRAM_SAME)) { + zram_clear_flag(zram, index, ZRAM_SAME); + atomic64_dec(&zram->stats.same_pages); + atomic64_dec(&zram->stats.pages_stored); + goto out; + } + + handle = zram_get_handle(zram, index); + if (!handle) + return; + + zs_free(zram->mem_pool, handle); + + atomic64_sub(zram_get_obj_size(zram, index), + &zram->stats.compr_data_size); + atomic64_dec(&zram->stats.pages_stored); + +out: + zram_set_handle(zram, index, 0); + zram_set_obj_size(zram, index, 0); + WARN_ON_ONCE(zram->table[index].flags & + ~(1UL << ZRAM_LOCK | 1UL << ZRAM_UNDER_WB)); +} + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +void update_zram_index(struct zram *zram, u32 index, unsigned long page) +{ + zram_slot_lock(zram, index); + put_anon_pages((struct page*)page); + + zram_free_page(zram, index); + zram_set_flag(zram, index, ZRAM_CACHED); + zram_set_page(zram, index, page); + zram_set_obj_size(zram, index, PAGE_SIZE); + zram_slot_unlock(zram, index); +} + +int async_compress_page(struct zram *zram, struct page* page) +{ + int ret = 0; + unsigned long alloced_pages; + unsigned long handle = 0; + unsigned int comp_len = 0; + void *src, *dst; + struct zcomp_strm *zstrm; + int index = get_zram_index(page); + +compress_again: + zram_slot_lock(zram, index); + if (!zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + zram_slot_unlock(zram, index); + return 0; + } + zram_slot_unlock(zram, index); + + zstrm = zcomp_stream_get(zram->comp); + src = kmap_atomic(page); + ret = zcomp_compress(zstrm, src, &comp_len); + kunmap_atomic(src); + + if (unlikely(ret)) { + zcomp_stream_put(zram->comp); + pr_err("Compression failed! err=%d\n", ret); + zs_free(zram->mem_pool, handle); + return ret; + } + + if (comp_len >= huge_class_size) + comp_len = PAGE_SIZE; + + if (!handle) + handle = zs_malloc(zram->mem_pool, comp_len, + __GFP_KSWAPD_RECLAIM | + __GFP_NOWARN | + __GFP_HIGHMEM | + __GFP_MOVABLE | + __GFP_CMA); + if (!handle) { + zcomp_stream_put(zram->comp); + atomic64_inc(&zram->stats.writestall); + handle = zs_malloc(zram->mem_pool, comp_len, + GFP_NOIO | __GFP_HIGHMEM | + __GFP_MOVABLE | __GFP_CMA); + if (handle) + goto compress_again; + return -ENOMEM; + } + + alloced_pages = zs_get_total_pages(zram->mem_pool); + update_used_max(zram, alloced_pages); + + if (zram->limit_pages && alloced_pages > zram->limit_pages) { + zcomp_stream_put(zram->comp); + zs_free(zram->mem_pool, handle); + return -ENOMEM; + } + + dst = zs_map_object(zram->mem_pool, handle, ZS_MM_WO); + + src = zstrm->buffer; + if (comp_len == PAGE_SIZE) + src = kmap_atomic(page); + memcpy(dst, src, comp_len); + if (comp_len == PAGE_SIZE) + kunmap_atomic(src); + + zcomp_stream_put(zram->comp); + zs_unmap_object(zram->mem_pool, handle); + atomic64_add(comp_len, &zram->stats.compr_data_size); + + /* + * Free memory associated with this sector + * before overwriting unused sectors. + */ + zram_slot_lock(zram, index); + if (!zram_test_flag(zram, index, ZRAM_CACHED_COMPRESS)) { + atomic64_sub(comp_len, &zram->stats.compr_data_size); + zs_free(zram->mem_pool, handle); + zram_slot_unlock(zram, index); + return 0; + } + zram_free_page(zram, index); + + if (comp_len == PAGE_SIZE) { + zram_set_flag(zram, index, ZRAM_HUGE); + atomic64_inc(&zram->stats.huge_pages); + } + + zram_set_handle(zram, index, handle); + zram_set_obj_size(zram, index, comp_len); +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_record(zram, index, page->mem_cgroup); +#endif + zram_slot_unlock(zram, index); + + /* Update stats */ + atomic64_inc(&zram->stats.pages_stored); + + return ret; +} +#endif + +static int __zram_bvec_read(struct zram *zram, struct page *page, u32 index, + struct bio *bio, bool partial_io) +{ + struct zcomp_strm *zstrm; + unsigned long handle; + unsigned int size; + void *src, *dst; + int ret; + + zram_slot_lock(zram, index); + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if (akcompress_cache_page_fault(zram, page, index)) + return 0; +#endif + +#ifdef CONFIG_HYBRIDSWAP_CORE + if (likely(!bio)) { + ret = hybridswap_page_fault(zram, index); + if (unlikely(ret)) { + pr_err("search in hybridswap failed! err=%d, page=%u\n", + ret, index); + zram_slot_unlock(zram, index); + return ret; + } + } +#endif + + if (zram_test_flag(zram, index, ZRAM_WB)) { + struct bio_vec bvec; + + zram_slot_unlock(zram, index); + + bvec.bv_page = page; + bvec.bv_len = PAGE_SIZE; + bvec.bv_offset = 0; + return read_from_bdev(zram, &bvec, + zram_get_element(zram, index), + bio, partial_io); + } + + handle = zram_get_handle(zram, index); + if (!handle || zram_test_flag(zram, index, ZRAM_SAME)) { + unsigned long value; + void *mem; + + value = handle ? zram_get_element(zram, index) : 0; + mem = kmap_atomic(page); + zram_fill_page(mem, PAGE_SIZE, value); + kunmap_atomic(mem); + zram_slot_unlock(zram, index); + return 0; + } + + size = zram_get_obj_size(zram, index); + + if (size != PAGE_SIZE) + zstrm = zcomp_stream_get(zram->comp); + + src = zs_map_object(zram->mem_pool, handle, ZS_MM_RO); + if (size == PAGE_SIZE) { + dst = kmap_atomic(page); + memcpy(dst, src, PAGE_SIZE); + kunmap_atomic(dst); + ret = 0; + } else { + dst = kmap_atomic(page); + ret = zcomp_decompress(zstrm, src, size, dst); + kunmap_atomic(dst); + zcomp_stream_put(zram->comp); + } + zs_unmap_object(zram->mem_pool, handle); + zram_slot_unlock(zram, index); + + /* Should NEVER happen. Return bio error if it does. */ + if (WARN_ON(ret)) + pr_err("Decompression failed! err=%d, page=%u\n", ret, index); + + return ret; +} + +static int zram_bvec_read(struct zram *zram, struct bio_vec *bvec, + u32 index, int offset, struct bio *bio) +{ + int ret; + struct page *page; + + page = bvec->bv_page; + if (is_partial_io(bvec)) { + /* Use a temporary buffer to decompress the page */ + page = alloc_page(GFP_NOIO|__GFP_HIGHMEM); + if (!page) + return -ENOMEM; + } + + ret = __zram_bvec_read(zram, page, index, bio, is_partial_io(bvec)); + if (unlikely(ret)) + goto out; + + if (is_partial_io(bvec)) { + void *dst = kmap_atomic(bvec->bv_page); + void *src = kmap_atomic(page); + + memcpy(dst + bvec->bv_offset, src + offset, bvec->bv_len); + kunmap_atomic(src); + kunmap_atomic(dst); + } +out: + if (is_partial_io(bvec)) + __free_page(page); + + return ret; +} + +static int __zram_bvec_write(struct zram *zram, struct bio_vec *bvec, + u32 index, struct bio *bio) +{ + int ret = 0; + unsigned long alloced_pages; + unsigned long handle = 0; + unsigned int comp_len = 0; + void *src, *dst, *mem; + struct zcomp_strm *zstrm; + struct page *page = bvec->bv_page; + unsigned long element = 0; + enum zram_pageflags flags = 0; + + mem = kmap_atomic(page); + if (page_same_filled(mem, &element)) { + kunmap_atomic(mem); + /* Free memory associated with this sector now. */ + flags = ZRAM_SAME; + atomic64_inc(&zram->stats.same_pages); + goto out; + } + kunmap_atomic(mem); + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + if ((current_is_kswapd() || current_is_mswapd(current)) && + add_anon_page2cache(zram, index, page)) { + return 0; + } +#endif + +compress_again: + zstrm = zcomp_stream_get(zram->comp); + src = kmap_atomic(page); + ret = zcomp_compress(zstrm, src, &comp_len); + kunmap_atomic(src); + + if (unlikely(ret)) { + zcomp_stream_put(zram->comp); + pr_err("Compression failed! err=%d\n", ret); + zs_free(zram->mem_pool, handle); + return ret; + } + + if (comp_len >= huge_class_size) + comp_len = PAGE_SIZE; + /* + * handle allocation has 2 paths: + * a) fast path is executed with preemption disabled (for + * per-cpu streams) and has __GFP_DIRECT_RECLAIM bit clear, + * since we can't sleep; + * b) slow path enables preemption and attempts to allocate + * the page with __GFP_DIRECT_RECLAIM bit set. we have to + * put per-cpu compression stream and, thus, to re-do + * the compression once handle is allocated. + * + * if we have a 'non-null' handle here then we are coming + * from the slow path and handle has already been allocated. + */ + if (!handle) + handle = zs_malloc(zram->mem_pool, comp_len, + __GFP_KSWAPD_RECLAIM | + __GFP_NOWARN | + __GFP_HIGHMEM | + __GFP_MOVABLE | + __GFP_CMA | + __GFP_OFFLINABLE); // NOTE: __GFP_OFFLINABLE only for QCOM 5.4 kernel + if (!handle) { + zcomp_stream_put(zram->comp); + atomic64_inc(&zram->stats.writestall); + handle = zs_malloc(zram->mem_pool, comp_len, + GFP_NOIO | __GFP_HIGHMEM | + __GFP_MOVABLE | __GFP_CMA | + __GFP_OFFLINABLE); // NOTE: __GFP_OFFLINABLE only for QCOM 5.4 kernel + if (handle) + goto compress_again; + return -ENOMEM; + } + + alloced_pages = zs_get_total_pages(zram->mem_pool); + update_used_max(zram, alloced_pages); + + if (zram->limit_pages && alloced_pages > zram->limit_pages) { + zcomp_stream_put(zram->comp); + zs_free(zram->mem_pool, handle); + return -ENOMEM; + } + + dst = zs_map_object(zram->mem_pool, handle, ZS_MM_WO); + + src = zstrm->buffer; + if (comp_len == PAGE_SIZE) + src = kmap_atomic(page); + memcpy(dst, src, comp_len); + if (comp_len == PAGE_SIZE) + kunmap_atomic(src); + + zcomp_stream_put(zram->comp); + zs_unmap_object(zram->mem_pool, handle); + atomic64_add(comp_len, &zram->stats.compr_data_size); +out: + /* + * Free memory associated with this sector + * before overwriting unused sectors. + */ + zram_slot_lock(zram, index); + zram_free_page(zram, index); + + if (comp_len == PAGE_SIZE) { + zram_set_flag(zram, index, ZRAM_HUGE); + atomic64_inc(&zram->stats.huge_pages); + } + + if (flags) { + zram_set_flag(zram, index, flags); + zram_set_element(zram, index, element); + } else { + zram_set_handle(zram, index, handle); + zram_set_obj_size(zram, index, comp_len); + } + +#ifdef CONFIG_HYBRIDSWAP_CORE + hybridswap_record(zram, index, page->mem_cgroup); +#endif + zram_slot_unlock(zram, index); + + /* Update stats */ + atomic64_inc(&zram->stats.pages_stored); + return ret; +} + +static int zram_bvec_write(struct zram *zram, struct bio_vec *bvec, + u32 index, int offset, struct bio *bio) +{ + int ret; + struct page *page = NULL; + void *src; + struct bio_vec vec; + + vec = *bvec; + if (is_partial_io(bvec)) { + void *dst; + /* + * This is a partial IO. We need to read the full page + * before to write the changes. + */ + page = alloc_page(GFP_NOIO|__GFP_HIGHMEM); + if (!page) + return -ENOMEM; + + ret = __zram_bvec_read(zram, page, index, bio, true); + if (ret) + goto out; + + src = kmap_atomic(bvec->bv_page); + dst = kmap_atomic(page); + memcpy(dst + offset, src + bvec->bv_offset, bvec->bv_len); + kunmap_atomic(dst); + kunmap_atomic(src); + + vec.bv_page = page; + vec.bv_len = PAGE_SIZE; + vec.bv_offset = 0; + } + + ret = __zram_bvec_write(zram, &vec, index, bio); +out: + if (is_partial_io(bvec)) + __free_page(page); + return ret; +} + +/* + * zram_bio_discard - handler on discard request + * @index: physical block index in PAGE_SIZE units + * @offset: byte offset within physical block + */ +static void zram_bio_discard(struct zram *zram, u32 index, + int offset, struct bio *bio) +{ + size_t n = bio->bi_iter.bi_size; + + /* + * zram manages data in physical block size units. Because logical block + * size isn't identical with physical block size on some arch, we + * could get a discard request pointing to a specific offset within a + * certain physical block. Although we can handle this request by + * reading that physiclal block and decompressing and partially zeroing + * and re-compressing and then re-storing it, this isn't reasonable + * because our intent with a discard request is to save memory. So + * skipping this logical block is appropriate here. + */ + if (offset) { + if (n <= (PAGE_SIZE - offset)) + return; + + n -= (PAGE_SIZE - offset); + index++; + } + + while (n >= PAGE_SIZE) { + zram_slot_lock(zram, index); + zram_free_page(zram, index); + zram_slot_unlock(zram, index); + atomic64_inc(&zram->stats.notify_free); + index++; + n -= PAGE_SIZE; + } +} + +/* + * Returns errno if it has some problem. Otherwise return 0 or 1. + * Returns 0 if IO request was done synchronously + * Returns 1 if IO request was successfully submitted. + */ +static int zram_bvec_rw(struct zram *zram, struct bio_vec *bvec, u32 index, + int offset, unsigned int op, struct bio *bio) +{ + unsigned long start_time = jiffies; + struct request_queue *q = zram->disk->queue; + int ret; + + generic_start_io_acct(q, op, bvec->bv_len >> SECTOR_SHIFT, + &zram->disk->part0); + + if (!op_is_write(op)) { + atomic64_inc(&zram->stats.num_reads); + ret = zram_bvec_read(zram, bvec, index, offset, bio); + flush_dcache_page(bvec->bv_page); + } else { + atomic64_inc(&zram->stats.num_writes); + ret = zram_bvec_write(zram, bvec, index, offset, bio); + } + + generic_end_io_acct(q, op, &zram->disk->part0, start_time); + + zram_slot_lock(zram, index); + zram_accessed(zram, index); + zram_slot_unlock(zram, index); + + if (unlikely(ret < 0)) { + if (!op_is_write(op)) + atomic64_inc(&zram->stats.failed_reads); + else + atomic64_inc(&zram->stats.failed_writes); + } + + return ret; +} + +static void __zram_make_request(struct zram *zram, struct bio *bio) +{ + int offset; + u32 index; + struct bio_vec bvec; + struct bvec_iter iter; + + index = bio->bi_iter.bi_sector >> SECTORS_PER_PAGE_SHIFT; + offset = (bio->bi_iter.bi_sector & + (SECTORS_PER_PAGE - 1)) << SECTOR_SHIFT; + + switch (bio_op(bio)) { + case REQ_OP_DISCARD: + case REQ_OP_WRITE_ZEROES: + zram_bio_discard(zram, index, offset, bio); + bio_endio(bio); + return; + default: + break; + } + + bio_for_each_segment(bvec, bio, iter) { + struct bio_vec bv = bvec; + unsigned int unwritten = bvec.bv_len; + + do { + bv.bv_len = min_t(unsigned int, PAGE_SIZE - offset, + unwritten); + if (zram_bvec_rw(zram, &bv, index, offset, + bio_op(bio), bio) < 0) + goto out; + + bv.bv_offset += bv.bv_len; + unwritten -= bv.bv_len; + + update_position(&index, &offset, &bv); + } while (unwritten); + } + + bio_endio(bio); + return; + +out: + bio_io_error(bio); +} + +/* + * Handler function for all zram I/O requests. + */ +static blk_qc_t zram_make_request(struct request_queue *queue, struct bio *bio) +{ + struct zram *zram = queue->queuedata; + + if (!valid_io_request(zram, bio->bi_iter.bi_sector, + bio->bi_iter.bi_size)) { + atomic64_inc(&zram->stats.invalid_io); + goto error; + } + + __zram_make_request(zram, bio); + return BLK_QC_T_NONE; + +error: + bio_io_error(bio); + return BLK_QC_T_NONE; +} + +static void zram_slot_free_notify(struct block_device *bdev, + unsigned long index) +{ + struct zram *zram; + + zram = bdev->bd_disk->private_data; + + atomic64_inc(&zram->stats.notify_free); + if (!zram_slot_trylock(zram, index)) { + atomic64_inc(&zram->stats.miss_free); + return; + } + +#ifdef CONFIG_HYBRIDSWAP_CORE + if (!hybridswap_delete(zram, index)) { + zram_slot_unlock(zram, index); + atomic64_inc(&zram->stats.miss_free); + return; + } +#endif + zram_free_page(zram, index); + zram_slot_unlock(zram, index); +} + +static int zram_rw_page(struct block_device *bdev, sector_t sector, + struct page *page, unsigned int op) +{ + int offset, ret; + u32 index; + struct zram *zram; + struct bio_vec bv; + + if (PageTransHuge(page)) + return -ENOTSUPP; + zram = bdev->bd_disk->private_data; + + if (!valid_io_request(zram, sector, PAGE_SIZE)) { + atomic64_inc(&zram->stats.invalid_io); + ret = -EINVAL; + goto out; + } + + index = sector >> SECTORS_PER_PAGE_SHIFT; + offset = (sector & (SECTORS_PER_PAGE - 1)) << SECTOR_SHIFT; + + bv.bv_page = page; + bv.bv_len = PAGE_SIZE; + bv.bv_offset = 0; + + ret = zram_bvec_rw(zram, &bv, index, offset, op, NULL); +out: + /* + * If I/O fails, just return error(ie, non-zero) without + * calling page_endio. + * It causes resubmit the I/O with bio request by upper functions + * of rw_page(e.g., swap_readpage, __swap_writepage) and + * bio->bi_end_io does things to handle the error + * (e.g., SetPageError, set_page_dirty and extra works). + */ + if (unlikely(ret < 0)) + return ret; + + switch (ret) { + case 0: + page_endio(page, op_is_write(op), 0); + break; + case 1: + ret = 0; + break; + default: + WARN_ON(1); + } + return ret; +} + +static void zram_reset_device(struct zram *zram) +{ + struct zcomp *comp; + u64 disksize; + + down_write(&zram->init_lock); + + zram->limit_pages = 0; + + if (!init_done(zram)) { + up_write(&zram->init_lock); + return; + } + + comp = zram->comp; + disksize = zram->disksize; + zram->disksize = 0; + + set_capacity(zram->disk, 0); + part_stat_set_all(&zram->disk->part0, 0); + + up_write(&zram->init_lock); + /* I/O operation under all of CPU are done so let's free */ + zram_meta_free(zram, disksize); + memset(&zram->stats, 0, sizeof(zram->stats)); + zcomp_destroy(comp); + reset_bdev(zram); +} + +static ssize_t disksize_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + u64 disksize; + struct zcomp *comp; + struct zram *zram = dev_to_zram(dev); + int err; + + disksize = memparse(buf, NULL); + if (!disksize) + return -EINVAL; + + down_write(&zram->init_lock); + if (init_done(zram)) { + pr_info("Cannot change disksize for initialized device\n"); + err = -EBUSY; + goto out_unlock; + } + + disksize = PAGE_ALIGN(disksize); + if (!zram_meta_alloc(zram, disksize)) { + err = -ENOMEM; + goto out_unlock; + } + + comp = zcomp_create(zram->compressor); + if (IS_ERR(comp)) { + pr_err("Cannot initialise %s compressing backend\n", + zram->compressor); + err = PTR_ERR(comp); + goto out_free_meta; + } + + zram->comp = comp; + zram->disksize = disksize; + set_capacity(zram->disk, zram->disksize >> SECTOR_SHIFT); + + revalidate_disk(zram->disk); + up_write(&zram->init_lock); + + return len; + +out_free_meta: + zram_meta_free(zram, disksize); +out_unlock: + up_write(&zram->init_lock); + return err; +} + +static ssize_t reset_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t len) +{ + int ret; + unsigned short do_reset; + struct zram *zram; + struct block_device *bdev; + + ret = kstrtou16(buf, 10, &do_reset); + if (ret) + return ret; + + if (!do_reset) + return -EINVAL; + + zram = dev_to_zram(dev); + bdev = bdget_disk(zram->disk, 0); + if (!bdev) + return -ENOMEM; + + mutex_lock(&bdev->bd_mutex); + /* Do not reset an active device or claimed device */ + if (bdev->bd_openers || zram->claim) { + mutex_unlock(&bdev->bd_mutex); + bdput(bdev); + return -EBUSY; + } + + /* From now on, anyone can't open /dev/zram[0-9] */ + zram->claim = true; + mutex_unlock(&bdev->bd_mutex); + + /* Make sure all the pending I/O are finished */ + fsync_bdev(bdev); + zram_reset_device(zram); + revalidate_disk(zram->disk); + bdput(bdev); + + mutex_lock(&bdev->bd_mutex); + zram->claim = false; + mutex_unlock(&bdev->bd_mutex); + + return len; +} + +static int zram_open(struct block_device *bdev, fmode_t mode) +{ + int ret = 0; + struct zram *zram; + + WARN_ON(!mutex_is_locked(&bdev->bd_mutex)); + + zram = bdev->bd_disk->private_data; + /* zram was claimed to reset so open request fails */ + if (zram->claim) + ret = -EBUSY; + + return ret; +} + +static const struct block_device_operations zram_devops = { + .open = zram_open, + .swap_slot_free_notify = zram_slot_free_notify, + .rw_page = zram_rw_page, + .owner = THIS_MODULE +}; + +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static const struct block_device_operations zram_wb_devops = { + .open = zram_open, + .submit_bio = zram_submit_bio, + .swap_slot_free_notify = zram_slot_free_notify, + .owner = THIS_MODULE +}; +#endif + +static DEVICE_ATTR_WO(compact); +static DEVICE_ATTR_RW(disksize); +static DEVICE_ATTR_RO(initstate); +static DEVICE_ATTR_WO(reset); +static DEVICE_ATTR_WO(mem_limit); +static DEVICE_ATTR_WO(mem_used_max); +static DEVICE_ATTR_WO(idle); +static DEVICE_ATTR_RW(max_comp_streams); +static DEVICE_ATTR_RW(comp_algorithm); +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK +static DEVICE_ATTR_RW(backing_dev); +static DEVICE_ATTR_WO(writeback); +static DEVICE_ATTR_RW(writeback_limit); +static DEVICE_ATTR_RW(writeback_limit_enable); +#endif +#ifdef CONFIG_HYBRIDSWAP +static DEVICE_ATTR_RO(hybridswap_vmstat); +static DEVICE_ATTR_RW(hybridswap_loglevel); +static DEVICE_ATTR_RW(hybridswap_enable); +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD +static DEVICE_ATTR_RW(hybridswap_swapd_pause); +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE +static DEVICE_ATTR_RW(hybridswap_core_enable); +static DEVICE_ATTR_RW(hybridswap_loop_device); +static DEVICE_ATTR_RW(hybridswap_dev_life); +static DEVICE_ATTR_RW(hybridswap_quota_day); +static DEVICE_ATTR_RO(hybridswap_report); +static DEVICE_ATTR_RO(hybridswap_stat_snap); +static DEVICE_ATTR_RO(hybridswap_meminfo); +static DEVICE_ATTR_RW(hybridswap_zram_increase); +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +static DEVICE_ATTR_RW(hybridswap_akcompress); +#endif + +static struct attribute *zram_disk_attrs[] = { + &dev_attr_disksize.attr, + &dev_attr_initstate.attr, + &dev_attr_reset.attr, + &dev_attr_compact.attr, + &dev_attr_mem_limit.attr, + &dev_attr_mem_used_max.attr, + &dev_attr_idle.attr, + &dev_attr_max_comp_streams.attr, + &dev_attr_comp_algorithm.attr, +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + &dev_attr_backing_dev.attr, + &dev_attr_writeback.attr, + &dev_attr_writeback_limit.attr, + &dev_attr_writeback_limit_enable.attr, +#endif + &dev_attr_io_stat.attr, + &dev_attr_mm_stat.attr, +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + &dev_attr_bd_stat.attr, +#endif + &dev_attr_debug_stat.attr, +#ifdef CONFIG_HYBRIDSWAP + &dev_attr_hybridswap_vmstat.attr, + &dev_attr_hybridswap_loglevel.attr, + &dev_attr_hybridswap_enable.attr, +#endif +#ifdef CONFIG_HYBRIDSWAP_SWAPD + &dev_attr_hybridswap_swapd_pause.attr, +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + &dev_attr_hybridswap_core_enable.attr, + &dev_attr_hybridswap_report.attr, + &dev_attr_hybridswap_meminfo.attr, + &dev_attr_hybridswap_stat_snap.attr, + &dev_attr_hybridswap_loop_device.attr, + &dev_attr_hybridswap_dev_life.attr, + &dev_attr_hybridswap_quota_day.attr, + &dev_attr_hybridswap_zram_increase.attr, +#endif +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + &dev_attr_hybridswap_akcompress.attr, +#endif + NULL, +}; + +static const struct attribute_group zram_disk_attr_group = { + .attrs = zram_disk_attrs, +}; + +static const struct attribute_group *zram_disk_attr_groups[] = { + &zram_disk_attr_group, + NULL, +}; + +/* + * Allocate and initialize new zram device. the function returns + * '>= 0' device_id upon success, and negative value otherwise. + */ +static int zram_add(void) +{ + struct zram *zram; + struct request_queue *queue; + int ret, device_id; + + zram = kzalloc(sizeof(struct zram), GFP_KERNEL); + if (!zram) + return -ENOMEM; + + ret = idr_alloc(&zram_index_idr, zram, 0, 0, GFP_KERNEL); + if (ret < 0) + goto out_free_dev; + device_id = ret; + + init_rwsem(&zram->init_lock); +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + spin_lock_init(&zram->wb_limit_lock); +#endif + queue = blk_alloc_queue(GFP_KERNEL); + if (!queue) { + pr_err("Error allocating disk queue for device %d\n", + device_id); + ret = -ENOMEM; + goto out_free_idr; + } + + blk_queue_make_request(queue, zram_make_request); + + /* gendisk structure */ + zram->disk = alloc_disk(1); + if (!zram->disk) { + pr_err("Error allocating disk structure for device %d\n", + device_id); + ret = -ENOMEM; + goto out_free_queue; + } + + zram->disk->major = zram_major; + zram->disk->first_minor = device_id; + zram->disk->fops = &zram_devops; + zram->disk->queue = queue; + zram->disk->queue->queuedata = zram; + zram->disk->private_data = zram; + snprintf(zram->disk->disk_name, 16, "zram%d", device_id); + + /* Actual capacity set using syfs (/sys/block/zram/disksize */ + set_capacity(zram->disk, 0); + /* zram devices sort of resembles non-rotational disks */ + blk_queue_flag_set(QUEUE_FLAG_NONROT, zram->disk->queue); + blk_queue_flag_clear(QUEUE_FLAG_ADD_RANDOM, zram->disk->queue); + + /* + * To ensure that we always get PAGE_SIZE aligned + * and n*PAGE_SIZED sized I/O requests. + */ + blk_queue_physical_block_size(zram->disk->queue, PAGE_SIZE); + blk_queue_logical_block_size(zram->disk->queue, + ZRAM_LOGICAL_BLOCK_SIZE); + blk_queue_io_min(zram->disk->queue, PAGE_SIZE); + blk_queue_io_opt(zram->disk->queue, PAGE_SIZE); + zram->disk->queue->limits.discard_granularity = PAGE_SIZE; + blk_queue_max_discard_sectors(zram->disk->queue, UINT_MAX); + blk_queue_flag_set(QUEUE_FLAG_DISCARD, zram->disk->queue); + + /* + * zram_bio_discard() will clear all logical blocks if logical block + * size is identical with physical block size(PAGE_SIZE). But if it is + * different, we will skip discarding some parts of logical blocks in + * the part of the request range which isn't aligned to physical block + * size. So we can't ensure that all discarded logical blocks are + * zeroed. + */ + if (ZRAM_LOGICAL_BLOCK_SIZE == PAGE_SIZE) + blk_queue_max_write_zeroes_sectors(zram->disk->queue, UINT_MAX); + + zram->disk->queue->backing_dev_info->capabilities |= + (BDI_CAP_STABLE_WRITES | BDI_CAP_SYNCHRONOUS_IO); + device_add_disk(NULL, zram->disk, zram_disk_attr_groups); + + strlcpy(zram->compressor, default_compressor, sizeof(zram->compressor)); + + zram_debugfs_register(zram); + pr_info("Added device: %s\n", zram->disk->disk_name); + return device_id; + +out_free_queue: + blk_cleanup_queue(queue); +out_free_idr: + idr_remove(&zram_index_idr, device_id); +out_free_dev: + kfree(zram); + return ret; +} + +static int zram_remove(struct zram *zram) +{ + struct block_device *bdev; + + bdev = bdget_disk(zram->disk, 0); + if (!bdev) + return -ENOMEM; + + mutex_lock(&bdev->bd_mutex); + if (bdev->bd_openers || zram->claim) { + mutex_unlock(&bdev->bd_mutex); + bdput(bdev); + return -EBUSY; + } + + zram->claim = true; + mutex_unlock(&bdev->bd_mutex); + + zram_debugfs_unregister(zram); + + /* Make sure all the pending I/O are finished */ + fsync_bdev(bdev); + zram_reset_device(zram); + bdput(bdev); + + pr_info("Removed device: %s\n", zram->disk->disk_name); + + del_gendisk(zram->disk); + blk_cleanup_queue(zram->disk->queue); + put_disk(zram->disk); + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + destroy_akcompressd_task(zram); +#endif + kfree(zram); + return 0; +} + +/* zram-control sysfs attributes */ + +/* + * NOTE: hot_add attribute is not the usual read-only sysfs attribute. In a + * sense that reading from this file does alter the state of your system -- it + * creates a new un-initialized zram device and returns back this device's + * device_id (or an error code if it fails to create a new device). + */ +static ssize_t hot_add_show(struct class *class, + struct class_attribute *attr, + char *buf) +{ + int ret; + + mutex_lock(&zram_index_mutex); + ret = zram_add(); + mutex_unlock(&zram_index_mutex); + + if (ret < 0) + return ret; + return scnprintf(buf, PAGE_SIZE, "%d\n", ret); +} +static struct class_attribute class_attr_hot_add = + __ATTR(hot_add, 0400, hot_add_show, NULL); + +static ssize_t hot_remove_store(struct class *class, + struct class_attribute *attr, + const char *buf, + size_t count) +{ + struct zram *zram; + int ret, dev_id; + + /* dev_id is gendisk->first_minor, which is `int' */ + ret = kstrtoint(buf, 10, &dev_id); + if (ret) + return ret; + if (dev_id < 0) + return -EINVAL; + + mutex_lock(&zram_index_mutex); + + zram = idr_find(&zram_index_idr, dev_id); + if (zram) { + ret = zram_remove(zram); + if (!ret) + idr_remove(&zram_index_idr, dev_id); + } else { + ret = -ENODEV; + } + + mutex_unlock(&zram_index_mutex); + return ret ? ret : count; +} +static CLASS_ATTR_WO(hot_remove); + +static struct attribute *zram_control_class_attrs[] = { + &class_attr_hot_add.attr, + &class_attr_hot_remove.attr, + NULL, +}; +ATTRIBUTE_GROUPS(zram_control_class); + +static struct class zram_control_class = { + .name = "zram-control", + .owner = THIS_MODULE, + .class_groups = zram_control_class_groups, +}; + +static int zram_remove_cb(int id, void *ptr, void *data) +{ + zram_remove(ptr); + return 0; +} + +static void destroy_devices(void) +{ + class_unregister(&zram_control_class); + idr_for_each(&zram_index_idr, &zram_remove_cb, NULL); + zram_debugfs_destroy(); + idr_destroy(&zram_index_idr); + unregister_blkdev(zram_major, "zram"); + cpuhp_remove_multi_state(CPUHP_ZCOMP_PREPARE); +} + +static int __init zram_init(void) +{ + int ret; + + ret = cpuhp_setup_state_multi(CPUHP_ZCOMP_PREPARE, "block/zram:prepare", + zcomp_cpu_up_prepare, zcomp_cpu_dead); + if (ret < 0) + return ret; + + ret = class_register(&zram_control_class); + if (ret) { + pr_err("Unable to register zram-control class\n"); + cpuhp_remove_multi_state(CPUHP_ZCOMP_PREPARE); + return ret; + } + + zram_debugfs_create(); + zram_major = register_blkdev(0, "zram"); + if (zram_major <= 0) { + pr_err("Unable to get major number\n"); + class_unregister(&zram_control_class); + cpuhp_remove_multi_state(CPUHP_ZCOMP_PREPARE); + return -EBUSY; + } + + while (num_devices != 0) { + mutex_lock(&zram_index_mutex); + ret = zram_add(); + mutex_unlock(&zram_index_mutex); + if (ret < 0) + goto out_error; + num_devices--; + } + +#ifdef CONFIG_HYBRIDSWAP + ret = hybridswap_pre_init(); + if (ret) + goto out_error; +#endif + return 0; + +out_error: + destroy_devices(); + return ret; +} + +static void __exit zram_exit(void) +{ + destroy_devices(); +} + +module_init(zram_init); +module_exit(zram_exit); + +module_param(num_devices, uint, 0); +MODULE_PARM_DESC(num_devices, "Number of pre-created zram devices"); + +MODULE_LICENSE("Dual BSD/GPL"); +MODULE_AUTHOR("Nitin Gupta "); +MODULE_DESCRIPTION("Compressed RAM Block Device"); diff --git a/drivers/moto_swap/zram-5.4/zram_drv.h b/drivers/moto_swap/zram-5.4/zram_drv.h new file mode 100644 index 000000000000..7e8e5ab8c148 --- /dev/null +++ b/drivers/moto_swap/zram-5.4/zram_drv.h @@ -0,0 +1,150 @@ +/* + * Compressed RAM block device + * + * Copyright (C) 2008, 2009, 2010 Nitin Gupta + * 2012, 2013 Minchan Kim + * + * This code is released using a dual license strategy: BSD/GPL + * You can choose the licence that better fits your requirements. + * + * Released under the terms of 3-clause BSD License + * Released under the terms of GNU General Public License Version 2.0 + * + */ + +#ifndef _ZRAM_DRV_H_ +#define _ZRAM_DRV_H_ + +#include +#include +#include + +#include "zcomp.h" + +#define SECTORS_PER_PAGE_SHIFT (PAGE_SHIFT - SECTOR_SHIFT) +#define SECTORS_PER_PAGE (1 << SECTORS_PER_PAGE_SHIFT) +#define ZRAM_LOGICAL_BLOCK_SHIFT 12 +#define ZRAM_LOGICAL_BLOCK_SIZE (1 << ZRAM_LOGICAL_BLOCK_SHIFT) +#define ZRAM_SECTOR_PER_LOGICAL_BLOCK \ + (1 << (ZRAM_LOGICAL_BLOCK_SHIFT - SECTOR_SHIFT)) + + +/* + * The lower ZRAM_FLAG_SHIFT bits of table.flags is for + * object size (excluding header), the higher bits is for + * zram_pageflags. + * + * zram is mainly used for memory efficiency so we want to keep memory + * footprint small so we can squeeze size and flags into a field. + * The lower ZRAM_FLAG_SHIFT bits is for object size (excluding header), + * the higher bits is for zram_pageflags. + */ +#define ZRAM_FLAG_SHIFT 24 + +/* Flags for zram pages (table[page_no].flags) */ +enum zram_pageflags { + /* zram slot is locked */ + ZRAM_LOCK = ZRAM_FLAG_SHIFT, + ZRAM_SAME, /* Page consists the same element */ + ZRAM_WB, /* page is stored on backing_device */ + ZRAM_UNDER_WB, /* page is under writeback */ + ZRAM_HUGE, /* Incompressible page */ + ZRAM_IDLE, /* not accessed page since last idle marking */ +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + ZRAM_CACHED, /* page is cached in async compress cache buffer */ + ZRAM_CACHED_COMPRESS, /* page is under async compress */ +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + ZRAM_BATCHING_OUT, + ZRAM_FROM_HYBRIDSWAP, + ZRAM_MCGID_CLEAR, + ZRAM_IN_BD, /* zram stored in back device */ +#endif + __NR_ZRAM_PAGEFLAGS, +}; + +/*-- Data structures */ + +/* Allocated for each disk page */ +struct zram_table_entry { + union { + unsigned long handle; + unsigned long element; +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS + unsigned long page; +#endif + }; + unsigned long flags; +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + ktime_t ac_time; +#endif +}; + +struct zram_stats { + atomic64_t compr_data_size; /* compressed size of pages stored */ + atomic64_t num_reads; /* failed + successful */ + atomic64_t num_writes; /* --do-- */ + atomic64_t failed_reads; /* can happen when memory is too low */ + atomic64_t failed_writes; /* can happen when memory is too low */ + atomic64_t invalid_io; /* non-page-aligned I/O requests */ + atomic64_t notify_free; /* no. of swap slot free notifications */ + atomic64_t same_pages; /* no. of same element filled pages */ + atomic64_t huge_pages; /* no. of huge pages */ + atomic64_t pages_stored; /* no. of pages currently stored */ + atomic_long_t max_used_pages; /* no. of maximum pages stored */ + atomic64_t writestall; /* no. of write slow paths */ + atomic64_t miss_free; /* no. of missed free */ +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + atomic64_t bd_count; /* no. of pages in backing device */ + atomic64_t bd_reads; /* no. of reads from backing device */ + atomic64_t bd_writes; /* no. of writes from backing device */ +#endif +}; + +struct zram { + struct zram_table_entry *table; + struct zs_pool *mem_pool; + struct zcomp *comp; + struct gendisk *disk; + /* Prevent concurrent execution of device init */ + struct rw_semaphore init_lock; + /* + * the number of pages zram can consume for storing compressed data + */ + unsigned long limit_pages; + + struct zram_stats stats; + /* + * This is the limit on amount of *uncompressed* worth of data + * we can store in a disk. + */ + u64 disksize; /* bytes */ + char compressor[CRYPTO_MAX_ALG_NAME]; + /* + * zram is claimed so open request will be failed + */ + bool claim; /* Protected by bdev->bd_mutex */ + struct file *backing_dev; +#ifdef CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK + spinlock_t wb_limit_lock; + bool wb_limit_enable; + u64 bd_wb_limit; + struct block_device *bdev; + unsigned int old_block_size; + unsigned long *bitmap; + unsigned long nr_pages; +#endif +#ifdef CONFIG_HYBRIDSWAP_ZRAM_MEMORY_TRACKING + struct dentry *debugfs_dir; +#endif +#if (defined CONFIG_HYBRIDSWAP_ZRAM_WRITEBACK) || (defined CONFIG_HYBRIDSWAP_CORE) + struct block_device *bdev; + unsigned int old_block_size; + unsigned long nr_pages; + unsigned long increase_nr_pages; +#endif +#ifdef CONFIG_HYBRIDSWAP_CORE + struct hyb_info *infos; +#endif +}; +#endif diff --git a/drivers/moto_swap/zram-5.4/zram_drv_internal.h b/drivers/moto_swap/zram-5.4/zram_drv_internal.h new file mode 100644 index 000000000000..3c102cf38773 --- /dev/null +++ b/drivers/moto_swap/zram-5.4/zram_drv_internal.h @@ -0,0 +1,39 @@ +#ifndef _ZRAM_DRV_INTERNAL_H_ +#define _ZRAM_DRV_INTERNAL_H_ +#ifdef BIT +#undef BIT +#define BIT(nr) (1lu << (nr)) +#endif + +#define zram_slot_lock(zram, index) (bit_spin_lock(ZRAM_LOCK, &zram->table[index].flags)) + +#define zram_slot_unlock(zram, index) (bit_spin_unlock(ZRAM_LOCK, &zram->table[index].flags)) + +#define init_done(zram) (zram->disksize) + +#define dev_to_zram(dev) ((struct zram *)dev_to_disk(dev)->private_data) + +#define zram_get_handle(zram, index) (zram->table[index].handle) + +#define zram_set_handle(zram, index, handle_val) (zram->table[index].handle = handle_val) + +#define zram_test_flag(zram, index, flag) (zram->table[index].flags & BIT(flag)) + +#define zram_set_flag(zram, index, flag) (zram->table[index].flags |= BIT(flag)) + +#define zram_clear_flag(zram, index, flag) (zram->table[index].flags &= ~BIT(flag)) + +#define zram_set_element(zram, index, element) (zram->table[index].element = element) + +#define zram_get_obj_size(zram, index) (zram->table[index].flags & (BIT(ZRAM_FLAG_SHIFT) - 1)) + +#define zram_set_obj_size(zram, index, size) do {\ + unsigned long flags = zram->table[index].flags >> ZRAM_FLAG_SHIFT; \ + zram->table[index].flags = (flags << ZRAM_FLAG_SHIFT) | size; \ +} while(0) + +#ifdef CONFIG_HYBRIDSWAP_ASYNC_COMPRESS +extern int async_compress_page(struct zram *zram, struct page* page); +extern void update_zram_index(struct zram *zram, u32 index, unsigned long page); +#endif +#endif