1163 lines
38 KiB
Diff
1163 lines
38 KiB
Diff
diff --git a/drivers/block/zram/backend_842.c b/drivers/block/zram/backend_842.c
|
|
--- a/drivers/block/zram/backend_842.c
|
|
+++ b/drivers/block/zram/backend_842.c
|
|
@@ -1,5 +1,7 @@
|
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
|
|
+#define pr_fmt(fmt) "842: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/slab.h>
|
|
#include <linux/sw842.h>
|
|
@@ -13,6 +15,14 @@ static void release_params_842(struct zcomp_params *params)
|
|
|
|
static int setup_params_842(struct zcomp_params *params)
|
|
{
|
|
+ if (params->dict_sz) {
|
|
+ pr_err("dictionary is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
+ if (params->level != ZCOMP_PARAM_NOT_SET) {
|
|
+ pr_err("compression level is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
return 0;
|
|
}
|
|
|
|
diff --git a/drivers/block/zram/backend_deflate.c b/drivers/block/zram/backend_deflate.c
|
|
--- a/drivers/block/zram/backend_deflate.c
|
|
+++ b/drivers/block/zram/backend_deflate.c
|
|
@@ -1,5 +1,7 @@
|
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
|
|
+#define pr_fmt(fmt) "deflate: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/slab.h>
|
|
#include <linux/vmalloc.h>
|
|
@@ -22,15 +24,26 @@ static void deflate_release_params(struct zcomp_params *params)
|
|
|
|
static int deflate_setup_params(struct zcomp_params *params)
|
|
{
|
|
- if (params->level == ZCOMP_PARAM_NOT_SET)
|
|
+ if (params->dict_sz) {
|
|
+ pr_err("dictionary is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
+
|
|
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
|
|
params->level = Z_DEFAULT_COMPRESSION;
|
|
+ } else if (params->level < Z_DEFAULT_COMPRESSION ||
|
|
+ params->level > Z_BEST_COMPRESSION) {
|
|
+ pr_err("invalid compression level %d\n", params->level);
|
|
+ return -EINVAL;
|
|
+ }
|
|
+
|
|
if (params->deflate.winbits == ZCOMP_PARAM_NOT_SET) {
|
|
params->deflate.winbits = DEFLATE_DEF_WINBITS;
|
|
} else {
|
|
s32 wb = params->deflate.winbits;
|
|
|
|
if ((wb < -15 || wb > -9) && (wb < 9 || wb > 15)) {
|
|
- pr_err("invalid deflate winbits: %d\n", wb);
|
|
+ pr_err("invalid winbits %d\n", wb);
|
|
return -EINVAL;
|
|
}
|
|
}
|
|
diff --git a/drivers/block/zram/backend_lz4.c b/drivers/block/zram/backend_lz4.c
|
|
--- a/drivers/block/zram/backend_lz4.c
|
|
+++ b/drivers/block/zram/backend_lz4.c
|
|
@@ -1,3 +1,7 @@
|
|
+// SPDX-License-Identifier: GPL-2.0-or-later
|
|
+
|
|
+#define pr_fmt(fmt) "lz4: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/lz4.h>
|
|
#include <linux/slab.h>
|
|
@@ -28,8 +32,12 @@ static int lz4_setup_params(struct zcomp_params *params)
|
|
LZ4_stream_t *dict_stream;
|
|
int ret;
|
|
|
|
- if (params->level == ZCOMP_PARAM_NOT_SET)
|
|
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
|
|
params->level = LZ4_ACCELERATION_DEFAULT;
|
|
+ } else if (params->level < LZ4_ACCELERATION_DEFAULT) {
|
|
+ pr_err("invalid compression level %d\n", params->level);
|
|
+ return -EINVAL;
|
|
+ }
|
|
|
|
if (!params->dict || !params->dict_sz)
|
|
return 0;
|
|
diff --git a/drivers/block/zram/backend_lz4hc.c b/drivers/block/zram/backend_lz4hc.c
|
|
--- a/drivers/block/zram/backend_lz4hc.c
|
|
+++ b/drivers/block/zram/backend_lz4hc.c
|
|
@@ -1,3 +1,7 @@
|
|
+// SPDX-License-Identifier: GPL-2.0-or-later
|
|
+
|
|
+#define pr_fmt(fmt) "lz4hc: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/lz4.h>
|
|
#include <linux/slab.h>
|
|
@@ -18,8 +22,18 @@ static void lz4hc_release_params(struct zcomp_params *params)
|
|
|
|
static int lz4hc_setup_params(struct zcomp_params *params)
|
|
{
|
|
- if (params->level == ZCOMP_PARAM_NOT_SET)
|
|
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
|
|
params->level = LZ4HC_DEFAULT_CLEVEL;
|
|
+ } else if (params->level < 1 || params->level > LZ4HC_MAX_CLEVEL) {
|
|
+ /*
|
|
+ * Use < 1 rather than < LZ4HC_MIN_CLEVEL here because
|
|
+ * LZ4HC_compress_generic() only clamps levels below 1
|
|
+ * (levels 1 and 2 are valid). LZ4HC_MIN_CLEVEL (3) is
|
|
+ * advisory and not enforced by the library.
|
|
+ */
|
|
+ pr_err("invalid compression level %d\n", params->level);
|
|
+ return -EINVAL;
|
|
+ }
|
|
|
|
return 0;
|
|
}
|
|
diff --git a/drivers/block/zram/backend_lzo.c b/drivers/block/zram/backend_lzo.c
|
|
--- a/drivers/block/zram/backend_lzo.c
|
|
+++ b/drivers/block/zram/backend_lzo.c
|
|
@@ -1,5 +1,7 @@
|
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
|
|
+#define pr_fmt(fmt) "lzo: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/slab.h>
|
|
#include <linux/lzo.h>
|
|
@@ -12,6 +14,14 @@ static void lzo_release_params(struct zcomp_params *params)
|
|
|
|
static int lzo_setup_params(struct zcomp_params *params)
|
|
{
|
|
+ if (params->dict_sz) {
|
|
+ pr_err("dictionary is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
+ if (params->level != ZCOMP_PARAM_NOT_SET) {
|
|
+ pr_err("compression level is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
return 0;
|
|
}
|
|
|
|
diff --git a/drivers/block/zram/backend_lzorle.c b/drivers/block/zram/backend_lzorle.c
|
|
--- a/drivers/block/zram/backend_lzorle.c
|
|
+++ b/drivers/block/zram/backend_lzorle.c
|
|
@@ -1,5 +1,7 @@
|
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
|
|
+#define pr_fmt(fmt) "lzo-rle: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/slab.h>
|
|
#include <linux/lzo.h>
|
|
@@ -12,6 +14,14 @@ static void lzorle_release_params(struct zcomp_params *params)
|
|
|
|
static int lzorle_setup_params(struct zcomp_params *params)
|
|
{
|
|
+ if (params->dict_sz) {
|
|
+ pr_err("dictionary is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
+ if (params->level != ZCOMP_PARAM_NOT_SET) {
|
|
+ pr_err("compression level is not supported\n");
|
|
+ return -EOPNOTSUPP;
|
|
+ }
|
|
return 0;
|
|
}
|
|
|
|
diff --git a/drivers/block/zram/backend_zstd.c b/drivers/block/zram/backend_zstd.c
|
|
--- a/drivers/block/zram/backend_zstd.c
|
|
+++ b/drivers/block/zram/backend_zstd.c
|
|
@@ -1,5 +1,7 @@
|
|
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
|
|
+#define pr_fmt(fmt) "zstd: " fmt
|
|
+
|
|
#include <linux/kernel.h>
|
|
#include <linux/slab.h>
|
|
#include <linux/vmalloc.h>
|
|
@@ -58,8 +60,13 @@ static int zstd_setup_params(struct zcomp_params *params)
|
|
return -ENOMEM;
|
|
|
|
params->drv_data = zp;
|
|
- if (params->level == ZCOMP_PARAM_NOT_SET)
|
|
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
|
|
params->level = zstd_default_clevel();
|
|
+ } else if (params->level < zstd_min_clevel() ||
|
|
+ params->level > zstd_max_clevel()) {
|
|
+ pr_err("invalid compression level %d\n", params->level);
|
|
+ goto error;
|
|
+ }
|
|
|
|
zp->cprm = zstd_get_params(params->level, PAGE_SIZE);
|
|
|
|
@@ -85,7 +92,6 @@ static int zstd_setup_params(struct zcomp_params *params)
|
|
return 0;
|
|
|
|
error:
|
|
- zstd_release_params(params);
|
|
return -EINVAL;
|
|
}
|
|
|
|
@@ -161,7 +167,6 @@ static int zstd_create(struct zcomp_params *params, struct zcomp_ctx *ctx)
|
|
return 0;
|
|
|
|
error:
|
|
- zstd_release_params(params);
|
|
zstd_destroy(ctx);
|
|
return -EINVAL;
|
|
}
|
|
diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c
|
|
--- a/drivers/block/zram/zram_drv.c
|
|
+++ b/drivers/block/zram/zram_drv.c
|
|
@@ -1664,6 +1664,17 @@ static void comp_algorithm_set(struct zram *zram, u32 prio, const char *alg)
|
|
zram->comp_algs[prio] = alg;
|
|
}
|
|
|
|
+static void comp_params_reset(struct zram *zram, u32 prio)
|
|
+{
|
|
+ struct zcomp_params *params = &zram->params[prio];
|
|
+
|
|
+ vfree(params->dict);
|
|
+ params->level = ZCOMP_PARAM_NOT_SET;
|
|
+ params->deflate.winbits = ZCOMP_PARAM_NOT_SET;
|
|
+ params->dict_sz = 0;
|
|
+ params->dict = NULL;
|
|
+}
|
|
+
|
|
static int __comp_algorithm_store(struct zram *zram, u32 prio, const char *buf)
|
|
{
|
|
const char *alg;
|
|
@@ -1684,20 +1695,10 @@ static int __comp_algorithm_store(struct zram *zram, u32 prio, const char *buf)
|
|
}
|
|
|
|
comp_algorithm_set(zram, prio, alg);
|
|
+ comp_params_reset(zram, prio);
|
|
return 0;
|
|
}
|
|
|
|
-static void comp_params_reset(struct zram *zram, u32 prio)
|
|
-{
|
|
- struct zcomp_params *params = &zram->params[prio];
|
|
-
|
|
- vfree(params->dict);
|
|
- params->level = ZCOMP_PARAM_NOT_SET;
|
|
- params->deflate.winbits = ZCOMP_PARAM_NOT_SET;
|
|
- params->dict_sz = 0;
|
|
- params->dict = NULL;
|
|
-}
|
|
-
|
|
static int comp_params_store(struct zram *zram, u32 prio, s32 level,
|
|
const char *dict_path,
|
|
struct deflate_params *deflate_params)
|
|
@@ -1712,8 +1713,16 @@ static int comp_params_store(struct zram *zram, u32 prio, s32 level,
|
|
INT_MAX,
|
|
NULL,
|
|
READING_POLICY);
|
|
- if (sz < 0)
|
|
+ if (sz < 0) {
|
|
+ pr_err("failed to load dictionary %s (err=%zd)\n",
|
|
+ dict_path, sz);
|
|
+ return sz;
|
|
+ }
|
|
+ if (sz == 0) {
|
|
+ pr_err("failed to load dictionary %s (empty file)\n",
|
|
+ dict_path);
|
|
return -EINVAL;
|
|
+ }
|
|
}
|
|
|
|
zram->params[prio].dict_sz = sz;
|
|
diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
|
|
--- a/include/linux/memcontrol.h
|
|
+++ b/include/linux/memcontrol.h
|
|
@@ -952,6 +952,8 @@ unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item);
|
|
bool memcg_stat_item_valid(int idx);
|
|
bool memcg_vm_event_item_valid(enum vm_event_item idx);
|
|
unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx);
|
|
+unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
|
|
+ enum node_stat_item idx);
|
|
unsigned long lruvec_page_state_local(struct lruvec *lruvec,
|
|
enum node_stat_item idx);
|
|
|
|
@@ -1404,6 +1406,12 @@ static inline unsigned long lruvec_page_state(struct lruvec *lruvec,
|
|
return node_page_state(lruvec_pgdat(lruvec), idx);
|
|
}
|
|
|
|
+static inline unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
|
|
+ enum node_stat_item idx)
|
|
+{
|
|
+ return node_page_state_monotonic(lruvec_pgdat(lruvec), idx);
|
|
+}
|
|
+
|
|
static inline unsigned long lruvec_page_state_local(struct lruvec *lruvec,
|
|
enum node_stat_item idx)
|
|
{
|
|
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
|
|
--- a/include/linux/mmzone.h
|
|
+++ b/include/linux/mmzone.h
|
|
@@ -323,6 +323,8 @@ enum node_stat_item {
|
|
PGSCAN_PROACTIVE,
|
|
PGSCAN_ANON,
|
|
PGSCAN_FILE,
|
|
+ PGROTATE_ANON,
|
|
+ PGROTATE_FILE,
|
|
PGREFILL,
|
|
#ifdef CONFIG_HUGETLB_PAGE
|
|
NR_HUGETLB,
|
|
@@ -755,6 +757,12 @@ void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent,
|
|
|
|
#endif /* CONFIG_LRU_GEN */
|
|
|
|
+struct lru_cost {
|
|
+ unsigned long count;
|
|
+ unsigned long last_rotated;
|
|
+ unsigned long last_io;
|
|
+};
|
|
+
|
|
struct lruvec {
|
|
struct list_head lists[NR_LRU_LISTS];
|
|
/* per lruvec lru_lock for memcg */
|
|
@@ -763,9 +771,12 @@ struct lruvec {
|
|
* These track the cost of reclaiming one LRU - file or anon -
|
|
* over the other. As the observed cost of reclaiming one LRU
|
|
* increases, the reclaim scan balance tips toward the other.
|
|
+ * Updated and decayed at prepare_scan_control() time; cost_lock
|
|
+ * serialises that update.
|
|
*/
|
|
- unsigned long anon_cost;
|
|
- unsigned long file_cost;
|
|
+ struct lru_cost cost[ANON_AND_FILE];
|
|
+ /* Protects cost[]. */
|
|
+ spinlock_t cost_lock;
|
|
/* Non-resident age, driven by LRU movement */
|
|
atomic_long_t nonresident_age;
|
|
/* Refaults at the time of last reclaim cycle */
|
|
diff --git a/include/linux/swap.h b/include/linux/swap.h
|
|
--- a/include/linux/swap.h
|
|
+++ b/include/linux/swap.h
|
|
@@ -309,9 +309,6 @@ extern unsigned long totalreserve_pages;
|
|
|
|
|
|
/* linux/mm/swap.c */
|
|
-void lru_note_cost_unlock_irq(struct lruvec *lruvec, bool file,
|
|
- unsigned int nr_io, unsigned int nr_rotated);
|
|
-void lru_note_cost_refault(struct folio *);
|
|
void folio_add_lru(struct folio *);
|
|
void folio_add_lru_vma(struct folio *, struct vm_area_struct *);
|
|
void mark_page_accessed(struct page *);
|
|
diff --git a/include/linux/vmstat.h b/include/linux/vmstat.h
|
|
--- a/include/linux/vmstat.h
|
|
+++ b/include/linux/vmstat.h
|
|
@@ -20,7 +20,6 @@ struct reclaim_stat {
|
|
unsigned nr_congested;
|
|
unsigned nr_writeback;
|
|
unsigned nr_immediate;
|
|
- unsigned nr_pageout;
|
|
unsigned nr_activate[ANON_AND_FILE];
|
|
unsigned nr_ref_keep;
|
|
unsigned nr_unmap_fail;
|
|
@@ -194,6 +193,19 @@ unsigned long global_node_page_state_pages(enum node_stat_item item)
|
|
return x;
|
|
}
|
|
|
|
+/*
|
|
+ * Non-clamping variant of global_node_page_state() intended for callers that
|
|
+ * snapshot a monotonically-incremented counter and subtract two samples.
|
|
+ * Returns the raw wrapping value so that unsigned modular subtraction stays
|
|
+ * correct across a signed-long overflow (a real hazard on 32-bit) that the
|
|
+ * clamp in global_node_page_state() would otherwise turn into a huge spurious
|
|
+ * delta. Do NOT use for non-monotonic page-count reads.
|
|
+ */
|
|
+static inline unsigned long global_node_page_state_monotonic(enum node_stat_item item)
|
|
+{
|
|
+ return (unsigned long)atomic_long_read(&vm_node_stat[item]);
|
|
+}
|
|
+
|
|
static inline unsigned long global_node_page_state(enum node_stat_item item)
|
|
{
|
|
VM_WARN_ON_ONCE(vmstat_item_in_bytes(item));
|
|
@@ -259,11 +271,14 @@ extern unsigned long node_page_state(struct pglist_data *pgdat,
|
|
enum node_stat_item item);
|
|
extern unsigned long node_page_state_pages(struct pglist_data *pgdat,
|
|
enum node_stat_item item);
|
|
+extern unsigned long node_page_state_monotonic(struct pglist_data *pgdat,
|
|
+ enum node_stat_item item);
|
|
extern void fold_vm_numa_events(void);
|
|
#else
|
|
#define sum_zone_node_page_state(node, item) global_zone_page_state(item)
|
|
#define node_page_state(node, item) global_node_page_state(item)
|
|
#define node_page_state_pages(node, item) global_node_page_state_pages(item)
|
|
+#define node_page_state_monotonic(node, item) global_node_page_state_monotonic(item)
|
|
static inline void fold_vm_numa_events(void)
|
|
{
|
|
}
|
|
diff --git a/lib/maple_tree.c b/lib/maple_tree.c
|
|
--- a/lib/maple_tree.c
|
|
+++ b/lib/maple_tree.c
|
|
@@ -1501,14 +1501,26 @@ static inline void mas_parent_gap(struct ma_state *mas, unsigned char offset,
|
|
goto ascend;
|
|
}
|
|
|
|
+static __always_inline void mas_update_gap_known(struct ma_state *mas,
|
|
+ unsigned long gap)
|
|
+{
|
|
+ unsigned char pslot;
|
|
+ unsigned long p_gap;
|
|
+
|
|
+ pslot = mte_parent_slot(mas->node);
|
|
+ p_gap = ma_gaps(mte_parent(mas->node),
|
|
+ mas_parent_type(mas, mas->node))[pslot];
|
|
+
|
|
+ if (p_gap != gap)
|
|
+ mas_parent_gap(mas, pslot, gap);
|
|
+}
|
|
+
|
|
/*
|
|
* mas_update_gap() - Update a nodes gaps and propagate up if necessary.
|
|
* @mas: the maple state.
|
|
*/
|
|
static inline void mas_update_gap(struct ma_state *mas)
|
|
{
|
|
- unsigned char pslot;
|
|
- unsigned long p_gap;
|
|
unsigned long max_gap;
|
|
|
|
if (!mt_is_alloc(mas->tree))
|
|
@@ -1518,13 +1530,7 @@ static inline void mas_update_gap(struct ma_state *mas)
|
|
return;
|
|
|
|
max_gap = mas_max_gap(mas);
|
|
-
|
|
- pslot = mte_parent_slot(mas->node);
|
|
- p_gap = ma_gaps(mte_parent(mas->node),
|
|
- mas_parent_type(mas, mas->node))[pslot];
|
|
-
|
|
- if (p_gap != max_gap)
|
|
- mas_parent_gap(mas, pslot, max_gap);
|
|
+ mas_update_gap_known(mas, max_gap);
|
|
}
|
|
|
|
/*
|
|
@@ -2081,8 +2087,8 @@ static inline void mas_wmb_replace(struct ma_state *mas, struct maple_copy *cp)
|
|
mas->node = mt_slot_locked(mas->tree, cp->slot, 0);
|
|
/* Insert the new data in the tree */
|
|
mas_topiary_replace(mas, old_enode, cp->height);
|
|
- if (!mte_is_leaf(mas->node))
|
|
- mas_update_gap(mas);
|
|
+ if (mt_is_alloc(mas->tree) && !mte_is_root(mas->node))
|
|
+ mas_update_gap_known(mas, cp->gap[0]);
|
|
|
|
mtree_range_walk(mas);
|
|
}
|
|
@@ -3125,7 +3131,7 @@ static void mas_wr_spanning_store(struct ma_wr_state *wr_mas)
|
|
static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
|
|
{
|
|
unsigned char dst_offset, offset_end;
|
|
- unsigned char copy_size, node_pivots;
|
|
+ unsigned char copy_size, node_pivots, node_slots;
|
|
struct maple_node reuse, *newnode;
|
|
unsigned long *dst_pivots;
|
|
void __rcu **dst_slots;
|
|
@@ -3138,6 +3144,7 @@ static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
|
|
in_rcu = mt_in_rcu(mas->tree);
|
|
offset_end = wr_mas->offset_end;
|
|
node_pivots = mt_pivots[wr_mas->type];
|
|
+ node_slots = mt_slots[wr_mas->type];
|
|
/* Assume last adds an entry */
|
|
new_end = mas->end + 1 - offset_end + mas->offset;
|
|
if (mas->last == wr_mas->end_piv) {
|
|
@@ -3149,7 +3156,6 @@ static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
|
|
if (in_rcu) {
|
|
newnode = mas_pop_node(mas);
|
|
} else {
|
|
- memset(&reuse, 0, sizeof(struct maple_node));
|
|
newnode = &reuse;
|
|
}
|
|
|
|
@@ -3193,7 +3199,21 @@ static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
|
|
dst_pivots[new_end] = mas->max;
|
|
|
|
done:
|
|
- mas_leaf_set_meta(newnode, maple_leaf_64, new_end);
|
|
+ if (!in_rcu && new_end + 2 < node_slots) {
|
|
+ unsigned char clear_from = new_end + 1;
|
|
+
|
|
+ /*
|
|
+ * Note that the last slot is never cleared, since the metadata
|
|
+ * will be stored there or it has a value.
|
|
+ */
|
|
+ memset(dst_slots + clear_from, 0,
|
|
+ sizeof(void __rcu *) * (node_slots - clear_from));
|
|
+ if (clear_from < node_pivots)
|
|
+ memset(dst_pivots + clear_from, 0,
|
|
+ sizeof(unsigned long) * (node_pivots - clear_from));
|
|
+ }
|
|
+
|
|
+ mas_leaf_set_meta(newnode, wr_mas->type, new_end);
|
|
if (in_rcu) {
|
|
struct maple_enode *old_enode = mas->node;
|
|
|
|
@@ -3218,7 +3238,7 @@ static inline void mas_wr_slot_store(struct ma_wr_state *wr_mas)
|
|
void __rcu **slots = wr_mas->slots;
|
|
bool gap = false;
|
|
|
|
- gap |= !mt_slot_locked(mas->tree, slots, offset);
|
|
+ gap |= !wr_mas->content;
|
|
gap |= !mt_slot_locked(mas->tree, slots, offset + 1);
|
|
|
|
if (wr_mas->offset_end - offset == 1) {
|
|
@@ -3661,6 +3681,9 @@ static inline enum store_type mas_wr_store_type(struct ma_wr_state *wr_mas)
|
|
{
|
|
struct ma_state *mas = wr_mas->mas;
|
|
unsigned char new_end;
|
|
+ bool appending;
|
|
+ bool one_slot;
|
|
+ bool in_rcu;
|
|
|
|
if (unlikely(mas_is_none(mas) || mas_is_ptr(mas)))
|
|
return wr_store_root;
|
|
@@ -3680,21 +3703,30 @@ static inline enum store_type mas_wr_store_type(struct ma_wr_state *wr_mas)
|
|
return wr_new_root;
|
|
|
|
new_end = mas_wr_new_end(wr_mas);
|
|
+ in_rcu = mt_in_rcu(mas->tree);
|
|
+ appending = mas->offset == mas->end;
|
|
+ one_slot = wr_mas->offset_end - mas->offset == 1;
|
|
+
|
|
/* Potential spanning rebalance collapsing a node */
|
|
if (new_end < mt_min_slots[wr_mas->type]) {
|
|
if (!mte_is_root(mas->node))
|
|
return wr_rebalance;
|
|
+ if (!in_rcu) {
|
|
+ if (appending)
|
|
+ return wr_append;
|
|
+ else if (mas->end == new_end && one_slot)
|
|
+ return wr_slot_store;
|
|
+ }
|
|
return wr_node_store;
|
|
}
|
|
|
|
if (new_end >= mt_slots[wr_mas->type])
|
|
return wr_split_store;
|
|
|
|
- if (!mt_in_rcu(mas->tree) && (mas->offset == mas->end))
|
|
+ if (!in_rcu && appending)
|
|
return wr_append;
|
|
|
|
- if ((new_end == mas->end) && (!mt_in_rcu(mas->tree) ||
|
|
- (wr_mas->offset_end - mas->offset == 1)))
|
|
+ if (new_end == mas->end && (!in_rcu || one_slot))
|
|
return wr_slot_store;
|
|
|
|
return wr_node_store;
|
|
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
|
|
--- a/mm/khugepaged.c
|
|
+++ b/mm/khugepaged.c
|
|
@@ -1442,10 +1442,10 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s
|
|
|
|
result = SCAN_SUCCEED;
|
|
out_up_write:
|
|
- if (anon_vma_locked)
|
|
- anon_vma_unlock_write(vma->anon_vma);
|
|
if (pte)
|
|
pte_unmap(pte);
|
|
+ if (anon_vma_locked)
|
|
+ anon_vma_unlock_write(vma->anon_vma);
|
|
mmap_write_unlock(mm);
|
|
out_nolock:
|
|
if (folio)
|
|
diff --git a/mm/ksm.c b/mm/ksm.c
|
|
--- a/mm/ksm.c
|
|
+++ b/mm/ksm.c
|
|
@@ -3061,10 +3061,9 @@ int __ksm_enter(struct mm_struct *mm)
|
|
|
|
slot = &mm_slot->slot;
|
|
|
|
+ spin_lock(&ksm_mmlist_lock);
|
|
/* Check ksm_run too? Would need tighter locking */
|
|
needs_wakeup = list_empty(&ksm_mm_head.slot.mm_node);
|
|
-
|
|
- spin_lock(&ksm_mmlist_lock);
|
|
mm_slot_insert(mm_slots_hash, mm, slot);
|
|
/*
|
|
* When KSM_RUN_MERGE (or KSM_RUN_STOP),
|
|
diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c
|
|
--- a/mm/memcontrol-v1.c
|
|
+++ b/mm/memcontrol-v1.c
|
|
@@ -1998,8 +1998,8 @@ void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s)
|
|
for_each_online_pgdat(pgdat) {
|
|
mz = memcg->nodeinfo[pgdat->node_id];
|
|
|
|
- anon_cost += mz->lruvec.anon_cost;
|
|
- file_cost += mz->lruvec.file_cost;
|
|
+ anon_cost += mz->lruvec.cost[WORKINGSET_ANON].count;
|
|
+ file_cost += mz->lruvec.cost[WORKINGSET_FILE].count;
|
|
}
|
|
seq_buf_printf(s, "anon_cost %lu\n", anon_cost);
|
|
seq_buf_printf(s, "file_cost %lu\n", file_cost);
|
|
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
|
|
--- a/mm/memcontrol.c
|
|
+++ b/mm/memcontrol.c
|
|
@@ -393,6 +393,7 @@ static const unsigned int memcg_node_stat_items[] = {
|
|
NR_SHMEM_THPS,
|
|
NR_FILE_THPS,
|
|
NR_ANON_THPS,
|
|
+ NR_VMSCAN_WRITE,
|
|
NR_VMALLOC,
|
|
NR_KERNEL_STACK_KB,
|
|
NR_PAGETABLE,
|
|
@@ -419,6 +420,8 @@ static const unsigned int memcg_node_stat_items[] = {
|
|
PGSCAN_PROACTIVE,
|
|
PGSCAN_ANON,
|
|
PGSCAN_FILE,
|
|
+ PGROTATE_ANON,
|
|
+ PGROTATE_FILE,
|
|
PGREFILL,
|
|
#ifdef CONFIG_HUGETLB_PAGE
|
|
NR_HUGETLB,
|
|
@@ -502,6 +505,42 @@ unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx)
|
|
return x;
|
|
}
|
|
|
|
+/**
|
|
+ * lruvec_page_state_monotonic - non-clamping lruvec stat read for delta sampling
|
|
+ * @lruvec: the LRU vector to read from
|
|
+ * @idx: the node_stat_item to read
|
|
+ *
|
|
+ * Returns the raw state[idx] value cast to unsigned long, skipping the
|
|
+ * clamp-negative-to-zero step in lruvec_page_state(). Intended for callers
|
|
+ * that snapshot a monotonically-incremented counter and subtract two
|
|
+ * samples: unsigned modular arithmetic then yields the correct delta across
|
|
+ * a signed-long wraparound (a real hazard on 32-bit) that the clamp would
|
|
+ * otherwise turn into a huge spurious delta.
|
|
+ *
|
|
+ * Do NOT use for non-monotonic page-count reads where a transient negative
|
|
+ * reading from per-CPU delta skew must present as zero.
|
|
+ *
|
|
+ * XXX: This helper (and its node/global peers) exists because some
|
|
+ * monotonically-incremented event counters are stored in
|
|
+ * enum node_stat_item.
|
|
+ */
|
|
+unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
|
|
+ enum node_stat_item idx)
|
|
+{
|
|
+ struct mem_cgroup_per_node *pn;
|
|
+ int i;
|
|
+
|
|
+ if (mem_cgroup_disabled())
|
|
+ return node_page_state_monotonic(lruvec_pgdat(lruvec), idx);
|
|
+
|
|
+ i = memcg_stats_index(idx);
|
|
+ if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx))
|
|
+ return 0;
|
|
+
|
|
+ pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
|
|
+ return (unsigned long)READ_ONCE(pn->lruvec_stats->state[i]);
|
|
+}
|
|
+
|
|
unsigned long lruvec_page_state_local(struct lruvec *lruvec,
|
|
enum node_stat_item idx)
|
|
{
|
|
@@ -2095,7 +2134,12 @@ static bool consume_stock(struct mem_cgroup *memcg, unsigned int nr_pages)
|
|
|
|
stock_pages = READ_ONCE(stock->nr_pages[i]);
|
|
if (stock_pages >= nr_pages) {
|
|
- WRITE_ONCE(stock->nr_pages[i], stock_pages - nr_pages);
|
|
+ stock_pages -= nr_pages;
|
|
+ WRITE_ONCE(stock->nr_pages[i], stock_pages);
|
|
+ if (!stock_pages) {
|
|
+ css_put(&memcg->css);
|
|
+ WRITE_ONCE(stock->cached[i], NULL);
|
|
+ }
|
|
ret = true;
|
|
}
|
|
break;
|
|
diff --git a/mm/migrate_device.c b/mm/migrate_device.c
|
|
--- a/mm/migrate_device.c
|
|
+++ b/mm/migrate_device.c
|
|
@@ -872,7 +872,7 @@ static int migrate_vma_insert_huge_pmd_page(struct migrate_vma *migrate,
|
|
|
|
if (flush) {
|
|
pte_free(vma->vm_mm, pgtable);
|
|
- flush_cache_page(vma, addr, addr + HPAGE_PMD_SIZE);
|
|
+ flush_cache_range(vma, addr, addr + HPAGE_PMD_SIZE);
|
|
pmdp_invalidate(vma, addr, pmdp);
|
|
} else {
|
|
pgtable_trans_huge_deposit(vma->vm_mm, pmdp, pgtable);
|
|
diff --git a/mm/mmzone.c b/mm/mmzone.c
|
|
--- a/mm/mmzone.c
|
|
+++ b/mm/mmzone.c
|
|
@@ -78,6 +78,7 @@ void lruvec_init(struct lruvec *lruvec)
|
|
|
|
memset(lruvec, 0, sizeof(struct lruvec));
|
|
spin_lock_init(&lruvec->lru_lock);
|
|
+ spin_lock_init(&lruvec->cost_lock);
|
|
zswap_lruvec_state_init(lruvec);
|
|
|
|
for_each_lru(lru)
|
|
diff --git a/mm/mremap.c b/mm/mremap.c
|
|
--- a/mm/mremap.c
|
|
+++ b/mm/mremap.c
|
|
@@ -264,7 +264,7 @@ static int move_ptes(struct pagetable_move_control *pmc,
|
|
|
|
for (; old_addr < old_end; old_ptep += nr_ptes, old_addr += nr_ptes * PAGE_SIZE,
|
|
new_ptep += nr_ptes, new_addr += nr_ptes * PAGE_SIZE) {
|
|
- VM_WARN_ON_ONCE(!pte_none(*new_ptep));
|
|
+ VM_WARN_ON_ONCE(!pte_none(ptep_get(new_ptep)));
|
|
|
|
nr_ptes = 1;
|
|
max_nr_ptes = (old_end - old_addr) >> PAGE_SHIFT;
|
|
diff --git a/mm/shmem.c b/mm/shmem.c
|
|
--- a/mm/shmem.c
|
|
+++ b/mm/shmem.c
|
|
@@ -3612,6 +3612,7 @@ static long shmem_fallocate(struct file *file, int mode, loff_t offset,
|
|
struct shmem_inode_info *info = SHMEM_I(inode);
|
|
struct shmem_falloc shmem_falloc;
|
|
pgoff_t start, index, end, undo_fallocend;
|
|
+ loff_t aligned_end;
|
|
int error;
|
|
|
|
if (mode & ~(FALLOC_FL_KEEP_SIZE | FALLOC_FL_PUNCH_HOLE))
|
|
@@ -3668,8 +3669,15 @@ static long shmem_fallocate(struct file *file, int mode, loff_t offset,
|
|
goto out;
|
|
}
|
|
|
|
+ /* Check for wraparound */
|
|
+ if (check_add_overflow(offset + len, (loff_t)PAGE_SIZE - 1,
|
|
+ &aligned_end)) {
|
|
+ error = -EFBIG;
|
|
+ goto out;
|
|
+ }
|
|
+
|
|
start = offset >> PAGE_SHIFT;
|
|
- end = (offset + len + PAGE_SIZE - 1) >> PAGE_SHIFT;
|
|
+ end = aligned_end >> PAGE_SHIFT;
|
|
/* Try to avoid a swapstorm if len is impossible to satisfy */
|
|
if (sbinfo->max_blocks && end - start > sbinfo->max_blocks) {
|
|
error = -ENOSPC;
|
|
diff --git a/mm/swap.c b/mm/swap.c
|
|
--- a/mm/swap.c
|
|
+++ b/mm/swap.c
|
|
@@ -272,73 +272,6 @@ void folio_rotate_reclaimable(struct folio *folio)
|
|
folio_batch_add_and_move(folio, lru_move_tail);
|
|
}
|
|
|
|
-void lru_note_cost_unlock_irq(struct lruvec *lruvec, bool file,
|
|
- unsigned int nr_io, unsigned int nr_rotated)
|
|
- __releases(lruvec->lru_lock)
|
|
- __releases(rcu)
|
|
-{
|
|
- unsigned long cost;
|
|
-
|
|
- /*
|
|
- * Reflect the relative cost of incurring IO and spending CPU
|
|
- * time on rotations. This doesn't attempt to make a precise
|
|
- * comparison, it just says: if reloads are about comparable
|
|
- * between the LRU lists, or rotations are overwhelmingly
|
|
- * different between them, adjust scan balance for CPU work.
|
|
- */
|
|
- cost = nr_io * SWAP_CLUSTER_MAX + nr_rotated;
|
|
- if (!cost) {
|
|
- spin_unlock_irq(&lruvec->lru_lock);
|
|
- rcu_read_unlock();
|
|
- return;
|
|
- }
|
|
-
|
|
- for (;;) {
|
|
- unsigned long lrusize;
|
|
-
|
|
- /* Record cost event */
|
|
- if (file)
|
|
- lruvec->file_cost += cost;
|
|
- else
|
|
- lruvec->anon_cost += cost;
|
|
-
|
|
- /*
|
|
- * Decay previous events
|
|
- *
|
|
- * Because workloads change over time (and to avoid
|
|
- * overflow) we keep these statistics as a floating
|
|
- * average, which ends up weighing recent refaults
|
|
- * more than old ones.
|
|
- */
|
|
- lrusize = lruvec_page_state(lruvec, NR_INACTIVE_ANON) +
|
|
- lruvec_page_state(lruvec, NR_ACTIVE_ANON) +
|
|
- lruvec_page_state(lruvec, NR_INACTIVE_FILE) +
|
|
- lruvec_page_state(lruvec, NR_ACTIVE_FILE);
|
|
-
|
|
- if (lruvec->file_cost + lruvec->anon_cost > lrusize / 4) {
|
|
- lruvec->file_cost /= 2;
|
|
- lruvec->anon_cost /= 2;
|
|
- }
|
|
-
|
|
- spin_unlock_irq(&lruvec->lru_lock);
|
|
- lruvec = parent_lruvec(lruvec);
|
|
- if (!lruvec) {
|
|
- rcu_read_unlock();
|
|
- break;
|
|
- }
|
|
- spin_lock_irq(&lruvec->lru_lock);
|
|
- }
|
|
-}
|
|
-
|
|
-void lru_note_cost_refault(struct folio *folio)
|
|
-{
|
|
- struct lruvec *lruvec;
|
|
-
|
|
- lruvec = folio_lruvec_lock_irq(folio);
|
|
- lru_note_cost_unlock_irq(lruvec, folio_is_file_lru(folio),
|
|
- folio_nr_pages(folio), 0);
|
|
-}
|
|
-
|
|
static void lru_activate(struct lruvec *lruvec, struct folio *folio)
|
|
{
|
|
long nr_pages = folio_nr_pages(folio);
|
|
@@ -1173,8 +1106,6 @@ void lru_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int
|
|
|
|
child_lruvec = mem_cgroup_lruvec(memcg, NODE_DATA(nid));
|
|
parent_lruvec = mem_cgroup_lruvec(parent, NODE_DATA(nid));
|
|
- parent_lruvec->anon_cost += child_lruvec->anon_cost;
|
|
- parent_lruvec->file_cost += child_lruvec->file_cost;
|
|
|
|
for_each_lru(lru)
|
|
lruvec_reparent_lru(child_lruvec, parent_lruvec, lru, nid);
|
|
diff --git a/mm/vma.c b/mm/vma.c
|
|
--- a/mm/vma.c
|
|
+++ b/mm/vma.c
|
|
@@ -1866,10 +1866,10 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap,
|
|
bool *need_rmap_locks)
|
|
{
|
|
struct vm_area_struct *vma = *vmap;
|
|
- unsigned long vma_start = vma->vm_start;
|
|
+ unsigned long old_vma_start = vma->vm_start;
|
|
struct mm_struct *mm = vma->vm_mm;
|
|
struct vm_area_struct *new_vma;
|
|
- bool faulted_in_anon_vma = true;
|
|
+ bool can_self_merge = false;
|
|
VMA_ITERATOR(vmi, mm, addr);
|
|
VMG_VMA_STATE(vmg, &vmi, NULL, vma, addr, addr + len);
|
|
|
|
@@ -1879,7 +1879,7 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap,
|
|
*/
|
|
if (unlikely(vma_is_anonymous(vma) && !vma->anon_vma)) {
|
|
pgoff = addr >> PAGE_SHIFT;
|
|
- faulted_in_anon_vma = false;
|
|
+ can_self_merge = true;
|
|
}
|
|
|
|
/*
|
|
@@ -1899,24 +1899,21 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap,
|
|
new_vma = vma_merge_copied_range(&vmg);
|
|
|
|
if (new_vma) {
|
|
- /*
|
|
- * Source vma may have been merged into new_vma
|
|
- */
|
|
- if (unlikely(vma_start >= new_vma->vm_start &&
|
|
- vma_start < new_vma->vm_end)) {
|
|
+ /* Self-merged and VMA replaced. */
|
|
+ if (unlikely(new_vma->vm_start < old_vma_start &&
|
|
+ new_vma->vm_end > old_vma_start)) {
|
|
/*
|
|
- * The only way we can get a vma_merge with
|
|
- * self during an mremap is if the vma hasn't
|
|
- * been faulted in yet and we were allowed to
|
|
- * reset the dst vma->vm_pgoff to the
|
|
- * destination address of the mremap to allow
|
|
- * the merge to happen. mremap must change the
|
|
- * vm_pgoff linearity between src and dst vmas
|
|
- * (in turn preventing a vma_merge) to be
|
|
- * safe. It is only safe to keep the vm_pgoff
|
|
- * linear if there are no pages mapped yet.
|
|
+ * The only way a VMA can both self-merge and be
|
|
+ * replaced is if the remap places the new VMA
|
|
+ * immediately prior to its old self ('next') and
|
|
+ * immediately after another VMA ('prev') causing the
|
|
+ * next to be removed and prev to be expanded to cover
|
|
+ * the entire range.
|
|
+ *
|
|
+ * This should only be possible if the page offset was
|
|
+ * updated, i.e. the VMA is unfaulted.
|
|
*/
|
|
- VM_BUG_ON_VMA(faulted_in_anon_vma, new_vma);
|
|
+ VM_WARN_ON_ONCE_VMA(!can_self_merge, new_vma);
|
|
*vmap = vma = new_vma;
|
|
}
|
|
*need_rmap_locks = (new_vma->vm_pgoff <= vma->vm_pgoff);
|
|
diff --git a/mm/vmscan.c b/mm/vmscan.c
|
|
--- a/mm/vmscan.c
|
|
+++ b/mm/vmscan.c
|
|
@@ -669,7 +669,7 @@ static pageout_t pageout(struct folio *folio, struct address_space *mapping,
|
|
folio_clear_reclaim(folio);
|
|
|
|
trace_mm_vmscan_write_folio(folio);
|
|
- node_stat_add_folio(folio, NR_VMSCAN_WRITE);
|
|
+ lruvec_stat_mod_folio(folio, NR_VMSCAN_WRITE, folio_nr_pages(folio));
|
|
return PAGE_SUCCESS;
|
|
}
|
|
|
|
@@ -1420,8 +1420,6 @@ static unsigned int shrink_folio_list(struct list_head *folio_list,
|
|
sc->nr_scanned -= (nr_pages - 1);
|
|
nr_pages = 1;
|
|
}
|
|
- stat->nr_pageout += nr_pages;
|
|
-
|
|
if (folio_test_writeback(folio))
|
|
goto keep;
|
|
if (folio_test_dirty(folio))
|
|
@@ -2045,10 +2043,10 @@ static unsigned long shrink_inactive_list(unsigned long nr_to_scan,
|
|
item = PGSTEAL_KSWAPD + reclaimer_offset(sc);
|
|
mod_lruvec_state(lruvec, item, nr_reclaimed);
|
|
mod_lruvec_state(lruvec, PGSTEAL_ANON + file, nr_reclaimed);
|
|
+ if (nr_scanned > nr_reclaimed)
|
|
+ mod_lruvec_state(lruvec, PGROTATE_ANON + file,
|
|
+ nr_scanned - nr_reclaimed);
|
|
|
|
- lruvec_lock_irq(lruvec);
|
|
- lru_note_cost_unlock_irq(lruvec, file, stat.nr_pageout,
|
|
- nr_scanned - nr_reclaimed);
|
|
handle_reclaim_writeback(nr_taken, pgdat, sc, &stat);
|
|
trace_mm_vmscan_lru_shrink_inactive(pgdat->node_id,
|
|
nr_scanned, nr_reclaimed, &stat, sc->priority, file);
|
|
@@ -2154,9 +2152,9 @@ static void shrink_active_list(unsigned long nr_to_scan,
|
|
count_vm_events(PGDEACTIVATE, nr_deactivate);
|
|
count_memcg_events(lruvec_memcg(lruvec), PGDEACTIVATE, nr_deactivate);
|
|
mod_node_page_state(pgdat, NR_ISOLATED_ANON + file, -nr_taken);
|
|
+ if (nr_rotated)
|
|
+ mod_lruvec_state(lruvec, PGROTATE_ANON + file, nr_rotated);
|
|
|
|
- lruvec_lock_irq(lruvec);
|
|
- lru_note_cost_unlock_irq(lruvec, file, 0, nr_rotated);
|
|
trace_mm_vmscan_lru_shrink_active(pgdat->node_id, nr_taken, nr_activate,
|
|
nr_deactivate, nr_rotated, sc->priority, file);
|
|
}
|
|
@@ -2289,8 +2287,10 @@ enum scan_balance {
|
|
|
|
static void prepare_scan_control(pg_data_t *pgdat, struct scan_control *sc)
|
|
{
|
|
- unsigned long file;
|
|
+ struct lru_cost *anon_cost, *file_cost;
|
|
struct lruvec *target_lruvec;
|
|
+ unsigned long lrusize;
|
|
+ unsigned long file;
|
|
|
|
if (lru_gen_enabled() && !lru_gen_switching())
|
|
return;
|
|
@@ -2306,11 +2306,69 @@ static void prepare_scan_control(pg_data_t *pgdat, struct scan_control *sc)
|
|
|
|
/*
|
|
* Determine the scan balance between anon and file LRUs.
|
|
+ *
|
|
+ * The cost model is based on rotations, refaults and
|
|
+ * reclaim-driven writes (anon only) on each side.
|
|
+ *
|
|
+ * These event counters are monotonic, so each reclaim cycle
|
|
+ * the delta since the last scan is extracted and incorporated
|
|
+ * into a decaying average. This ensures currency, as workloads
|
|
+ * change over time, and avoids overflow in the calculations.
|
|
+ *
|
|
+ * Use lruvec_page_state_monotonic() so unsigned subtraction
|
|
+ * yields the correct delta across a signed-long wraparound of
|
|
+ * the underlying counter (a real hazard on 32-bit that the
|
|
+ * clamp in lruvec_page_state() would otherwise turn into a huge
|
|
+ * spurious delta).
|
|
*/
|
|
- spin_lock_irq(&target_lruvec->lru_lock);
|
|
- sc->anon_cost = target_lruvec->anon_cost;
|
|
- sc->file_cost = target_lruvec->file_cost;
|
|
- spin_unlock_irq(&target_lruvec->lru_lock);
|
|
+ spin_lock(&target_lruvec->cost_lock);
|
|
+
|
|
+ for (int f = 0; f <= 1; f++) {
|
|
+ struct lru_cost *cost = &target_lruvec->cost[f];
|
|
+ unsigned long rotated, io, nr_rotated, nr_io;
|
|
+
|
|
+ rotated = lruvec_page_state_monotonic(target_lruvec,
|
|
+ PGROTATE_ANON + f);
|
|
+ io = lruvec_page_state_monotonic(target_lruvec,
|
|
+ WORKINGSET_RESTORE_BASE + f);
|
|
+ if (f == WORKINGSET_ANON)
|
|
+ io += lruvec_page_state_monotonic(target_lruvec,
|
|
+ NR_VMSCAN_WRITE);
|
|
+
|
|
+ nr_rotated = rotated - cost->last_rotated;
|
|
+ nr_io = io - cost->last_io;
|
|
+
|
|
+ /*
|
|
+ * Reflect the relative cost of incurring IO and spending
|
|
+ * CPU time on rotations. This doesn't attempt to make a
|
|
+ * precise comparison, it just says: if reloads are about
|
|
+ * comparable between the LRU lists, or rotations are
|
|
+ * overwhelmingly different between them, adjust scan
|
|
+ * balance for CPU work.
|
|
+ */
|
|
+ cost->count += nr_io * SWAP_CLUSTER_MAX + nr_rotated;
|
|
+
|
|
+ cost->last_rotated = rotated;
|
|
+ cost->last_io = io;
|
|
+ }
|
|
+
|
|
+ anon_cost = &target_lruvec->cost[WORKINGSET_ANON];
|
|
+ file_cost = &target_lruvec->cost[WORKINGSET_FILE];
|
|
+
|
|
+ lrusize = lruvec_page_state(target_lruvec, NR_INACTIVE_ANON) +
|
|
+ lruvec_page_state(target_lruvec, NR_ACTIVE_ANON) +
|
|
+ lruvec_page_state(target_lruvec, NR_INACTIVE_FILE) +
|
|
+ lruvec_page_state(target_lruvec, NR_ACTIVE_FILE);
|
|
+
|
|
+ while (anon_cost->count + file_cost->count > lrusize / 4) {
|
|
+ anon_cost->count /= 2;
|
|
+ file_cost->count /= 2;
|
|
+ }
|
|
+
|
|
+ sc->anon_cost = anon_cost->count;
|
|
+ sc->file_cost = file_cost->count;
|
|
+
|
|
+ spin_unlock(&target_lruvec->cost_lock);
|
|
|
|
/*
|
|
* Target desirable inactive:active list ratios for the anon
|
|
@@ -4197,7 +4255,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
|
|
unsigned long end;
|
|
struct lru_gen_mm_walk *walk;
|
|
struct folio *last = NULL;
|
|
- int young = 1;
|
|
+ int young = nr;
|
|
pte_t *pte = pvmw->pte;
|
|
unsigned long addr = pvmw->address;
|
|
struct vm_area_struct *vma = pvmw->vma;
|
|
@@ -4819,7 +4877,8 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
|
|
struct reclaim_stat stat;
|
|
struct lru_gen_mm_walk *walk;
|
|
int scanned, reclaimed;
|
|
- int isolated = 0, type, type_scanned;
|
|
+ int isolated = 0, nr_isolated = 0, type, type_scanned;
|
|
+ unsigned long total_reclaimed = 0;
|
|
bool skip_retry = false;
|
|
struct mem_cgroup *memcg = lruvec_memcg(lruvec);
|
|
struct pglist_data *pgdat = lruvec_pgdat(lruvec);
|
|
@@ -4831,6 +4890,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
|
|
|
|
scanned = isolate_folios(nr_to_scan, lruvec, sc, swappiness,
|
|
&list, &isolated, &type, &type_scanned);
|
|
+ nr_isolated = isolated;
|
|
|
|
/* Scanning may have emptied the oldest gen, flush it */
|
|
if (scanned)
|
|
@@ -4843,6 +4903,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
|
|
retry:
|
|
reclaimed = shrink_folio_list(&list, pgdat, sc, &stat, false, memcg);
|
|
sc->nr_reclaimed += reclaimed;
|
|
+ total_reclaimed += reclaimed;
|
|
/* Retry pass is only meant for clean folios without new isolation */
|
|
if (isolated)
|
|
handle_reclaim_writeback(isolated, pgdat, sc, &stat);
|
|
@@ -4892,6 +4953,10 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
|
|
goto retry;
|
|
}
|
|
|
|
+ if (nr_isolated > total_reclaimed)
|
|
+ mod_lruvec_state(lruvec, PGROTATE_ANON + type,
|
|
+ nr_isolated - total_reclaimed);
|
|
+
|
|
return scanned;
|
|
}
|
|
|
|
diff --git a/mm/vmstat.c b/mm/vmstat.c
|
|
--- a/mm/vmstat.c
|
|
+++ b/mm/vmstat.c
|
|
@@ -1024,6 +1024,17 @@ unsigned long node_page_state(struct pglist_data *pgdat,
|
|
|
|
return node_page_state_pages(pgdat, item);
|
|
}
|
|
+
|
|
+/*
|
|
+ * Non-clamping variant of node_page_state() intended for callers that
|
|
+ * snapshot a monotonically-incremented counter and subtract two samples.
|
|
+ * See global_node_page_state_monotonic() for the rationale.
|
|
+ */
|
|
+unsigned long node_page_state_monotonic(struct pglist_data *pgdat,
|
|
+ enum node_stat_item item)
|
|
+{
|
|
+ return (unsigned long)atomic_long_read(&pgdat->vm_stat[item]);
|
|
+}
|
|
#endif
|
|
|
|
/*
|
|
@@ -1289,6 +1300,8 @@ const char * const vmstat_text[] = {
|
|
[I(PGSCAN_PROACTIVE)] = "pgscan_proactive",
|
|
[I(PGSCAN_ANON)] = "pgscan_anon",
|
|
[I(PGSCAN_FILE)] = "pgscan_file",
|
|
+ [I(PGROTATE_ANON)] = "pgrotate_anon",
|
|
+ [I(PGROTATE_FILE)] = "pgrotate_file",
|
|
[I(PGREFILL)] = "pgrefill",
|
|
#ifdef CONFIG_HUGETLB_PAGE
|
|
[I(NR_HUGETLB)] = "nr_hugetlb",
|
|
diff --git a/mm/workingset.c b/mm/workingset.c
|
|
--- a/mm/workingset.c
|
|
+++ b/mm/workingset.c
|
|
@@ -584,11 +584,6 @@ void workingset_refault(struct folio *folio, void *shadow)
|
|
/* Folio was active prior to eviction */
|
|
if (workingset) {
|
|
folio_set_workingset(folio);
|
|
- /*
|
|
- * XXX: Move to folio_add_lru() when it supports new vs
|
|
- * putback
|
|
- */
|
|
- lru_note_cost_refault(folio);
|
|
mod_lruvec_state(lruvec, WORKINGSET_RESTORE_BASE + file, nr);
|
|
}
|
|
out:
|
|
diff --git a/mm/zswap.c b/mm/zswap.c
|
|
--- a/mm/zswap.c
|
|
+++ b/mm/zswap.c
|
|
@@ -1217,7 +1217,7 @@ static unsigned long zswap_shrinker_count(struct shrinker *shrinker,
|
|
* Without memcg, use the zswap pool-wide metrics.
|
|
*/
|
|
if (!mem_cgroup_disabled()) {
|
|
- mem_cgroup_flush_stats(memcg);
|
|
+ mem_cgroup_flush_stats_ratelimited(memcg);
|
|
nr_backing = memcg_page_state(memcg, MEMCG_ZSWAP_B) >> PAGE_SHIFT;
|
|
nr_stored = memcg_page_state(memcg, MEMCG_ZSWAPPED);
|
|
} else {
|
|
@@ -1275,6 +1275,14 @@ static struct shrinker *zswap_alloc_shrinker(void)
|
|
return shrinker;
|
|
}
|
|
|
|
+/*
|
|
+ * Scan up to SWAP_CLUSTER_MAX pages on each per-node zswap LRU of @memcg
|
|
+ * and write back the reclaimable ones.
|
|
+ *
|
|
+ * Return: 0 if at least one entry was written back, -EAGAIN if entries
|
|
+ * were scanned but none could be written back, or -ENOENT if @memcg has
|
|
+ * writeback disabled, is a zombie cgroup, or has empty zswap LRUs.
|
|
+ */
|
|
static int shrink_memcg(struct mem_cgroup *memcg)
|
|
{
|
|
int nid, shrunk = 0, scanned = 0;
|
|
@@ -1290,13 +1298,14 @@ static int shrink_memcg(struct mem_cgroup *memcg)
|
|
return -ENOENT;
|
|
|
|
for_each_node_state(nid, N_NORMAL_MEMORY) {
|
|
- unsigned long nr_to_walk = 1;
|
|
+ unsigned long nr_to_walk = SWAP_CLUSTER_MAX;
|
|
|
|
shrunk += list_lru_walk_one(&zswap_list_lru, nid, memcg,
|
|
&shrink_memcg_cb, NULL, &nr_to_walk);
|
|
- scanned += 1 - nr_to_walk;
|
|
+ scanned += SWAP_CLUSTER_MAX - nr_to_walk;
|
|
}
|
|
|
|
+ /* Nothing was scanned: every LRU under @memcg was empty. */
|
|
if (!scanned)
|
|
return -ENOENT;
|
|
|