Files
omarchy-pkgs/pkgbuilds/linux-omarchy/0280-mm-updates.patch
T

1271 lines
40 KiB
Diff

diff --git a/drivers/block/zram/backend_842.c b/drivers/block/zram/backend_842.c
--- a/drivers/block/zram/backend_842.c
+++ b/drivers/block/zram/backend_842.c
@@ -1,5 +1,7 @@
// SPDX-License-Identifier: GPL-2.0-or-later
+#define pr_fmt(fmt) "842: " fmt
+
#include <linux/kernel.h>
#include <linux/slab.h>
#include <linux/sw842.h>
@@ -13,6 +15,14 @@ static void release_params_842(struct zcomp_params *params)
static int setup_params_842(struct zcomp_params *params)
{
+ if (params->dict_sz) {
+ pr_err("dictionary is not supported\n");
+ return -EOPNOTSUPP;
+ }
+ if (params->level != ZCOMP_PARAM_NOT_SET) {
+ pr_err("compression level is not supported\n");
+ return -EOPNOTSUPP;
+ }
return 0;
}
diff --git a/drivers/block/zram/backend_deflate.c b/drivers/block/zram/backend_deflate.c
--- a/drivers/block/zram/backend_deflate.c
+++ b/drivers/block/zram/backend_deflate.c
@@ -1,5 +1,7 @@
// SPDX-License-Identifier: GPL-2.0-or-later
+#define pr_fmt(fmt) "deflate: " fmt
+
#include <linux/kernel.h>
#include <linux/slab.h>
#include <linux/vmalloc.h>
@@ -22,15 +24,26 @@ static void deflate_release_params(struct zcomp_params *params)
static int deflate_setup_params(struct zcomp_params *params)
{
- if (params->level == ZCOMP_PARAM_NOT_SET)
+ if (params->dict_sz) {
+ pr_err("dictionary is not supported\n");
+ return -EOPNOTSUPP;
+ }
+
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
params->level = Z_DEFAULT_COMPRESSION;
+ } else if (params->level < Z_DEFAULT_COMPRESSION ||
+ params->level > Z_BEST_COMPRESSION) {
+ pr_err("invalid compression level %d\n", params->level);
+ return -EINVAL;
+ }
+
if (params->deflate.winbits == ZCOMP_PARAM_NOT_SET) {
params->deflate.winbits = DEFLATE_DEF_WINBITS;
} else {
s32 wb = params->deflate.winbits;
if ((wb < -15 || wb > -9) && (wb < 9 || wb > 15)) {
- pr_err("invalid deflate winbits: %d\n", wb);
+ pr_err("invalid winbits %d\n", wb);
return -EINVAL;
}
}
diff --git a/drivers/block/zram/backend_lz4.c b/drivers/block/zram/backend_lz4.c
--- a/drivers/block/zram/backend_lz4.c
+++ b/drivers/block/zram/backend_lz4.c
@@ -1,3 +1,7 @@
+// SPDX-License-Identifier: GPL-2.0-or-later
+
+#define pr_fmt(fmt) "lz4: " fmt
+
#include <linux/kernel.h>
#include <linux/lz4.h>
#include <linux/slab.h>
@@ -28,8 +32,12 @@ static int lz4_setup_params(struct zcomp_params *params)
LZ4_stream_t *dict_stream;
int ret;
- if (params->level == ZCOMP_PARAM_NOT_SET)
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
params->level = LZ4_ACCELERATION_DEFAULT;
+ } else if (params->level < LZ4_ACCELERATION_DEFAULT) {
+ pr_err("invalid compression level %d\n", params->level);
+ return -EINVAL;
+ }
if (!params->dict || !params->dict_sz)
return 0;
diff --git a/drivers/block/zram/backend_lz4hc.c b/drivers/block/zram/backend_lz4hc.c
--- a/drivers/block/zram/backend_lz4hc.c
+++ b/drivers/block/zram/backend_lz4hc.c
@@ -1,3 +1,7 @@
+// SPDX-License-Identifier: GPL-2.0-or-later
+
+#define pr_fmt(fmt) "lz4hc: " fmt
+
#include <linux/kernel.h>
#include <linux/lz4.h>
#include <linux/slab.h>
@@ -18,8 +22,18 @@ static void lz4hc_release_params(struct zcomp_params *params)
static int lz4hc_setup_params(struct zcomp_params *params)
{
- if (params->level == ZCOMP_PARAM_NOT_SET)
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
params->level = LZ4HC_DEFAULT_CLEVEL;
+ } else if (params->level < 1 || params->level > LZ4HC_MAX_CLEVEL) {
+ /*
+ * Use < 1 rather than < LZ4HC_MIN_CLEVEL here because
+ * LZ4HC_compress_generic() only clamps levels below 1
+ * (levels 1 and 2 are valid). LZ4HC_MIN_CLEVEL (3) is
+ * advisory and not enforced by the library.
+ */
+ pr_err("invalid compression level %d\n", params->level);
+ return -EINVAL;
+ }
return 0;
}
diff --git a/drivers/block/zram/backend_lzo.c b/drivers/block/zram/backend_lzo.c
--- a/drivers/block/zram/backend_lzo.c
+++ b/drivers/block/zram/backend_lzo.c
@@ -1,5 +1,7 @@
// SPDX-License-Identifier: GPL-2.0-or-later
+#define pr_fmt(fmt) "lzo: " fmt
+
#include <linux/kernel.h>
#include <linux/slab.h>
#include <linux/lzo.h>
@@ -12,6 +14,14 @@ static void lzo_release_params(struct zcomp_params *params)
static int lzo_setup_params(struct zcomp_params *params)
{
+ if (params->dict_sz) {
+ pr_err("dictionary is not supported\n");
+ return -EOPNOTSUPP;
+ }
+ if (params->level != ZCOMP_PARAM_NOT_SET) {
+ pr_err("compression level is not supported\n");
+ return -EOPNOTSUPP;
+ }
return 0;
}
diff --git a/drivers/block/zram/backend_lzorle.c b/drivers/block/zram/backend_lzorle.c
--- a/drivers/block/zram/backend_lzorle.c
+++ b/drivers/block/zram/backend_lzorle.c
@@ -1,5 +1,7 @@
// SPDX-License-Identifier: GPL-2.0-or-later
+#define pr_fmt(fmt) "lzo-rle: " fmt
+
#include <linux/kernel.h>
#include <linux/slab.h>
#include <linux/lzo.h>
@@ -12,6 +14,14 @@ static void lzorle_release_params(struct zcomp_params *params)
static int lzorle_setup_params(struct zcomp_params *params)
{
+ if (params->dict_sz) {
+ pr_err("dictionary is not supported\n");
+ return -EOPNOTSUPP;
+ }
+ if (params->level != ZCOMP_PARAM_NOT_SET) {
+ pr_err("compression level is not supported\n");
+ return -EOPNOTSUPP;
+ }
return 0;
}
diff --git a/drivers/block/zram/backend_zstd.c b/drivers/block/zram/backend_zstd.c
--- a/drivers/block/zram/backend_zstd.c
+++ b/drivers/block/zram/backend_zstd.c
@@ -1,5 +1,7 @@
// SPDX-License-Identifier: GPL-2.0-or-later
+#define pr_fmt(fmt) "zstd: " fmt
+
#include <linux/kernel.h>
#include <linux/slab.h>
#include <linux/vmalloc.h>
@@ -58,8 +60,13 @@ static int zstd_setup_params(struct zcomp_params *params)
return -ENOMEM;
params->drv_data = zp;
- if (params->level == ZCOMP_PARAM_NOT_SET)
+ if (params->level == ZCOMP_PARAM_NOT_SET) {
params->level = zstd_default_clevel();
+ } else if (params->level < zstd_min_clevel() ||
+ params->level > zstd_max_clevel()) {
+ pr_err("invalid compression level %d\n", params->level);
+ goto error;
+ }
zp->cprm = zstd_get_params(params->level, PAGE_SIZE);
@@ -85,7 +92,6 @@ static int zstd_setup_params(struct zcomp_params *params)
return 0;
error:
- zstd_release_params(params);
return -EINVAL;
}
@@ -161,7 +167,6 @@ static int zstd_create(struct zcomp_params *params, struct zcomp_ctx *ctx)
return 0;
error:
- zstd_release_params(params);
zstd_destroy(ctx);
return -EINVAL;
}
diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c
--- a/drivers/block/zram/zram_drv.c
+++ b/drivers/block/zram/zram_drv.c
@@ -1664,6 +1664,17 @@ static void comp_algorithm_set(struct zram *zram, u32 prio, const char *alg)
zram->comp_algs[prio] = alg;
}
+static void comp_params_reset(struct zram *zram, u32 prio)
+{
+ struct zcomp_params *params = &zram->params[prio];
+
+ vfree(params->dict);
+ params->level = ZCOMP_PARAM_NOT_SET;
+ params->deflate.winbits = ZCOMP_PARAM_NOT_SET;
+ params->dict_sz = 0;
+ params->dict = NULL;
+}
+
static int __comp_algorithm_store(struct zram *zram, u32 prio, const char *buf)
{
const char *alg;
@@ -1684,20 +1695,10 @@ static int __comp_algorithm_store(struct zram *zram, u32 prio, const char *buf)
}
comp_algorithm_set(zram, prio, alg);
+ comp_params_reset(zram, prio);
return 0;
}
-static void comp_params_reset(struct zram *zram, u32 prio)
-{
- struct zcomp_params *params = &zram->params[prio];
-
- vfree(params->dict);
- params->level = ZCOMP_PARAM_NOT_SET;
- params->deflate.winbits = ZCOMP_PARAM_NOT_SET;
- params->dict_sz = 0;
- params->dict = NULL;
-}
-
static int comp_params_store(struct zram *zram, u32 prio, s32 level,
const char *dict_path,
struct deflate_params *deflate_params)
@@ -1712,8 +1713,16 @@ static int comp_params_store(struct zram *zram, u32 prio, s32 level,
INT_MAX,
NULL,
READING_POLICY);
- if (sz < 0)
+ if (sz < 0) {
+ pr_err("failed to load dictionary %s (err=%zd)\n",
+ dict_path, sz);
+ return sz;
+ }
+ if (sz == 0) {
+ pr_err("failed to load dictionary %s (empty file)\n",
+ dict_path);
return -EINVAL;
+ }
}
zram->params[prio].dict_sz = sz;
diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h
--- a/include/linux/memcontrol.h
+++ b/include/linux/memcontrol.h
@@ -270,10 +270,15 @@ struct mem_cgroup {
#endif
int kmemcg_id;
- struct memcg_vmstats_percpu __percpu *vmstats_percpu;
-
#ifdef CONFIG_CGROUP_WRITEBACK
struct list_head cgwb_list;
+#endif
+
+ /* Keep the hot per-CPU stats pointer away from memory event counters. */
+ struct memcg_vmstats_percpu __percpu *vmstats_percpu
+ ____cacheline_aligned_in_smp;
+
+#ifdef CONFIG_CGROUP_WRITEBACK
struct wb_domain cgwb_domain;
struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT];
#endif
@@ -947,6 +952,8 @@ unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item);
bool memcg_stat_item_valid(int idx);
bool memcg_vm_event_item_valid(enum vm_event_item idx);
unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx);
+unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
+ enum node_stat_item idx);
unsigned long lruvec_page_state_local(struct lruvec *lruvec,
enum node_stat_item idx);
@@ -1399,6 +1406,12 @@ static inline unsigned long lruvec_page_state(struct lruvec *lruvec,
return node_page_state(lruvec_pgdat(lruvec), idx);
}
+static inline unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
+ enum node_stat_item idx)
+{
+ return node_page_state_monotonic(lruvec_pgdat(lruvec), idx);
+}
+
static inline unsigned long lruvec_page_state_local(struct lruvec *lruvec,
enum node_stat_item idx)
{
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
--- a/include/linux/mmzone.h
+++ b/include/linux/mmzone.h
@@ -323,6 +323,8 @@ enum node_stat_item {
PGSCAN_PROACTIVE,
PGSCAN_ANON,
PGSCAN_FILE,
+ PGROTATE_ANON,
+ PGROTATE_FILE,
PGREFILL,
#ifdef CONFIG_HUGETLB_PAGE
NR_HUGETLB,
@@ -755,6 +757,12 @@ void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent,
#endif /* CONFIG_LRU_GEN */
+struct lru_cost {
+ unsigned long count;
+ unsigned long last_rotated;
+ unsigned long last_io;
+};
+
struct lruvec {
struct list_head lists[NR_LRU_LISTS];
/* per lruvec lru_lock for memcg */
@@ -763,9 +771,12 @@ struct lruvec {
* These track the cost of reclaiming one LRU - file or anon -
* over the other. As the observed cost of reclaiming one LRU
* increases, the reclaim scan balance tips toward the other.
+ * Updated and decayed at prepare_scan_control() time; cost_lock
+ * serialises that update.
*/
- unsigned long anon_cost;
- unsigned long file_cost;
+ struct lru_cost cost[ANON_AND_FILE];
+ /* Protects cost[]. */
+ spinlock_t cost_lock;
/* Non-resident age, driven by LRU movement */
atomic_long_t nonresident_age;
/* Refaults at the time of last reclaim cycle */
diff --git a/include/linux/swap.h b/include/linux/swap.h
--- a/include/linux/swap.h
+++ b/include/linux/swap.h
@@ -309,9 +309,6 @@ extern unsigned long totalreserve_pages;
/* linux/mm/swap.c */
-void lru_note_cost_unlock_irq(struct lruvec *lruvec, bool file,
- unsigned int nr_io, unsigned int nr_rotated);
-void lru_note_cost_refault(struct folio *);
void folio_add_lru(struct folio *);
void folio_add_lru_vma(struct folio *, struct vm_area_struct *);
void mark_page_accessed(struct page *);
diff --git a/include/linux/vmstat.h b/include/linux/vmstat.h
--- a/include/linux/vmstat.h
+++ b/include/linux/vmstat.h
@@ -20,7 +20,6 @@ struct reclaim_stat {
unsigned nr_congested;
unsigned nr_writeback;
unsigned nr_immediate;
- unsigned nr_pageout;
unsigned nr_activate[ANON_AND_FILE];
unsigned nr_ref_keep;
unsigned nr_unmap_fail;
@@ -194,6 +193,19 @@ unsigned long global_node_page_state_pages(enum node_stat_item item)
return x;
}
+/*
+ * Non-clamping variant of global_node_page_state() intended for callers that
+ * snapshot a monotonically-incremented counter and subtract two samples.
+ * Returns the raw wrapping value so that unsigned modular subtraction stays
+ * correct across a signed-long overflow (a real hazard on 32-bit) that the
+ * clamp in global_node_page_state() would otherwise turn into a huge spurious
+ * delta. Do NOT use for non-monotonic page-count reads.
+ */
+static inline unsigned long global_node_page_state_monotonic(enum node_stat_item item)
+{
+ return (unsigned long)atomic_long_read(&vm_node_stat[item]);
+}
+
static inline unsigned long global_node_page_state(enum node_stat_item item)
{
VM_WARN_ON_ONCE(vmstat_item_in_bytes(item));
@@ -259,11 +271,14 @@ extern unsigned long node_page_state(struct pglist_data *pgdat,
enum node_stat_item item);
extern unsigned long node_page_state_pages(struct pglist_data *pgdat,
enum node_stat_item item);
+extern unsigned long node_page_state_monotonic(struct pglist_data *pgdat,
+ enum node_stat_item item);
extern void fold_vm_numa_events(void);
#else
#define sum_zone_node_page_state(node, item) global_zone_page_state(item)
#define node_page_state(node, item) global_node_page_state(item)
#define node_page_state_pages(node, item) global_node_page_state_pages(item)
+#define node_page_state_monotonic(node, item) global_node_page_state_monotonic(item)
static inline void fold_vm_numa_events(void)
{
}
diff --git a/lib/maple_tree.c b/lib/maple_tree.c
--- a/lib/maple_tree.c
+++ b/lib/maple_tree.c
@@ -1501,14 +1501,26 @@ static inline void mas_parent_gap(struct ma_state *mas, unsigned char offset,
goto ascend;
}
+static __always_inline void mas_update_gap_known(struct ma_state *mas,
+ unsigned long gap)
+{
+ unsigned char pslot;
+ unsigned long p_gap;
+
+ pslot = mte_parent_slot(mas->node);
+ p_gap = ma_gaps(mte_parent(mas->node),
+ mas_parent_type(mas, mas->node))[pslot];
+
+ if (p_gap != gap)
+ mas_parent_gap(mas, pslot, gap);
+}
+
/*
* mas_update_gap() - Update a nodes gaps and propagate up if necessary.
* @mas: the maple state.
*/
static inline void mas_update_gap(struct ma_state *mas)
{
- unsigned char pslot;
- unsigned long p_gap;
unsigned long max_gap;
if (!mt_is_alloc(mas->tree))
@@ -1518,13 +1530,7 @@ static inline void mas_update_gap(struct ma_state *mas)
return;
max_gap = mas_max_gap(mas);
-
- pslot = mte_parent_slot(mas->node);
- p_gap = ma_gaps(mte_parent(mas->node),
- mas_parent_type(mas, mas->node))[pslot];
-
- if (p_gap != max_gap)
- mas_parent_gap(mas, pslot, max_gap);
+ mas_update_gap_known(mas, max_gap);
}
/*
@@ -2081,8 +2087,8 @@ static inline void mas_wmb_replace(struct ma_state *mas, struct maple_copy *cp)
mas->node = mt_slot_locked(mas->tree, cp->slot, 0);
/* Insert the new data in the tree */
mas_topiary_replace(mas, old_enode, cp->height);
- if (!mte_is_leaf(mas->node))
- mas_update_gap(mas);
+ if (mt_is_alloc(mas->tree) && !mte_is_root(mas->node))
+ mas_update_gap_known(mas, cp->gap[0]);
mtree_range_walk(mas);
}
@@ -3125,7 +3131,7 @@ static void mas_wr_spanning_store(struct ma_wr_state *wr_mas)
static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
{
unsigned char dst_offset, offset_end;
- unsigned char copy_size, node_pivots;
+ unsigned char copy_size, node_pivots, node_slots;
struct maple_node reuse, *newnode;
unsigned long *dst_pivots;
void __rcu **dst_slots;
@@ -3138,6 +3144,7 @@ static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
in_rcu = mt_in_rcu(mas->tree);
offset_end = wr_mas->offset_end;
node_pivots = mt_pivots[wr_mas->type];
+ node_slots = mt_slots[wr_mas->type];
/* Assume last adds an entry */
new_end = mas->end + 1 - offset_end + mas->offset;
if (mas->last == wr_mas->end_piv) {
@@ -3149,7 +3156,6 @@ static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
if (in_rcu) {
newnode = mas_pop_node(mas);
} else {
- memset(&reuse, 0, sizeof(struct maple_node));
newnode = &reuse;
}
@@ -3193,7 +3199,21 @@ static inline void mas_wr_node_store(struct ma_wr_state *wr_mas)
dst_pivots[new_end] = mas->max;
done:
- mas_leaf_set_meta(newnode, maple_leaf_64, new_end);
+ if (!in_rcu && new_end + 2 < node_slots) {
+ unsigned char clear_from = new_end + 1;
+
+ /*
+ * Note that the last slot is never cleared, since the metadata
+ * will be stored there or it has a value.
+ */
+ memset(dst_slots + clear_from, 0,
+ sizeof(void __rcu *) * (node_slots - clear_from));
+ if (clear_from < node_pivots)
+ memset(dst_pivots + clear_from, 0,
+ sizeof(unsigned long) * (node_pivots - clear_from));
+ }
+
+ mas_leaf_set_meta(newnode, wr_mas->type, new_end);
if (in_rcu) {
struct maple_enode *old_enode = mas->node;
@@ -3218,7 +3238,7 @@ static inline void mas_wr_slot_store(struct ma_wr_state *wr_mas)
void __rcu **slots = wr_mas->slots;
bool gap = false;
- gap |= !mt_slot_locked(mas->tree, slots, offset);
+ gap |= !wr_mas->content;
gap |= !mt_slot_locked(mas->tree, slots, offset + 1);
if (wr_mas->offset_end - offset == 1) {
@@ -3661,6 +3681,9 @@ static inline enum store_type mas_wr_store_type(struct ma_wr_state *wr_mas)
{
struct ma_state *mas = wr_mas->mas;
unsigned char new_end;
+ bool appending;
+ bool one_slot;
+ bool in_rcu;
if (unlikely(mas_is_none(mas) || mas_is_ptr(mas)))
return wr_store_root;
@@ -3680,21 +3703,30 @@ static inline enum store_type mas_wr_store_type(struct ma_wr_state *wr_mas)
return wr_new_root;
new_end = mas_wr_new_end(wr_mas);
+ in_rcu = mt_in_rcu(mas->tree);
+ appending = mas->offset == mas->end;
+ one_slot = wr_mas->offset_end - mas->offset == 1;
+
/* Potential spanning rebalance collapsing a node */
if (new_end < mt_min_slots[wr_mas->type]) {
if (!mte_is_root(mas->node))
return wr_rebalance;
+ if (!in_rcu) {
+ if (appending)
+ return wr_append;
+ else if (mas->end == new_end && one_slot)
+ return wr_slot_store;
+ }
return wr_node_store;
}
if (new_end >= mt_slots[wr_mas->type])
return wr_split_store;
- if (!mt_in_rcu(mas->tree) && (mas->offset == mas->end))
+ if (!in_rcu && appending)
return wr_append;
- if ((new_end == mas->end) && (!mt_in_rcu(mas->tree) ||
- (wr_mas->offset_end - mas->offset == 1)))
+ if (new_end == mas->end && (!in_rcu || one_slot))
return wr_slot_store;
return wr_node_store;
@@ -3793,35 +3825,40 @@ int mas_alloc_cyclic(struct ma_state *mas, unsigned long *startp,
void *entry, unsigned long range_lo, unsigned long range_hi,
unsigned long *next, gfp_t gfp)
{
- unsigned long min = range_lo;
- int ret = 0;
+ int ret;
+ unsigned long min;
+
+ min = range_lo;
+ do {
+ range_lo = max(min, *next);
+ ret = mas_empty_area(mas, range_lo, range_hi, 1);
+ if (ret < 0 && range_lo > min) {
+ mas_reset(mas);
+ ret = mas_empty_area(mas, min, range_hi, 1);
+ if (ret == 0)
+ ret = 1;
+ }
+ if (ret < 0)
+ goto out;
+
+ mas_insert(mas, entry);
+ } while (mas_nomem(mas, gfp));
+
+ if (mas_is_err(mas)) {
+ ret = xa_err(mas->node);
+ goto out;
+ }
- range_lo = max(min, *next);
- ret = mas_empty_area(mas, range_lo, range_hi, 1);
if ((mas->tree->ma_flags & MT_FLAGS_ALLOC_WRAPPED) && ret == 0) {
mas->tree->ma_flags &= ~MT_FLAGS_ALLOC_WRAPPED;
ret = 1;
}
- if (ret < 0 && range_lo > min) {
- mas_reset(mas);
- ret = mas_empty_area(mas, min, range_hi, 1);
- if (ret == 0)
- ret = 1;
- }
- if (ret < 0)
- return ret;
-
- do {
- mas_insert(mas, entry);
- } while (mas_nomem(mas, gfp));
- if (mas_is_err(mas))
- return xa_err(mas->node);
-
*startp = mas->index;
*next = *startp + 1;
if (*next == 0)
mas->tree->ma_flags |= MT_FLAGS_ALLOC_WRAPPED;
+out:
mas_destroy(mas);
return ret;
}
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -1442,10 +1442,10 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s
result = SCAN_SUCCEED;
out_up_write:
- if (anon_vma_locked)
- anon_vma_unlock_write(vma->anon_vma);
if (pte)
pte_unmap(pte);
+ if (anon_vma_locked)
+ anon_vma_unlock_write(vma->anon_vma);
mmap_write_unlock(mm);
out_nolock:
if (folio)
diff --git a/mm/ksm.c b/mm/ksm.c
--- a/mm/ksm.c
+++ b/mm/ksm.c
@@ -3061,10 +3061,9 @@ int __ksm_enter(struct mm_struct *mm)
slot = &mm_slot->slot;
+ spin_lock(&ksm_mmlist_lock);
/* Check ksm_run too? Would need tighter locking */
needs_wakeup = list_empty(&ksm_mm_head.slot.mm_node);
-
- spin_lock(&ksm_mmlist_lock);
mm_slot_insert(mm_slots_hash, mm, slot);
/*
* When KSM_RUN_MERGE (or KSM_RUN_STOP),
diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c
--- a/mm/memcontrol-v1.c
+++ b/mm/memcontrol-v1.c
@@ -1998,8 +1998,8 @@ void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s)
for_each_online_pgdat(pgdat) {
mz = memcg->nodeinfo[pgdat->node_id];
- anon_cost += mz->lruvec.anon_cost;
- file_cost += mz->lruvec.file_cost;
+ anon_cost += mz->lruvec.cost[WORKINGSET_ANON].count;
+ file_cost += mz->lruvec.cost[WORKINGSET_FILE].count;
}
seq_buf_printf(s, "anon_cost %lu\n", anon_cost);
seq_buf_printf(s, "file_cost %lu\n", file_cost);
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -393,6 +393,7 @@ static const unsigned int memcg_node_stat_items[] = {
NR_SHMEM_THPS,
NR_FILE_THPS,
NR_ANON_THPS,
+ NR_VMSCAN_WRITE,
NR_VMALLOC,
NR_KERNEL_STACK_KB,
NR_PAGETABLE,
@@ -419,6 +420,8 @@ static const unsigned int memcg_node_stat_items[] = {
PGSCAN_PROACTIVE,
PGSCAN_ANON,
PGSCAN_FILE,
+ PGROTATE_ANON,
+ PGROTATE_FILE,
PGREFILL,
#ifdef CONFIG_HUGETLB_PAGE
NR_HUGETLB,
@@ -502,6 +505,42 @@ unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx)
return x;
}
+/**
+ * lruvec_page_state_monotonic - non-clamping lruvec stat read for delta sampling
+ * @lruvec: the LRU vector to read from
+ * @idx: the node_stat_item to read
+ *
+ * Returns the raw state[idx] value cast to unsigned long, skipping the
+ * clamp-negative-to-zero step in lruvec_page_state(). Intended for callers
+ * that snapshot a monotonically-incremented counter and subtract two
+ * samples: unsigned modular arithmetic then yields the correct delta across
+ * a signed-long wraparound (a real hazard on 32-bit) that the clamp would
+ * otherwise turn into a huge spurious delta.
+ *
+ * Do NOT use for non-monotonic page-count reads where a transient negative
+ * reading from per-CPU delta skew must present as zero.
+ *
+ * XXX: This helper (and its node/global peers) exists because some
+ * monotonically-incremented event counters are stored in
+ * enum node_stat_item.
+ */
+unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec,
+ enum node_stat_item idx)
+{
+ struct mem_cgroup_per_node *pn;
+ int i;
+
+ if (mem_cgroup_disabled())
+ return node_page_state_monotonic(lruvec_pgdat(lruvec), idx);
+
+ i = memcg_stats_index(idx);
+ if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx))
+ return 0;
+
+ pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec);
+ return (unsigned long)READ_ONCE(pn->lruvec_stats->state[i]);
+}
+
unsigned long lruvec_page_state_local(struct lruvec *lruvec,
enum node_stat_item idx)
{
@@ -2095,7 +2134,12 @@ static bool consume_stock(struct mem_cgroup *memcg, unsigned int nr_pages)
stock_pages = READ_ONCE(stock->nr_pages[i]);
if (stock_pages >= nr_pages) {
- WRITE_ONCE(stock->nr_pages[i], stock_pages - nr_pages);
+ stock_pages -= nr_pages;
+ WRITE_ONCE(stock->nr_pages[i], stock_pages);
+ if (!stock_pages) {
+ css_put(&memcg->css);
+ WRITE_ONCE(stock->cached[i], NULL);
+ }
ret = true;
}
break;
diff --git a/mm/migrate_device.c b/mm/migrate_device.c
--- a/mm/migrate_device.c
+++ b/mm/migrate_device.c
@@ -872,7 +872,7 @@ static int migrate_vma_insert_huge_pmd_page(struct migrate_vma *migrate,
if (flush) {
pte_free(vma->vm_mm, pgtable);
- flush_cache_page(vma, addr, addr + HPAGE_PMD_SIZE);
+ flush_cache_range(vma, addr, addr + HPAGE_PMD_SIZE);
pmdp_invalidate(vma, addr, pmdp);
} else {
pgtable_trans_huge_deposit(vma->vm_mm, pmdp, pgtable);
diff --git a/mm/mmzone.c b/mm/mmzone.c
--- a/mm/mmzone.c
+++ b/mm/mmzone.c
@@ -78,6 +78,7 @@ void lruvec_init(struct lruvec *lruvec)
memset(lruvec, 0, sizeof(struct lruvec));
spin_lock_init(&lruvec->lru_lock);
+ spin_lock_init(&lruvec->cost_lock);
zswap_lruvec_state_init(lruvec);
for_each_lru(lru)
diff --git a/mm/mremap.c b/mm/mremap.c
--- a/mm/mremap.c
+++ b/mm/mremap.c
@@ -264,7 +264,7 @@ static int move_ptes(struct pagetable_move_control *pmc,
for (; old_addr < old_end; old_ptep += nr_ptes, old_addr += nr_ptes * PAGE_SIZE,
new_ptep += nr_ptes, new_addr += nr_ptes * PAGE_SIZE) {
- VM_WARN_ON_ONCE(!pte_none(*new_ptep));
+ VM_WARN_ON_ONCE(!pte_none(ptep_get(new_ptep)));
nr_ptes = 1;
max_nr_ptes = (old_end - old_addr) >> PAGE_SHIFT;
diff --git a/mm/shmem.c b/mm/shmem.c
--- a/mm/shmem.c
+++ b/mm/shmem.c
@@ -3612,6 +3612,7 @@ static long shmem_fallocate(struct file *file, int mode, loff_t offset,
struct shmem_inode_info *info = SHMEM_I(inode);
struct shmem_falloc shmem_falloc;
pgoff_t start, index, end, undo_fallocend;
+ loff_t aligned_end;
int error;
if (mode & ~(FALLOC_FL_KEEP_SIZE | FALLOC_FL_PUNCH_HOLE))
@@ -3668,8 +3669,15 @@ static long shmem_fallocate(struct file *file, int mode, loff_t offset,
goto out;
}
+ /* Check for wraparound */
+ if (check_add_overflow(offset + len, (loff_t)PAGE_SIZE - 1,
+ &aligned_end)) {
+ error = -EFBIG;
+ goto out;
+ }
+
start = offset >> PAGE_SHIFT;
- end = (offset + len + PAGE_SIZE - 1) >> PAGE_SHIFT;
+ end = aligned_end >> PAGE_SHIFT;
/* Try to avoid a swapstorm if len is impossible to satisfy */
if (sbinfo->max_blocks && end - start > sbinfo->max_blocks) {
error = -ENOSPC;
diff --git a/mm/swap.c b/mm/swap.c
--- a/mm/swap.c
+++ b/mm/swap.c
@@ -272,73 +272,6 @@ void folio_rotate_reclaimable(struct folio *folio)
folio_batch_add_and_move(folio, lru_move_tail);
}
-void lru_note_cost_unlock_irq(struct lruvec *lruvec, bool file,
- unsigned int nr_io, unsigned int nr_rotated)
- __releases(lruvec->lru_lock)
- __releases(rcu)
-{
- unsigned long cost;
-
- /*
- * Reflect the relative cost of incurring IO and spending CPU
- * time on rotations. This doesn't attempt to make a precise
- * comparison, it just says: if reloads are about comparable
- * between the LRU lists, or rotations are overwhelmingly
- * different between them, adjust scan balance for CPU work.
- */
- cost = nr_io * SWAP_CLUSTER_MAX + nr_rotated;
- if (!cost) {
- spin_unlock_irq(&lruvec->lru_lock);
- rcu_read_unlock();
- return;
- }
-
- for (;;) {
- unsigned long lrusize;
-
- /* Record cost event */
- if (file)
- lruvec->file_cost += cost;
- else
- lruvec->anon_cost += cost;
-
- /*
- * Decay previous events
- *
- * Because workloads change over time (and to avoid
- * overflow) we keep these statistics as a floating
- * average, which ends up weighing recent refaults
- * more than old ones.
- */
- lrusize = lruvec_page_state(lruvec, NR_INACTIVE_ANON) +
- lruvec_page_state(lruvec, NR_ACTIVE_ANON) +
- lruvec_page_state(lruvec, NR_INACTIVE_FILE) +
- lruvec_page_state(lruvec, NR_ACTIVE_FILE);
-
- if (lruvec->file_cost + lruvec->anon_cost > lrusize / 4) {
- lruvec->file_cost /= 2;
- lruvec->anon_cost /= 2;
- }
-
- spin_unlock_irq(&lruvec->lru_lock);
- lruvec = parent_lruvec(lruvec);
- if (!lruvec) {
- rcu_read_unlock();
- break;
- }
- spin_lock_irq(&lruvec->lru_lock);
- }
-}
-
-void lru_note_cost_refault(struct folio *folio)
-{
- struct lruvec *lruvec;
-
- lruvec = folio_lruvec_lock_irq(folio);
- lru_note_cost_unlock_irq(lruvec, folio_is_file_lru(folio),
- folio_nr_pages(folio), 0);
-}
-
static void lru_activate(struct lruvec *lruvec, struct folio *folio)
{
long nr_pages = folio_nr_pages(folio);
@@ -1153,7 +1086,16 @@ static void lruvec_reparent_lru(struct lruvec *child_lruvec,
for_each_managed_zone_pgdat(zone, NODE_DATA(nid), zid, MAX_NR_ZONES - 1) {
unsigned long size = mem_cgroup_get_zone_lru_size(child_lruvec, lru, zid);
+ if (!size)
+ continue;
+
+ /*
+ * The folios are accounted to the parent from now on, so the
+ * size has to be moved, not just copied. Leaving it behind
+ * makes the dying child describe folios it no longer owns.
+ */
mem_cgroup_update_lru_size(parent_lruvec, lru, zid, size);
+ mem_cgroup_update_lru_size(child_lruvec, lru, zid, -(long)size);
}
}
@@ -1164,8 +1106,6 @@ void lru_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int
child_lruvec = mem_cgroup_lruvec(memcg, NODE_DATA(nid));
parent_lruvec = mem_cgroup_lruvec(parent, NODE_DATA(nid));
- parent_lruvec->anon_cost += child_lruvec->anon_cost;
- parent_lruvec->file_cost += child_lruvec->file_cost;
for_each_lru(lru)
lruvec_reparent_lru(child_lruvec, parent_lruvec, lru, nid);
diff --git a/mm/vma.c b/mm/vma.c
--- a/mm/vma.c
+++ b/mm/vma.c
@@ -1866,10 +1866,10 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap,
bool *need_rmap_locks)
{
struct vm_area_struct *vma = *vmap;
- unsigned long vma_start = vma->vm_start;
+ unsigned long old_vma_start = vma->vm_start;
struct mm_struct *mm = vma->vm_mm;
struct vm_area_struct *new_vma;
- bool faulted_in_anon_vma = true;
+ bool can_self_merge = false;
VMA_ITERATOR(vmi, mm, addr);
VMG_VMA_STATE(vmg, &vmi, NULL, vma, addr, addr + len);
@@ -1879,7 +1879,7 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap,
*/
if (unlikely(vma_is_anonymous(vma) && !vma->anon_vma)) {
pgoff = addr >> PAGE_SHIFT;
- faulted_in_anon_vma = false;
+ can_self_merge = true;
}
/*
@@ -1899,24 +1899,21 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap,
new_vma = vma_merge_copied_range(&vmg);
if (new_vma) {
- /*
- * Source vma may have been merged into new_vma
- */
- if (unlikely(vma_start >= new_vma->vm_start &&
- vma_start < new_vma->vm_end)) {
+ /* Self-merged and VMA replaced. */
+ if (unlikely(new_vma->vm_start < old_vma_start &&
+ new_vma->vm_end > old_vma_start)) {
/*
- * The only way we can get a vma_merge with
- * self during an mremap is if the vma hasn't
- * been faulted in yet and we were allowed to
- * reset the dst vma->vm_pgoff to the
- * destination address of the mremap to allow
- * the merge to happen. mremap must change the
- * vm_pgoff linearity between src and dst vmas
- * (in turn preventing a vma_merge) to be
- * safe. It is only safe to keep the vm_pgoff
- * linear if there are no pages mapped yet.
+ * The only way a VMA can both self-merge and be
+ * replaced is if the remap places the new VMA
+ * immediately prior to its old self ('next') and
+ * immediately after another VMA ('prev') causing the
+ * next to be removed and prev to be expanded to cover
+ * the entire range.
+ *
+ * This should only be possible if the page offset was
+ * updated, i.e. the VMA is unfaulted.
*/
- VM_BUG_ON_VMA(faulted_in_anon_vma, new_vma);
+ VM_WARN_ON_ONCE_VMA(!can_self_merge, new_vma);
*vmap = vma = new_vma;
}
*need_rmap_locks = (new_vma->vm_pgoff <= vma->vm_pgoff);
diff --git a/mm/vmscan.c b/mm/vmscan.c
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -669,7 +669,7 @@ static pageout_t pageout(struct folio *folio, struct address_space *mapping,
folio_clear_reclaim(folio);
trace_mm_vmscan_write_folio(folio);
- node_stat_add_folio(folio, NR_VMSCAN_WRITE);
+ lruvec_stat_mod_folio(folio, NR_VMSCAN_WRITE, folio_nr_pages(folio));
return PAGE_SUCCESS;
}
@@ -1420,8 +1420,6 @@ static unsigned int shrink_folio_list(struct list_head *folio_list,
sc->nr_scanned -= (nr_pages - 1);
nr_pages = 1;
}
- stat->nr_pageout += nr_pages;
-
if (folio_test_writeback(folio))
goto keep;
if (folio_test_dirty(folio))
@@ -2045,10 +2043,10 @@ static unsigned long shrink_inactive_list(unsigned long nr_to_scan,
item = PGSTEAL_KSWAPD + reclaimer_offset(sc);
mod_lruvec_state(lruvec, item, nr_reclaimed);
mod_lruvec_state(lruvec, PGSTEAL_ANON + file, nr_reclaimed);
+ if (nr_scanned > nr_reclaimed)
+ mod_lruvec_state(lruvec, PGROTATE_ANON + file,
+ nr_scanned - nr_reclaimed);
- lruvec_lock_irq(lruvec);
- lru_note_cost_unlock_irq(lruvec, file, stat.nr_pageout,
- nr_scanned - nr_reclaimed);
handle_reclaim_writeback(nr_taken, pgdat, sc, &stat);
trace_mm_vmscan_lru_shrink_inactive(pgdat->node_id,
nr_scanned, nr_reclaimed, &stat, sc->priority, file);
@@ -2154,9 +2152,9 @@ static void shrink_active_list(unsigned long nr_to_scan,
count_vm_events(PGDEACTIVATE, nr_deactivate);
count_memcg_events(lruvec_memcg(lruvec), PGDEACTIVATE, nr_deactivate);
mod_node_page_state(pgdat, NR_ISOLATED_ANON + file, -nr_taken);
+ if (nr_rotated)
+ mod_lruvec_state(lruvec, PGROTATE_ANON + file, nr_rotated);
- lruvec_lock_irq(lruvec);
- lru_note_cost_unlock_irq(lruvec, file, 0, nr_rotated);
trace_mm_vmscan_lru_shrink_active(pgdat->node_id, nr_taken, nr_activate,
nr_deactivate, nr_rotated, sc->priority, file);
}
@@ -2289,8 +2287,10 @@ enum scan_balance {
static void prepare_scan_control(pg_data_t *pgdat, struct scan_control *sc)
{
- unsigned long file;
+ struct lru_cost *anon_cost, *file_cost;
struct lruvec *target_lruvec;
+ unsigned long lrusize;
+ unsigned long file;
if (lru_gen_enabled() && !lru_gen_switching())
return;
@@ -2306,11 +2306,69 @@ static void prepare_scan_control(pg_data_t *pgdat, struct scan_control *sc)
/*
* Determine the scan balance between anon and file LRUs.
+ *
+ * The cost model is based on rotations, refaults and
+ * reclaim-driven writes (anon only) on each side.
+ *
+ * These event counters are monotonic, so each reclaim cycle
+ * the delta since the last scan is extracted and incorporated
+ * into a decaying average. This ensures currency, as workloads
+ * change over time, and avoids overflow in the calculations.
+ *
+ * Use lruvec_page_state_monotonic() so unsigned subtraction
+ * yields the correct delta across a signed-long wraparound of
+ * the underlying counter (a real hazard on 32-bit that the
+ * clamp in lruvec_page_state() would otherwise turn into a huge
+ * spurious delta).
*/
- spin_lock_irq(&target_lruvec->lru_lock);
- sc->anon_cost = target_lruvec->anon_cost;
- sc->file_cost = target_lruvec->file_cost;
- spin_unlock_irq(&target_lruvec->lru_lock);
+ spin_lock(&target_lruvec->cost_lock);
+
+ for (int f = 0; f <= 1; f++) {
+ struct lru_cost *cost = &target_lruvec->cost[f];
+ unsigned long rotated, io, nr_rotated, nr_io;
+
+ rotated = lruvec_page_state_monotonic(target_lruvec,
+ PGROTATE_ANON + f);
+ io = lruvec_page_state_monotonic(target_lruvec,
+ WORKINGSET_RESTORE_BASE + f);
+ if (f == WORKINGSET_ANON)
+ io += lruvec_page_state_monotonic(target_lruvec,
+ NR_VMSCAN_WRITE);
+
+ nr_rotated = rotated - cost->last_rotated;
+ nr_io = io - cost->last_io;
+
+ /*
+ * Reflect the relative cost of incurring IO and spending
+ * CPU time on rotations. This doesn't attempt to make a
+ * precise comparison, it just says: if reloads are about
+ * comparable between the LRU lists, or rotations are
+ * overwhelmingly different between them, adjust scan
+ * balance for CPU work.
+ */
+ cost->count += nr_io * SWAP_CLUSTER_MAX + nr_rotated;
+
+ cost->last_rotated = rotated;
+ cost->last_io = io;
+ }
+
+ anon_cost = &target_lruvec->cost[WORKINGSET_ANON];
+ file_cost = &target_lruvec->cost[WORKINGSET_FILE];
+
+ lrusize = lruvec_page_state(target_lruvec, NR_INACTIVE_ANON) +
+ lruvec_page_state(target_lruvec, NR_ACTIVE_ANON) +
+ lruvec_page_state(target_lruvec, NR_INACTIVE_FILE) +
+ lruvec_page_state(target_lruvec, NR_ACTIVE_FILE);
+
+ while (anon_cost->count + file_cost->count > lrusize / 4) {
+ anon_cost->count /= 2;
+ file_cost->count /= 2;
+ }
+
+ sc->anon_cost = anon_cost->count;
+ sc->file_cost = file_cost->count;
+
+ spin_unlock(&target_lruvec->cost_lock);
/*
* Target desirable inactive:active list ratios for the anon
@@ -4197,7 +4255,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
unsigned long end;
struct lru_gen_mm_walk *walk;
struct folio *last = NULL;
- int young = 1;
+ int young = nr;
pte_t *pte = pvmw->pte;
unsigned long addr = pvmw->address;
struct vm_area_struct *vma = pvmw->vma;
@@ -4571,7 +4629,12 @@ void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent,
for_each_managed_zone_pgdat(zone, NODE_DATA(nid), zid, MAX_NR_ZONES - 1) {
unsigned long size = mem_cgroup_get_zone_lru_size(child_lruvec, lru, zid);
+ if (!size)
+ continue;
+
+ /* Move the accounting, do not duplicate it. */
mem_cgroup_update_lru_size(parent_lruvec, lru, zid, size);
+ mem_cgroup_update_lru_size(child_lruvec, lru, zid, -(long)size);
}
}
}
@@ -4814,7 +4877,8 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
struct reclaim_stat stat;
struct lru_gen_mm_walk *walk;
int scanned, reclaimed;
- int isolated = 0, type, type_scanned;
+ int isolated = 0, nr_isolated = 0, type, type_scanned;
+ unsigned long total_reclaimed = 0;
bool skip_retry = false;
struct mem_cgroup *memcg = lruvec_memcg(lruvec);
struct pglist_data *pgdat = lruvec_pgdat(lruvec);
@@ -4826,6 +4890,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
scanned = isolate_folios(nr_to_scan, lruvec, sc, swappiness,
&list, &isolated, &type, &type_scanned);
+ nr_isolated = isolated;
/* Scanning may have emptied the oldest gen, flush it */
if (scanned)
@@ -4838,6 +4903,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
retry:
reclaimed = shrink_folio_list(&list, pgdat, sc, &stat, false, memcg);
sc->nr_reclaimed += reclaimed;
+ total_reclaimed += reclaimed;
/* Retry pass is only meant for clean folios without new isolation */
if (isolated)
handle_reclaim_writeback(isolated, pgdat, sc, &stat);
@@ -4887,6 +4953,10 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec,
goto retry;
}
+ if (nr_isolated > total_reclaimed)
+ mod_lruvec_state(lruvec, PGROTATE_ANON + type,
+ nr_isolated - total_reclaimed);
+
return scanned;
}
diff --git a/mm/vmstat.c b/mm/vmstat.c
--- a/mm/vmstat.c
+++ b/mm/vmstat.c
@@ -1024,6 +1024,17 @@ unsigned long node_page_state(struct pglist_data *pgdat,
return node_page_state_pages(pgdat, item);
}
+
+/*
+ * Non-clamping variant of node_page_state() intended for callers that
+ * snapshot a monotonically-incremented counter and subtract two samples.
+ * See global_node_page_state_monotonic() for the rationale.
+ */
+unsigned long node_page_state_monotonic(struct pglist_data *pgdat,
+ enum node_stat_item item)
+{
+ return (unsigned long)atomic_long_read(&pgdat->vm_stat[item]);
+}
#endif
/*
@@ -1289,6 +1300,8 @@ const char * const vmstat_text[] = {
[I(PGSCAN_PROACTIVE)] = "pgscan_proactive",
[I(PGSCAN_ANON)] = "pgscan_anon",
[I(PGSCAN_FILE)] = "pgscan_file",
+ [I(PGROTATE_ANON)] = "pgrotate_anon",
+ [I(PGROTATE_FILE)] = "pgrotate_file",
[I(PGREFILL)] = "pgrefill",
#ifdef CONFIG_HUGETLB_PAGE
[I(NR_HUGETLB)] = "nr_hugetlb",
diff --git a/mm/workingset.c b/mm/workingset.c
--- a/mm/workingset.c
+++ b/mm/workingset.c
@@ -584,11 +584,6 @@ void workingset_refault(struct folio *folio, void *shadow)
/* Folio was active prior to eviction */
if (workingset) {
folio_set_workingset(folio);
- /*
- * XXX: Move to folio_add_lru() when it supports new vs
- * putback
- */
- lru_note_cost_refault(folio);
mod_lruvec_state(lruvec, WORKINGSET_RESTORE_BASE + file, nr);
}
out:
diff --git a/mm/zswap.c b/mm/zswap.c
--- a/mm/zswap.c
+++ b/mm/zswap.c
@@ -1217,7 +1217,7 @@ static unsigned long zswap_shrinker_count(struct shrinker *shrinker,
* Without memcg, use the zswap pool-wide metrics.
*/
if (!mem_cgroup_disabled()) {
- mem_cgroup_flush_stats(memcg);
+ mem_cgroup_flush_stats_ratelimited(memcg);
nr_backing = memcg_page_state(memcg, MEMCG_ZSWAP_B) >> PAGE_SHIFT;
nr_stored = memcg_page_state(memcg, MEMCG_ZSWAPPED);
} else {
@@ -1275,6 +1275,14 @@ static struct shrinker *zswap_alloc_shrinker(void)
return shrinker;
}
+/*
+ * Scan up to SWAP_CLUSTER_MAX pages on each per-node zswap LRU of @memcg
+ * and write back the reclaimable ones.
+ *
+ * Return: 0 if at least one entry was written back, -EAGAIN if entries
+ * were scanned but none could be written back, or -ENOENT if @memcg has
+ * writeback disabled, is a zombie cgroup, or has empty zswap LRUs.
+ */
static int shrink_memcg(struct mem_cgroup *memcg)
{
int nid, shrunk = 0, scanned = 0;
@@ -1290,13 +1298,14 @@ static int shrink_memcg(struct mem_cgroup *memcg)
return -ENOENT;
for_each_node_state(nid, N_NORMAL_MEMORY) {
- unsigned long nr_to_walk = 1;
+ unsigned long nr_to_walk = SWAP_CLUSTER_MAX;
shrunk += list_lru_walk_one(&zswap_list_lru, nid, memcg,
&shrink_memcg_cb, NULL, &nr_to_walk);
- scanned += 1 - nr_to_walk;
+ scanned += SWAP_CLUSTER_MAX - nr_to_walk;
}
+ /* Nothing was scanned: every LRU under @memcg was empty. */
if (!scanned)
return -ENOENT;