382 lines
15 KiB
Diff
382 lines
15 KiB
Diff
diff --git c/include/linux/mmzone.h i/include/linux/mmzone.h
|
|
--- c/include/linux/mmzone.h
|
|
+++ i/include/linux/mmzone.h
|
|
@@ -1465,6 +1465,39 @@ struct memory_failure_stats {
|
|
};
|
|
#endif
|
|
|
|
+/*
|
|
+ * Per-pgdat state machine for the kswapd "opportunistic compaction" hint.
|
|
+ *
|
|
+ * wakeup_kswapd() collapses the gfp flags of all wakers that arrive between
|
|
+ * two kswapd runs into a single tri-state, which kswapd then forwards to the
|
|
+ * shrinkers via shrink_control::opportunistic_compaction:
|
|
+ *
|
|
+ * KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION
|
|
+ * Initial state after kswapd consumes the previous value. No waker has
|
|
+ * been observed yet for the upcoming run.
|
|
+ *
|
|
+ * KSWAPD_NO_OPPORTUNISTIC_COMPACTION
|
|
+ * At least one waker is an order-0 allocation, or a high-order
|
|
+ * allocation that cannot tolerate failure (i.e., not eligible for
|
|
+ * opportunistic behaviour). Shrinkers must do their normal best-effort
|
|
+ * work; the hint is cleared.
|
|
+ *
|
|
+ * KSWAPD_OPPORTUNISTIC_COMPACTION
|
|
+ * All wakers seen so far are high-order allocations that may fail
|
|
+ * (__GFP_NORETRY or __GFP_RETRY_MAYFAIL, without __GFP_NOFAIL). Shrinkers
|
|
+ * may skip work that is unlikely to produce a contiguous high-order
|
|
+ * block (e.g., evicting working-set pages).
|
|
+ *
|
|
+ * The state is sticky in the "NO" direction within a single kswapd run: once
|
|
+ * any non-eligible waker is observed, subsequent eligible wakers cannot
|
|
+ * upgrade it back to KSWAPD_OPPORTUNISTIC_COMPACTION.
|
|
+ */
|
|
+enum kswapd_opportunistic_compaction_type {
|
|
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION = 0,
|
|
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION,
|
|
+ KSWAPD_OPPORTUNISTIC_COMPACTION,
|
|
+};
|
|
+
|
|
/*
|
|
* On NUMA machines, each NUMA node would have a pg_data_t to describe
|
|
* it's memory layout. On UMA machines there is a single pglist_data which
|
|
@@ -1529,6 +1562,13 @@ typedef struct pglist_data {
|
|
#endif
|
|
struct task_struct *kswapd; /* Protected by kswapd_lock */
|
|
int kswapd_order;
|
|
+ /*
|
|
+ * Aggregated opportunistic-compaction hint for the next kswapd run.
|
|
+ * Updated by wakeup_kswapd() based on the gfp flags / order of each
|
|
+ * waker, and consumed (and reset) by kswapd before balance_pgdat().
|
|
+ * See enum kswapd_opportunistic_compaction_type for the state machine.
|
|
+ */
|
|
+ atomic_t kswapd_opportunistic_compaction;
|
|
enum zone_type kswapd_highest_zoneidx;
|
|
|
|
atomic_t kswapd_failures; /* Number of 'reclaimed == 0' runs */
|
|
diff --git c/include/linux/shrinker.h i/include/linux/shrinker.h
|
|
--- c/include/linux/shrinker.h
|
|
+++ i/include/linux/shrinker.h
|
|
@@ -37,6 +37,26 @@ struct shrink_control {
|
|
/* current node being shrunk (for NUMA aware shrinkers) */
|
|
int nid;
|
|
|
|
+ /*
|
|
+ * Opportunistic compaction hint.
|
|
+ *
|
|
+ * Set by the reclaim path to tell shrinkers that this pass is
|
|
+ * driven by an order > 0 allocation that the caller is willing to
|
|
+ * have fail (e.g., __GFP_NORETRY / __GFP_RETRY_MAYFAIL without
|
|
+ * __GFP_NOFAIL). Such allocations only really benefit from
|
|
+ * shrinking when doing so frees up a contiguous, high-order block;
|
|
+ * thrashing working sets in the hope of producing one is typically
|
|
+ * counter-productive.
|
|
+ *
|
|
+ * Shrinkers that can produce naturally-aligned high-order folios
|
|
+ * (see shrink_control::order) should treat this as a hint to skip
|
|
+ * costly work that is unlikely to help compaction (for example,
|
|
+ * evicting hot/working-set pages just to free single pages).
|
|
+ *
|
|
+ * Only meaningful when @order > 0; ignored otherwise.
|
|
+ */
|
|
+ bool opportunistic_compaction;
|
|
+
|
|
/*
|
|
* How many objects scan_objects should scan and try to reclaim.
|
|
* This is reset before every call, so it is safe for callees
|
|
diff --git c/mm/internal.h i/mm/internal.h
|
|
--- c/mm/internal.h
|
|
+++ i/mm/internal.h
|
|
@@ -1767,7 +1767,7 @@ void __meminit __init_page_from_nid(unsigned long pfn, int nid);
|
|
|
|
/* shrinker related functions */
|
|
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
|
|
- int priority);
|
|
+ int priority, bool opportunistic_compaction);
|
|
|
|
int shmem_add_to_page_cache(struct folio *folio,
|
|
struct address_space *mapping,
|
|
diff --git c/mm/shrinker.c i/mm/shrinker.c
|
|
--- c/mm/shrinker.c
|
|
+++ i/mm/shrinker.c
|
|
@@ -476,7 +476,7 @@ static unsigned long do_shrink_slab(struct shrink_control *shrinkctl,
|
|
|
|
#ifdef CONFIG_MEMCG
|
|
static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
|
|
- struct mem_cgroup *memcg, int priority)
|
|
+ struct mem_cgroup *memcg, int priority, bool opportunistic_compaction)
|
|
{
|
|
struct shrinker_info *info;
|
|
unsigned long ret, freed = 0;
|
|
@@ -538,6 +538,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
|
|
.gfp_mask = gfp_mask,
|
|
.nid = nid,
|
|
.memcg = memcg,
|
|
+ .opportunistic_compaction = opportunistic_compaction,
|
|
};
|
|
struct shrinker *shrinker;
|
|
int shrinker_id = calc_shrinker_id(index, offset);
|
|
@@ -597,7 +598,8 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
|
|
}
|
|
#else /* !CONFIG_MEMCG */
|
|
static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
|
|
- struct mem_cgroup *memcg, int priority)
|
|
+ struct mem_cgroup *memcg, int priority,
|
|
+ bool opportunistic_compaction)
|
|
{
|
|
return 0;
|
|
}
|
|
@@ -609,6 +611,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
|
|
* @nid: node whose slab caches to target
|
|
* @memcg: memory cgroup whose slab caches to target
|
|
* @priority: the reclaim priority
|
|
+ * @opportunistic_compaction: do compaction opportunistically (e.g., do not swap working sets)
|
|
*
|
|
* Call the shrink functions to age shrinkable caches.
|
|
*
|
|
@@ -624,7 +627,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
|
|
* Returns the number of reclaimed slab objects.
|
|
*/
|
|
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
|
|
- int priority)
|
|
+ int priority, bool opportunistic_compaction)
|
|
{
|
|
unsigned long ret, freed = 0;
|
|
struct shrinker *shrinker;
|
|
@@ -637,7 +640,8 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
|
|
* oom.
|
|
*/
|
|
if (!mem_cgroup_disabled() && !mem_cgroup_is_root(memcg))
|
|
- return shrink_slab_memcg(gfp_mask, nid, memcg, priority);
|
|
+ return shrink_slab_memcg(gfp_mask, nid, memcg, priority,
|
|
+ opportunistic_compaction);
|
|
|
|
/*
|
|
* lockless algorithm of global shrink.
|
|
@@ -666,6 +670,7 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
|
|
.gfp_mask = gfp_mask,
|
|
.nid = nid,
|
|
.memcg = memcg,
|
|
+ .opportunistic_compaction = opportunistic_compaction,
|
|
};
|
|
|
|
if (!shrinker_try_get(shrinker))
|
|
diff --git c/mm/vmscan.c i/mm/vmscan.c
|
|
--- c/mm/vmscan.c
|
|
+++ i/mm/vmscan.c
|
|
@@ -96,6 +96,14 @@ struct scan_control {
|
|
/* Swappiness value for proactive reclaim. Always use sc_swappiness()! */
|
|
int *proactive_swappiness;
|
|
|
|
+ /*
|
|
+ * Opportunistic compaction hint snapshotted from the pgdat at the
|
|
+ * start of this reclaim pass. Forwarded to shrinkers through
|
|
+ * shrink_control::opportunistic_compaction so they can skip
|
|
+ * non-productive work for failable high-order allocations.
|
|
+ */
|
|
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
|
|
+
|
|
/* Can active folios be deactivated as part of reclaim? */
|
|
#define DEACTIVATE_ANON 1
|
|
#define DEACTIVATE_FILE 2
|
|
@@ -198,6 +206,29 @@ struct scan_control {
|
|
*/
|
|
int vm_swappiness = 60;
|
|
|
|
+/*
|
|
+ * Is @gfp_flags a high-order allocation that is eligible for the
|
|
+ * "opportunistic compaction" treatment in kswapd / shrinkers?
|
|
+ *
|
|
+ * The caller must be willing to tolerate failure (__GFP_NORETRY or
|
|
+ * __GFP_RETRY_MAYFAIL) and must not have set __GFP_NOFAIL. For such
|
|
+ * allocations there is little value in burning working-set pages just to
|
|
+ * scrape together a single high-order block: if compaction can't easily
|
|
+ * succeed, the caller would rather see the allocation fail.
|
|
+ */
|
|
+static bool gfp_opportunistic_compaction(gfp_t gfp_flags)
|
|
+{
|
|
+ return (gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL)) &&
|
|
+ !(gfp_flags & __GFP_NOFAIL);
|
|
+}
|
|
+
|
|
+static bool sc_opportunistic_compaction(struct scan_control *sc)
|
|
+{
|
|
+ return sc->order && (sc->kswapd_opportunistic_compaction ==
|
|
+ KSWAPD_OPPORTUNISTIC_COMPACTION || (!current_is_kswapd() &&
|
|
+ gfp_opportunistic_compaction(sc->gfp_mask)));
|
|
+}
|
|
+
|
|
static int sc_swappiness(struct scan_control *sc, struct mem_cgroup *memcg)
|
|
{
|
|
if (sc->proactive && sc->proactive_swappiness)
|
|
@@ -411,7 +442,7 @@ static unsigned long drop_slab_node(int nid)
|
|
|
|
memcg = mem_cgroup_iter(NULL, NULL, NULL);
|
|
do {
|
|
- freed += shrink_slab(GFP_KERNEL, nid, memcg, 0);
|
|
+ freed += shrink_slab(GFP_KERNEL, nid, memcg, 0, false);
|
|
} while ((memcg = mem_cgroup_iter(NULL, memcg, NULL)) != NULL);
|
|
|
|
return freed;
|
|
@@ -5081,6 +5112,7 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc)
|
|
unsigned long reclaimed = sc->nr_reclaimed;
|
|
struct mem_cgroup *memcg = lruvec_memcg(lruvec);
|
|
struct pglist_data *pgdat = lruvec_pgdat(lruvec);
|
|
+ bool opportunistic_compaction = sc_opportunistic_compaction(sc);
|
|
|
|
/* lru_gen_age_node() called mem_cgroup_calculate_protection() */
|
|
if (mem_cgroup_below_min(NULL, memcg))
|
|
@@ -5096,7 +5128,8 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc)
|
|
|
|
need_rotate = try_to_shrink_lruvec(lruvec, sc);
|
|
|
|
- shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority);
|
|
+ shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority,
|
|
+ opportunistic_compaction);
|
|
|
|
if (!sc->proactive)
|
|
vmpressure(sc->gfp_mask, sc->order, memcg, false,
|
|
@@ -6163,6 +6196,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc)
|
|
struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat);
|
|
unsigned long reclaimed;
|
|
unsigned long scanned;
|
|
+ bool opportunistic_compaction = sc_opportunistic_compaction(sc);
|
|
|
|
/*
|
|
* This loop can become CPU-bound when target memcgs
|
|
@@ -6200,7 +6234,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc)
|
|
shrink_lruvec(lruvec, sc);
|
|
|
|
shrink_slab(sc->gfp_mask, pgdat->node_id, memcg,
|
|
- sc->priority);
|
|
+ sc->priority, opportunistic_compaction);
|
|
|
|
/* Record the group's reclaim efficiency */
|
|
if (!sc->proactive)
|
|
@@ -7133,8 +7167,14 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx)
|
|
* found to have free_pages <= high_wmark_pages(zone), any page in that zone
|
|
* or lower is eligible for reclaim until at least one usable zone is
|
|
* balanced.
|
|
+ *
|
|
+ * @kswapd_opportunistic_compaction is the aggregated hint produced by
|
|
+ * wakeup_kswapd() for this run; it is propagated into scan_control so that
|
|
+ * shrinkers can skip costly work that is unlikely to help compaction when
|
|
+ * all wakers are failable high-order allocations.
|
|
*/
|
|
-static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx)
|
|
+static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx,
|
|
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction)
|
|
{
|
|
int i;
|
|
unsigned long nr_soft_reclaimed;
|
|
@@ -7148,6 +7188,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx)
|
|
.gfp_mask = GFP_KERNEL,
|
|
.order = order,
|
|
.may_unmap = 1,
|
|
+ .kswapd_opportunistic_compaction = kswapd_opportunistic_compaction,
|
|
};
|
|
|
|
trace_mm_vmscan_balance_pgdat_begin(pgdat->node_id, order,
|
|
@@ -7372,8 +7413,10 @@ static enum zone_type kswapd_highest_zoneidx(pg_data_t *pgdat,
|
|
return curr_idx == MAX_NR_ZONES ? prev_highest_zoneidx : curr_idx;
|
|
}
|
|
|
|
-static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
|
|
- unsigned int highest_zoneidx)
|
|
+static void
|
|
+kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
|
|
+ unsigned int highest_zoneidx,
|
|
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction)
|
|
{
|
|
long remaining = 0;
|
|
DEFINE_WAIT(wait);
|
|
@@ -7419,6 +7462,11 @@ static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_o
|
|
|
|
if (READ_ONCE(pgdat->kswapd_order) < reclaim_order)
|
|
WRITE_ONCE(pgdat->kswapd_order, reclaim_order);
|
|
+
|
|
+ if (kswapd_opportunistic_compaction ==
|
|
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION)
|
|
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
|
|
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
|
|
}
|
|
|
|
finish_wait(&pgdat->kswapd_wait, &wait);
|
|
@@ -7475,6 +7523,7 @@ static int kswapd(void *p)
|
|
unsigned int highest_zoneidx = MAX_NR_ZONES - 1;
|
|
pg_data_t *pgdat = (pg_data_t *)p;
|
|
struct task_struct *tsk = current;
|
|
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
|
|
|
|
/*
|
|
* Tell the memory management that we're a "memory allocator",
|
|
@@ -7492,6 +7541,8 @@ static int kswapd(void *p)
|
|
set_freezable();
|
|
|
|
WRITE_ONCE(pgdat->kswapd_order, 0);
|
|
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
|
|
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
|
|
WRITE_ONCE(pgdat->kswapd_highest_zoneidx, MAX_NR_ZONES);
|
|
atomic_set(&pgdat->nr_writeback_throttled, 0);
|
|
for ( ; ; ) {
|
|
@@ -7500,13 +7551,18 @@ static int kswapd(void *p)
|
|
alloc_order = reclaim_order = READ_ONCE(pgdat->kswapd_order);
|
|
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
|
|
highest_zoneidx);
|
|
+ kswapd_opportunistic_compaction =
|
|
+ atomic_read(&pgdat->kswapd_opportunistic_compaction);
|
|
|
|
kswapd_try_sleep:
|
|
kswapd_try_to_sleep(pgdat, alloc_order, reclaim_order,
|
|
- highest_zoneidx);
|
|
+ highest_zoneidx, kswapd_opportunistic_compaction);
|
|
|
|
/* Read the new order and highest_zoneidx */
|
|
alloc_order = READ_ONCE(pgdat->kswapd_order);
|
|
+ kswapd_opportunistic_compaction =
|
|
+ atomic_xchg(&pgdat->kswapd_opportunistic_compaction,
|
|
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
|
|
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
|
|
highest_zoneidx);
|
|
WRITE_ONCE(pgdat->kswapd_order, 0);
|
|
@@ -7533,7 +7589,8 @@ static int kswapd(void *p)
|
|
trace_mm_vmscan_kswapd_wake(pgdat->node_id, highest_zoneidx,
|
|
alloc_order);
|
|
reclaim_order = balance_pgdat(pgdat, alloc_order,
|
|
- highest_zoneidx);
|
|
+ highest_zoneidx,
|
|
+ kswapd_opportunistic_compaction);
|
|
if (reclaim_order < alloc_order)
|
|
goto kswapd_try_sleep;
|
|
}
|
|
@@ -7571,6 +7628,28 @@ void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order,
|
|
if (READ_ONCE(pgdat->kswapd_order) < order)
|
|
WRITE_ONCE(pgdat->kswapd_order, order);
|
|
|
|
+ /*
|
|
+ * Fold this waker into the per-pgdat opportunistic-compaction hint
|
|
+ * that kswapd will pick up at the start of its next run.
|
|
+ *
|
|
+ * The state is sticky in the "NO" direction: once any waker in this
|
|
+ * batch is order-0 or a non-failable high-order allocation, the hint
|
|
+ * stays cleared until kswapd consumes it. Only when every waker so
|
|
+ * far is a failable high-order allocation do we set
|
|
+ * KSWAPD_OPPORTUNISTIC_COMPACTION, asking shrinkers to skip work
|
|
+ * that won't realistically help compaction.
|
|
+ */
|
|
+ if (atomic_read(&pgdat->kswapd_opportunistic_compaction) !=
|
|
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION) {
|
|
+ if (!order || !gfp_opportunistic_compaction(gfp_flags))
|
|
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
|
|
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
|
|
+ else if (order && gfp_opportunistic_compaction(gfp_flags))
|
|
+ atomic_cmpxchg(&pgdat->kswapd_opportunistic_compaction,
|
|
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION,
|
|
+ KSWAPD_OPPORTUNISTIC_COMPACTION);
|
|
+ }
|
|
+
|
|
if (!waitqueue_active(&pgdat->kswapd_wait))
|
|
return;
|
|
|