Files
omarchy-pkgs/pkgbuilds/linux-omarchy-bore/8204-mm-opportunistic-compaction.patch
T

382 lines
15 KiB
Diff

diff --git c/include/linux/mmzone.h i/include/linux/mmzone.h
--- c/include/linux/mmzone.h
+++ i/include/linux/mmzone.h
@@ -1465,6 +1465,39 @@ struct memory_failure_stats {
};
#endif
+/*
+ * Per-pgdat state machine for the kswapd "opportunistic compaction" hint.
+ *
+ * wakeup_kswapd() collapses the gfp flags of all wakers that arrive between
+ * two kswapd runs into a single tri-state, which kswapd then forwards to the
+ * shrinkers via shrink_control::opportunistic_compaction:
+ *
+ * KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION
+ * Initial state after kswapd consumes the previous value. No waker has
+ * been observed yet for the upcoming run.
+ *
+ * KSWAPD_NO_OPPORTUNISTIC_COMPACTION
+ * At least one waker is an order-0 allocation, or a high-order
+ * allocation that cannot tolerate failure (i.e., not eligible for
+ * opportunistic behaviour). Shrinkers must do their normal best-effort
+ * work; the hint is cleared.
+ *
+ * KSWAPD_OPPORTUNISTIC_COMPACTION
+ * All wakers seen so far are high-order allocations that may fail
+ * (__GFP_NORETRY or __GFP_RETRY_MAYFAIL, without __GFP_NOFAIL). Shrinkers
+ * may skip work that is unlikely to produce a contiguous high-order
+ * block (e.g., evicting working-set pages).
+ *
+ * The state is sticky in the "NO" direction within a single kswapd run: once
+ * any non-eligible waker is observed, subsequent eligible wakers cannot
+ * upgrade it back to KSWAPD_OPPORTUNISTIC_COMPACTION.
+ */
+enum kswapd_opportunistic_compaction_type {
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION = 0,
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION,
+ KSWAPD_OPPORTUNISTIC_COMPACTION,
+};
+
/*
* On NUMA machines, each NUMA node would have a pg_data_t to describe
* it's memory layout. On UMA machines there is a single pglist_data which
@@ -1529,6 +1562,13 @@ typedef struct pglist_data {
#endif
struct task_struct *kswapd; /* Protected by kswapd_lock */
int kswapd_order;
+ /*
+ * Aggregated opportunistic-compaction hint for the next kswapd run.
+ * Updated by wakeup_kswapd() based on the gfp flags / order of each
+ * waker, and consumed (and reset) by kswapd before balance_pgdat().
+ * See enum kswapd_opportunistic_compaction_type for the state machine.
+ */
+ atomic_t kswapd_opportunistic_compaction;
enum zone_type kswapd_highest_zoneidx;
atomic_t kswapd_failures; /* Number of 'reclaimed == 0' runs */
diff --git c/include/linux/shrinker.h i/include/linux/shrinker.h
--- c/include/linux/shrinker.h
+++ i/include/linux/shrinker.h
@@ -37,6 +37,26 @@ struct shrink_control {
/* current node being shrunk (for NUMA aware shrinkers) */
int nid;
+ /*
+ * Opportunistic compaction hint.
+ *
+ * Set by the reclaim path to tell shrinkers that this pass is
+ * driven by an order > 0 allocation that the caller is willing to
+ * have fail (e.g., __GFP_NORETRY / __GFP_RETRY_MAYFAIL without
+ * __GFP_NOFAIL). Such allocations only really benefit from
+ * shrinking when doing so frees up a contiguous, high-order block;
+ * thrashing working sets in the hope of producing one is typically
+ * counter-productive.
+ *
+ * Shrinkers that can produce naturally-aligned high-order folios
+ * (see shrink_control::order) should treat this as a hint to skip
+ * costly work that is unlikely to help compaction (for example,
+ * evicting hot/working-set pages just to free single pages).
+ *
+ * Only meaningful when @order > 0; ignored otherwise.
+ */
+ bool opportunistic_compaction;
+
/*
* How many objects scan_objects should scan and try to reclaim.
* This is reset before every call, so it is safe for callees
diff --git c/mm/internal.h i/mm/internal.h
--- c/mm/internal.h
+++ i/mm/internal.h
@@ -1767,7 +1767,7 @@ void __meminit __init_page_from_nid(unsigned long pfn, int nid);
/* shrinker related functions */
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
- int priority);
+ int priority, bool opportunistic_compaction);
int shmem_add_to_page_cache(struct folio *folio,
struct address_space *mapping,
diff --git c/mm/shrinker.c i/mm/shrinker.c
--- c/mm/shrinker.c
+++ i/mm/shrinker.c
@@ -474,7 +474,7 @@ static unsigned long do_shrink_slab(struct shrink_control *shrinkctl,
#ifdef CONFIG_MEMCG
static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
- struct mem_cgroup *memcg, int priority)
+ struct mem_cgroup *memcg, int priority, bool opportunistic_compaction)
{
struct shrinker_info *info;
unsigned long ret, freed = 0;
@@ -536,6 +536,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
.gfp_mask = gfp_mask,
.nid = nid,
.memcg = memcg,
+ .opportunistic_compaction = opportunistic_compaction,
};
struct shrinker *shrinker;
int shrinker_id = calc_shrinker_id(index, offset);
@@ -595,7 +596,8 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
}
#else /* !CONFIG_MEMCG */
static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
- struct mem_cgroup *memcg, int priority)
+ struct mem_cgroup *memcg, int priority,
+ bool opportunistic_compaction)
{
return 0;
}
@@ -607,6 +609,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
* @nid: node whose slab caches to target
* @memcg: memory cgroup whose slab caches to target
* @priority: the reclaim priority
+ * @opportunistic_compaction: do compaction opportunistically (e.g., do not swap working sets)
*
* Call the shrink functions to age shrinkable caches.
*
@@ -622,7 +625,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
* Returns the number of reclaimed slab objects.
*/
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
- int priority)
+ int priority, bool opportunistic_compaction)
{
unsigned long ret, freed = 0;
struct shrinker *shrinker;
@@ -635,7 +638,8 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
* oom.
*/
if (!mem_cgroup_disabled() && !mem_cgroup_is_root(memcg))
- return shrink_slab_memcg(gfp_mask, nid, memcg, priority);
+ return shrink_slab_memcg(gfp_mask, nid, memcg, priority,
+ opportunistic_compaction);
/*
* lockless algorithm of global shrink.
@@ -664,6 +668,7 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
.gfp_mask = gfp_mask,
.nid = nid,
.memcg = memcg,
+ .opportunistic_compaction = opportunistic_compaction,
};
if (!shrinker_try_get(shrinker))
diff --git c/mm/vmscan.c i/mm/vmscan.c
--- c/mm/vmscan.c
+++ i/mm/vmscan.c
@@ -96,6 +96,14 @@ struct scan_control {
/* Swappiness value for proactive reclaim. Always use sc_swappiness()! */
int *proactive_swappiness;
+ /*
+ * Opportunistic compaction hint snapshotted from the pgdat at the
+ * start of this reclaim pass. Forwarded to shrinkers through
+ * shrink_control::opportunistic_compaction so they can skip
+ * non-productive work for failable high-order allocations.
+ */
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
+
/* Can active folios be deactivated as part of reclaim? */
#define DEACTIVATE_ANON 1
#define DEACTIVATE_FILE 2
@@ -198,6 +206,29 @@ struct scan_control {
*/
int vm_swappiness = 60;
+/*
+ * Is @gfp_flags a high-order allocation that is eligible for the
+ * "opportunistic compaction" treatment in kswapd / shrinkers?
+ *
+ * The caller must be willing to tolerate failure (__GFP_NORETRY or
+ * __GFP_RETRY_MAYFAIL) and must not have set __GFP_NOFAIL. For such
+ * allocations there is little value in burning working-set pages just to
+ * scrape together a single high-order block: if compaction can't easily
+ * succeed, the caller would rather see the allocation fail.
+ */
+static bool gfp_opportunistic_compaction(gfp_t gfp_flags)
+{
+ return (gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL)) &&
+ !(gfp_flags & __GFP_NOFAIL);
+}
+
+static bool sc_opportunistic_compaction(struct scan_control *sc)
+{
+ return sc->order && (sc->kswapd_opportunistic_compaction ==
+ KSWAPD_OPPORTUNISTIC_COMPACTION || (!current_is_kswapd() &&
+ gfp_opportunistic_compaction(sc->gfp_mask)));
+}
+
static int sc_swappiness(struct scan_control *sc, struct mem_cgroup *memcg)
{
if (sc->proactive && sc->proactive_swappiness)
@@ -411,7 +442,7 @@ static unsigned long drop_slab_node(int nid)
memcg = mem_cgroup_iter(NULL, NULL, NULL);
do {
- freed += shrink_slab(GFP_KERNEL, nid, memcg, 0);
+ freed += shrink_slab(GFP_KERNEL, nid, memcg, 0, false);
} while ((memcg = mem_cgroup_iter(NULL, memcg, NULL)) != NULL);
return freed;
@@ -5081,6 +5112,7 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc)
unsigned long reclaimed = sc->nr_reclaimed;
struct mem_cgroup *memcg = lruvec_memcg(lruvec);
struct pglist_data *pgdat = lruvec_pgdat(lruvec);
+ bool opportunistic_compaction = sc_opportunistic_compaction(sc);
/* lru_gen_age_node() called mem_cgroup_calculate_protection() */
if (mem_cgroup_below_min(NULL, memcg))
@@ -5096,7 +5128,8 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc)
need_rotate = try_to_shrink_lruvec(lruvec, sc);
- shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority);
+ shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority,
+ opportunistic_compaction);
if (!sc->proactive)
vmpressure(sc->gfp_mask, sc->order, memcg, false,
@@ -6163,6 +6196,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc)
struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat);
unsigned long reclaimed;
unsigned long scanned;
+ bool opportunistic_compaction = sc_opportunistic_compaction(sc);
/*
* This loop can become CPU-bound when target memcgs
@@ -6200,7 +6234,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc)
shrink_lruvec(lruvec, sc);
shrink_slab(sc->gfp_mask, pgdat->node_id, memcg,
- sc->priority);
+ sc->priority, opportunistic_compaction);
/* Record the group's reclaim efficiency */
if (!sc->proactive)
@@ -7133,8 +7167,14 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx)
* found to have free_pages <= high_wmark_pages(zone), any page in that zone
* or lower is eligible for reclaim until at least one usable zone is
* balanced.
+ *
+ * @kswapd_opportunistic_compaction is the aggregated hint produced by
+ * wakeup_kswapd() for this run; it is propagated into scan_control so that
+ * shrinkers can skip costly work that is unlikely to help compaction when
+ * all wakers are failable high-order allocations.
*/
-static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx)
+static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx,
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction)
{
int i;
unsigned long nr_soft_reclaimed;
@@ -7148,6 +7188,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx)
.gfp_mask = GFP_KERNEL,
.order = order,
.may_unmap = 1,
+ .kswapd_opportunistic_compaction = kswapd_opportunistic_compaction,
};
trace_mm_vmscan_balance_pgdat_begin(pgdat->node_id, order,
@@ -7372,8 +7413,10 @@ static enum zone_type kswapd_highest_zoneidx(pg_data_t *pgdat,
return curr_idx == MAX_NR_ZONES ? prev_highest_zoneidx : curr_idx;
}
-static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
- unsigned int highest_zoneidx)
+static void
+kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
+ unsigned int highest_zoneidx,
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction)
{
long remaining = 0;
DEFINE_WAIT(wait);
@@ -7419,6 +7462,11 @@ static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_o
if (READ_ONCE(pgdat->kswapd_order) < reclaim_order)
WRITE_ONCE(pgdat->kswapd_order, reclaim_order);
+
+ if (kswapd_opportunistic_compaction ==
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION)
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
}
finish_wait(&pgdat->kswapd_wait, &wait);
@@ -7475,6 +7523,7 @@ static int kswapd(void *p)
unsigned int highest_zoneidx = MAX_NR_ZONES - 1;
pg_data_t *pgdat = (pg_data_t *)p;
struct task_struct *tsk = current;
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
/*
* Tell the memory management that we're a "memory allocator",
@@ -7492,6 +7541,8 @@ static int kswapd(void *p)
set_freezable();
WRITE_ONCE(pgdat->kswapd_order, 0);
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
WRITE_ONCE(pgdat->kswapd_highest_zoneidx, MAX_NR_ZONES);
atomic_set(&pgdat->nr_writeback_throttled, 0);
for ( ; ; ) {
@@ -7500,13 +7551,18 @@ static int kswapd(void *p)
alloc_order = reclaim_order = READ_ONCE(pgdat->kswapd_order);
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
highest_zoneidx);
+ kswapd_opportunistic_compaction =
+ atomic_read(&pgdat->kswapd_opportunistic_compaction);
kswapd_try_sleep:
kswapd_try_to_sleep(pgdat, alloc_order, reclaim_order,
- highest_zoneidx);
+ highest_zoneidx, kswapd_opportunistic_compaction);
/* Read the new order and highest_zoneidx */
alloc_order = READ_ONCE(pgdat->kswapd_order);
+ kswapd_opportunistic_compaction =
+ atomic_xchg(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
highest_zoneidx);
WRITE_ONCE(pgdat->kswapd_order, 0);
@@ -7533,7 +7589,8 @@ static int kswapd(void *p)
trace_mm_vmscan_kswapd_wake(pgdat->node_id, highest_zoneidx,
alloc_order);
reclaim_order = balance_pgdat(pgdat, alloc_order,
- highest_zoneidx);
+ highest_zoneidx,
+ kswapd_opportunistic_compaction);
if (reclaim_order < alloc_order)
goto kswapd_try_sleep;
}
@@ -7571,6 +7628,28 @@ void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order,
if (READ_ONCE(pgdat->kswapd_order) < order)
WRITE_ONCE(pgdat->kswapd_order, order);
+ /*
+ * Fold this waker into the per-pgdat opportunistic-compaction hint
+ * that kswapd will pick up at the start of its next run.
+ *
+ * The state is sticky in the "NO" direction: once any waker in this
+ * batch is order-0 or a non-failable high-order allocation, the hint
+ * stays cleared until kswapd consumes it. Only when every waker so
+ * far is a failable high-order allocation do we set
+ * KSWAPD_OPPORTUNISTIC_COMPACTION, asking shrinkers to skip work
+ * that won't realistically help compaction.
+ */
+ if (atomic_read(&pgdat->kswapd_opportunistic_compaction) !=
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION) {
+ if (!order || !gfp_opportunistic_compaction(gfp_flags))
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
+ else if (order && gfp_opportunistic_compaction(gfp_flags))
+ atomic_cmpxchg(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION,
+ KSWAPD_OPPORTUNISTIC_COMPACTION);
+ }
+
if (!waitqueue_active(&pgdat->kswapd_wait))
return;