diff --git c/include/linux/mmzone.h i/include/linux/mmzone.h --- c/include/linux/mmzone.h +++ i/include/linux/mmzone.h @@ -1465,6 +1465,39 @@ struct memory_failure_stats { }; #endif +/* + * Per-pgdat state machine for the kswapd "opportunistic compaction" hint. + * + * wakeup_kswapd() collapses the gfp flags of all wakers that arrive between + * two kswapd runs into a single tri-state, which kswapd then forwards to the + * shrinkers via shrink_control::opportunistic_compaction: + * + * KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION + * Initial state after kswapd consumes the previous value. No waker has + * been observed yet for the upcoming run. + * + * KSWAPD_NO_OPPORTUNISTIC_COMPACTION + * At least one waker is an order-0 allocation, or a high-order + * allocation that cannot tolerate failure (i.e., not eligible for + * opportunistic behaviour). Shrinkers must do their normal best-effort + * work; the hint is cleared. + * + * KSWAPD_OPPORTUNISTIC_COMPACTION + * All wakers seen so far are high-order allocations that may fail + * (__GFP_NORETRY or __GFP_RETRY_MAYFAIL, without __GFP_NOFAIL). Shrinkers + * may skip work that is unlikely to produce a contiguous high-order + * block (e.g., evicting working-set pages). + * + * The state is sticky in the "NO" direction within a single kswapd run: once + * any non-eligible waker is observed, subsequent eligible wakers cannot + * upgrade it back to KSWAPD_OPPORTUNISTIC_COMPACTION. + */ +enum kswapd_opportunistic_compaction_type { + KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION = 0, + KSWAPD_NO_OPPORTUNISTIC_COMPACTION, + KSWAPD_OPPORTUNISTIC_COMPACTION, +}; + /* * On NUMA machines, each NUMA node would have a pg_data_t to describe * it's memory layout. On UMA machines there is a single pglist_data which @@ -1529,6 +1562,13 @@ typedef struct pglist_data { #endif struct task_struct *kswapd; /* Protected by kswapd_lock */ int kswapd_order; + /* + * Aggregated opportunistic-compaction hint for the next kswapd run. + * Updated by wakeup_kswapd() based on the gfp flags / order of each + * waker, and consumed (and reset) by kswapd before balance_pgdat(). + * See enum kswapd_opportunistic_compaction_type for the state machine. + */ + atomic_t kswapd_opportunistic_compaction; enum zone_type kswapd_highest_zoneidx; atomic_t kswapd_failures; /* Number of 'reclaimed == 0' runs */ diff --git c/include/linux/shrinker.h i/include/linux/shrinker.h --- c/include/linux/shrinker.h +++ i/include/linux/shrinker.h @@ -37,6 +37,26 @@ struct shrink_control { /* current node being shrunk (for NUMA aware shrinkers) */ int nid; + /* + * Opportunistic compaction hint. + * + * Set by the reclaim path to tell shrinkers that this pass is + * driven by an order > 0 allocation that the caller is willing to + * have fail (e.g., __GFP_NORETRY / __GFP_RETRY_MAYFAIL without + * __GFP_NOFAIL). Such allocations only really benefit from + * shrinking when doing so frees up a contiguous, high-order block; + * thrashing working sets in the hope of producing one is typically + * counter-productive. + * + * Shrinkers that can produce naturally-aligned high-order folios + * (see shrink_control::order) should treat this as a hint to skip + * costly work that is unlikely to help compaction (for example, + * evicting hot/working-set pages just to free single pages). + * + * Only meaningful when @order > 0; ignored otherwise. + */ + bool opportunistic_compaction; + /* * How many objects scan_objects should scan and try to reclaim. * This is reset before every call, so it is safe for callees diff --git c/mm/internal.h i/mm/internal.h --- c/mm/internal.h +++ i/mm/internal.h @@ -1767,7 +1767,7 @@ void __meminit __init_page_from_nid(unsigned long pfn, int nid); /* shrinker related functions */ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg, - int priority); + int priority, bool opportunistic_compaction); int shmem_add_to_page_cache(struct folio *folio, struct address_space *mapping, diff --git c/mm/shrinker.c i/mm/shrinker.c --- c/mm/shrinker.c +++ i/mm/shrinker.c @@ -474,7 +474,7 @@ static unsigned long do_shrink_slab(struct shrink_control *shrinkctl, #ifdef CONFIG_MEMCG static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid, - struct mem_cgroup *memcg, int priority) + struct mem_cgroup *memcg, int priority, bool opportunistic_compaction) { struct shrinker_info *info; unsigned long ret, freed = 0; @@ -536,6 +536,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid, .gfp_mask = gfp_mask, .nid = nid, .memcg = memcg, + .opportunistic_compaction = opportunistic_compaction, }; struct shrinker *shrinker; int shrinker_id = calc_shrinker_id(index, offset); @@ -595,7 +596,8 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid, } #else /* !CONFIG_MEMCG */ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid, - struct mem_cgroup *memcg, int priority) + struct mem_cgroup *memcg, int priority, + bool opportunistic_compaction) { return 0; } @@ -607,6 +609,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid, * @nid: node whose slab caches to target * @memcg: memory cgroup whose slab caches to target * @priority: the reclaim priority + * @opportunistic_compaction: do compaction opportunistically (e.g., do not swap working sets) * * Call the shrink functions to age shrinkable caches. * @@ -622,7 +625,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid, * Returns the number of reclaimed slab objects. */ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg, - int priority) + int priority, bool opportunistic_compaction) { unsigned long ret, freed = 0; struct shrinker *shrinker; @@ -635,7 +638,8 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg, * oom. */ if (!mem_cgroup_disabled() && !mem_cgroup_is_root(memcg)) - return shrink_slab_memcg(gfp_mask, nid, memcg, priority); + return shrink_slab_memcg(gfp_mask, nid, memcg, priority, + opportunistic_compaction); /* * lockless algorithm of global shrink. @@ -664,6 +668,7 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg, .gfp_mask = gfp_mask, .nid = nid, .memcg = memcg, + .opportunistic_compaction = opportunistic_compaction, }; if (!shrinker_try_get(shrinker)) diff --git c/mm/vmscan.c i/mm/vmscan.c --- c/mm/vmscan.c +++ i/mm/vmscan.c @@ -96,6 +96,14 @@ struct scan_control { /* Swappiness value for proactive reclaim. Always use sc_swappiness()! */ int *proactive_swappiness; + /* + * Opportunistic compaction hint snapshotted from the pgdat at the + * start of this reclaim pass. Forwarded to shrinkers through + * shrink_control::opportunistic_compaction so they can skip + * non-productive work for failable high-order allocations. + */ + enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction; + /* Can active folios be deactivated as part of reclaim? */ #define DEACTIVATE_ANON 1 #define DEACTIVATE_FILE 2 @@ -198,6 +206,29 @@ struct scan_control { */ int vm_swappiness = 60; +/* + * Is @gfp_flags a high-order allocation that is eligible for the + * "opportunistic compaction" treatment in kswapd / shrinkers? + * + * The caller must be willing to tolerate failure (__GFP_NORETRY or + * __GFP_RETRY_MAYFAIL) and must not have set __GFP_NOFAIL. For such + * allocations there is little value in burning working-set pages just to + * scrape together a single high-order block: if compaction can't easily + * succeed, the caller would rather see the allocation fail. + */ +static bool gfp_opportunistic_compaction(gfp_t gfp_flags) +{ + return (gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL)) && + !(gfp_flags & __GFP_NOFAIL); +} + +static bool sc_opportunistic_compaction(struct scan_control *sc) +{ + return sc->order && (sc->kswapd_opportunistic_compaction == + KSWAPD_OPPORTUNISTIC_COMPACTION || (!current_is_kswapd() && + gfp_opportunistic_compaction(sc->gfp_mask))); +} + static int sc_swappiness(struct scan_control *sc, struct mem_cgroup *memcg) { if (sc->proactive && sc->proactive_swappiness) @@ -411,7 +442,7 @@ static unsigned long drop_slab_node(int nid) memcg = mem_cgroup_iter(NULL, NULL, NULL); do { - freed += shrink_slab(GFP_KERNEL, nid, memcg, 0); + freed += shrink_slab(GFP_KERNEL, nid, memcg, 0, false); } while ((memcg = mem_cgroup_iter(NULL, memcg, NULL)) != NULL); return freed; @@ -5081,6 +5112,7 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc) unsigned long reclaimed = sc->nr_reclaimed; struct mem_cgroup *memcg = lruvec_memcg(lruvec); struct pglist_data *pgdat = lruvec_pgdat(lruvec); + bool opportunistic_compaction = sc_opportunistic_compaction(sc); /* lru_gen_age_node() called mem_cgroup_calculate_protection() */ if (mem_cgroup_below_min(NULL, memcg)) @@ -5096,7 +5128,8 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc) need_rotate = try_to_shrink_lruvec(lruvec, sc); - shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority); + shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority, + opportunistic_compaction); if (!sc->proactive) vmpressure(sc->gfp_mask, sc->order, memcg, false, @@ -6163,6 +6196,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc) struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat); unsigned long reclaimed; unsigned long scanned; + bool opportunistic_compaction = sc_opportunistic_compaction(sc); /* * This loop can become CPU-bound when target memcgs @@ -6200,7 +6234,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc) shrink_lruvec(lruvec, sc); shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, - sc->priority); + sc->priority, opportunistic_compaction); /* Record the group's reclaim efficiency */ if (!sc->proactive) @@ -7133,8 +7167,14 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx) * found to have free_pages <= high_wmark_pages(zone), any page in that zone * or lower is eligible for reclaim until at least one usable zone is * balanced. + * + * @kswapd_opportunistic_compaction is the aggregated hint produced by + * wakeup_kswapd() for this run; it is propagated into scan_control so that + * shrinkers can skip costly work that is unlikely to help compaction when + * all wakers are failable high-order allocations. */ -static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) +static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx, + enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction) { int i; unsigned long nr_soft_reclaimed; @@ -7148,6 +7188,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) .gfp_mask = GFP_KERNEL, .order = order, .may_unmap = 1, + .kswapd_opportunistic_compaction = kswapd_opportunistic_compaction, }; trace_mm_vmscan_balance_pgdat_begin(pgdat->node_id, order, @@ -7372,8 +7413,10 @@ static enum zone_type kswapd_highest_zoneidx(pg_data_t *pgdat, return curr_idx == MAX_NR_ZONES ? prev_highest_zoneidx : curr_idx; } -static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order, - unsigned int highest_zoneidx) +static void +kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order, + unsigned int highest_zoneidx, + enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction) { long remaining = 0; DEFINE_WAIT(wait); @@ -7419,6 +7462,11 @@ static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_o if (READ_ONCE(pgdat->kswapd_order) < reclaim_order) WRITE_ONCE(pgdat->kswapd_order, reclaim_order); + + if (kswapd_opportunistic_compaction == + KSWAPD_NO_OPPORTUNISTIC_COMPACTION) + atomic_set(&pgdat->kswapd_opportunistic_compaction, + KSWAPD_NO_OPPORTUNISTIC_COMPACTION); } finish_wait(&pgdat->kswapd_wait, &wait); @@ -7475,6 +7523,7 @@ static int kswapd(void *p) unsigned int highest_zoneidx = MAX_NR_ZONES - 1; pg_data_t *pgdat = (pg_data_t *)p; struct task_struct *tsk = current; + enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction; /* * Tell the memory management that we're a "memory allocator", @@ -7492,6 +7541,8 @@ static int kswapd(void *p) set_freezable(); WRITE_ONCE(pgdat->kswapd_order, 0); + atomic_set(&pgdat->kswapd_opportunistic_compaction, + KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION); WRITE_ONCE(pgdat->kswapd_highest_zoneidx, MAX_NR_ZONES); atomic_set(&pgdat->nr_writeback_throttled, 0); for ( ; ; ) { @@ -7500,13 +7551,18 @@ static int kswapd(void *p) alloc_order = reclaim_order = READ_ONCE(pgdat->kswapd_order); highest_zoneidx = kswapd_highest_zoneidx(pgdat, highest_zoneidx); + kswapd_opportunistic_compaction = + atomic_read(&pgdat->kswapd_opportunistic_compaction); kswapd_try_sleep: kswapd_try_to_sleep(pgdat, alloc_order, reclaim_order, - highest_zoneidx); + highest_zoneidx, kswapd_opportunistic_compaction); /* Read the new order and highest_zoneidx */ alloc_order = READ_ONCE(pgdat->kswapd_order); + kswapd_opportunistic_compaction = + atomic_xchg(&pgdat->kswapd_opportunistic_compaction, + KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION); highest_zoneidx = kswapd_highest_zoneidx(pgdat, highest_zoneidx); WRITE_ONCE(pgdat->kswapd_order, 0); @@ -7533,7 +7589,8 @@ static int kswapd(void *p) trace_mm_vmscan_kswapd_wake(pgdat->node_id, highest_zoneidx, alloc_order); reclaim_order = balance_pgdat(pgdat, alloc_order, - highest_zoneidx); + highest_zoneidx, + kswapd_opportunistic_compaction); if (reclaim_order < alloc_order) goto kswapd_try_sleep; } @@ -7571,6 +7628,28 @@ void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order, if (READ_ONCE(pgdat->kswapd_order) < order) WRITE_ONCE(pgdat->kswapd_order, order); + /* + * Fold this waker into the per-pgdat opportunistic-compaction hint + * that kswapd will pick up at the start of its next run. + * + * The state is sticky in the "NO" direction: once any waker in this + * batch is order-0 or a non-failable high-order allocation, the hint + * stays cleared until kswapd consumes it. Only when every waker so + * far is a failable high-order allocation do we set + * KSWAPD_OPPORTUNISTIC_COMPACTION, asking shrinkers to skip work + * that won't realistically help compaction. + */ + if (atomic_read(&pgdat->kswapd_opportunistic_compaction) != + KSWAPD_NO_OPPORTUNISTIC_COMPACTION) { + if (!order || !gfp_opportunistic_compaction(gfp_flags)) + atomic_set(&pgdat->kswapd_opportunistic_compaction, + KSWAPD_NO_OPPORTUNISTIC_COMPACTION); + else if (order && gfp_opportunistic_compaction(gfp_flags)) + atomic_cmpxchg(&pgdat->kswapd_opportunistic_compaction, + KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION, + KSWAPD_OPPORTUNISTIC_COMPACTION); + } + if (!waitqueue_active(&pgdat->kswapd_wait)) return;