Merge pull request #440 from kwilczynski/feature/add-linux-omarchy-kernels

Add new base and BORE-enabled Linux kernel packages for Omarchy
This commit is contained in:
Ryan Hughes authored and GitHub committed 2026-09-14 12:20:14 -07:00
commit a01153595b
195 files changed
+190517

No files matched your search

@@ -0,0 +1,4 @@
{
"source": "local",
"skip_build": true
}
@@ -0,0 +1,56 @@
diff --git a/kernel/fork.c b/kernel/fork.c
index f0e2e131a9a5..7b611da9a27a 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -127,6 +127,12 @@
#include <kunit/visibility.h>
+#ifdef CONFIG_USER_NS
+static int unprivileged_userns_clone = 1;
+#else
+#define unprivileged_userns_clone 1
+#endif
+
/*
* Minimum number of threads to boot the kernel
*/
@@ -2093,6 +2099,11 @@ __latent_entropy struct task_struct *copy_process(
return ERR_PTR(-EPERM);
}
+ if ((clone_flags & CLONE_NEWUSER) && !unprivileged_userns_clone) {
+ if (!capable(CAP_SYS_ADMIN))
+ return ERR_PTR(-EPERM);
+ }
+
/*
* Force any signals received before this point to be delivered
* before the fork happens. Collect up signals sent to multiple
@@ -3163,6 +3174,10 @@ static int check_unshare_flags(unsigned long unshare_flags)
if (!current_is_single_threaded())
return -EINVAL;
}
+ if ((unshare_flags & CLONE_NEWUSER) && !unprivileged_userns_clone) {
+ if (!capable(CAP_SYS_ADMIN))
+ return -EPERM;
+ }
return 0;
}
@@ -3398,6 +3413,15 @@ static const struct ctl_table fork_sysctl_table[] = {
.mode = 0644,
.proc_handler = sysctl_max_threads,
},
+#ifdef CONFIG_USER_NS
+ {
+ .procname = "unprivileged_userns_clone",
+ .data = &unprivileged_userns_clone,
+ .maxlen = sizeof(int),
+ .mode = 0644,
+ .proc_handler = proc_dointvec,
+ },
+#endif
};
static int __init init_fork_sysctl(void)
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,763 @@
diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h
--- a/arch/x86/include/asm/pgtable.h
+++ b/arch/x86/include/asm/pgtable.h
@@ -50,7 +50,7 @@ void ptdump_walk_user_pgd_level_checkwx(void);
extern spinlock_t pgd_lock;
extern struct list_head pgd_list;
-extern struct mm_struct *pgd_page_get_mm(struct page *page);
+struct mm_struct *pgd_page_get_mm(struct ptdesc *pt);
extern pmdval_t early_pmd_flags;
diff --git a/arch/x86/include/asm/pgtable_types.h b/arch/x86/include/asm/pgtable_types.h
--- a/arch/x86/include/asm/pgtable_types.h
+++ b/arch/x86/include/asm/pgtable_types.h
@@ -512,7 +512,7 @@ static inline pgprot_t pgprot_large_2_4k(pgprot_t pgprot)
return __pgprot(protval_large_2_4k(pgprot_val(pgprot)));
}
-
+struct ptdesc;
typedef struct page *pgtable_t;
extern pteval_t __supported_pte_mask;
diff --git a/arch/x86/include/asm/pkeys.h b/arch/x86/include/asm/pkeys.h
--- a/arch/x86/include/asm/pkeys.h
+++ b/arch/x86/include/asm/pkeys.h
@@ -88,6 +88,9 @@ int mm_pkey_alloc(struct mm_struct *mm)
u16 all_pkeys_mask = ((1U << arch_max_pkey()) - 1);
int ret;
+ if (!arch_pkeys_enabled())
+ return -1;
+
/*
* Are we out of pkeys? We must handle this specially
* because ffz() behavior is undefined if there are no
diff --git a/arch/x86/include/asm/tlbflush.h b/arch/x86/include/asm/tlbflush.h
--- a/arch/x86/include/asm/tlbflush.h
+++ b/arch/x86/include/asm/tlbflush.h
@@ -4,6 +4,7 @@
#include <linux/mm_types.h>
#include <linux/mmu_notifier.h>
+#include <linux/minmax.h>
#include <linux/sched.h>
#include <asm/barrier.h>
@@ -211,6 +212,12 @@ extern u16 invlpgb_count_max;
extern void initialize_tlbstate_and_flush(void);
+/*
+ * Keep stack-allocated flush_tlb_info cacheline aligned, but cap the
+ * alignment to avoid excessive stack usage on large-cacheline systems.
+ */
+#define FLUSH_TLB_INFO_ALIGN MIN(SMP_CACHE_BYTES, 64)
+
/*
* TLB flushing:
*
@@ -249,7 +256,7 @@ struct flush_tlb_info {
u8 stride_shift;
u8 freed_tables;
u8 trim_cpumask;
-};
+} __aligned(FLUSH_TLB_INFO_ALIGN);
void flush_tlb_local(void);
void flush_tlb_one_user(unsigned long addr);
diff --git a/arch/x86/kernel/kvm.c b/arch/x86/kernel/kvm.c
--- a/arch/x86/kernel/kvm.c
+++ b/arch/x86/kernel/kvm.c
@@ -663,8 +663,10 @@ static void kvm_flush_tlb_multi(const struct cpumask *cpumask,
u8 state;
int cpu;
struct kvm_steal_time *src;
- struct cpumask *flushmask = this_cpu_cpumask_var_ptr(__pv_cpu_mask);
+ struct cpumask *flushmask;
+ guard(preempt)();
+ flushmask = this_cpu_cpumask_var_ptr(__pv_cpu_mask);
cpumask_copy(flushmask, cpumask);
/*
* We have to call flush only on online vCPUs. And
diff --git a/arch/x86/mm/fault.c b/arch/x86/mm/fault.c
--- a/arch/x86/mm/fault.c
+++ b/arch/x86/mm/fault.c
@@ -275,17 +275,17 @@ void arch_sync_kernel_mappings(unsigned long start, unsigned long end)
for (addr = start & PMD_MASK;
addr >= TASK_SIZE_MAX && addr < VMALLOC_END;
addr += PMD_SIZE) {
- struct page *page;
+ struct ptdesc *ptdesc;
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
spinlock_t *pgt_lock;
/* the pgt_lock only for Xen */
- pgt_lock = &pgd_page_get_mm(page)->page_table_lock;
+ pgt_lock = &pgd_page_get_mm(ptdesc)->page_table_lock;
spin_lock(pgt_lock);
- vmalloc_sync_one(page_address(page), addr);
+ vmalloc_sync_one(ptdesc_address(ptdesc), addr);
spin_unlock(pgt_lock);
}
spin_unlock(&pgd_lock);
diff --git a/arch/x86/mm/init_64.c b/arch/x86/mm/init_64.c
--- a/arch/x86/mm/init_64.c
+++ b/arch/x86/mm/init_64.c
@@ -136,7 +136,7 @@ static void sync_global_pgds_l5(unsigned long start, unsigned long end)
for (addr = start; addr <= end; addr = ALIGN(addr + 1, PGDIR_SIZE)) {
const pgd_t *pgd_ref = pgd_offset_k(addr);
- struct page *page;
+ struct ptdesc *ptdesc;
/* Check for overflow */
if (addr < start)
@@ -146,13 +146,13 @@ static void sync_global_pgds_l5(unsigned long start, unsigned long end)
continue;
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
pgd_t *pgd;
spinlock_t *pgt_lock;
- pgd = (pgd_t *)page_address(page) + pgd_index(addr);
+ pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(addr);
/* the pgt_lock only for Xen */
- pgt_lock = &pgd_page_get_mm(page)->page_table_lock;
+ pgt_lock = &pgd_page_get_mm(ptdesc)->page_table_lock;
spin_lock(pgt_lock);
if (!pgd_none(*pgd_ref) && !pgd_none(*pgd))
@@ -174,7 +174,7 @@ static void sync_global_pgds_l4(unsigned long start, unsigned long end)
for (addr = start; addr <= end; addr = ALIGN(addr + 1, PGDIR_SIZE)) {
pgd_t *pgd_ref = pgd_offset_k(addr);
const p4d_t *p4d_ref;
- struct page *page;
+ struct ptdesc *ptdesc;
/*
* With folded p4d, pgd_none() is always false, we need to
@@ -187,15 +187,15 @@ static void sync_global_pgds_l4(unsigned long start, unsigned long end)
continue;
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
pgd_t *pgd;
p4d_t *p4d;
spinlock_t *pgt_lock;
- pgd = (pgd_t *)page_address(page) + pgd_index(addr);
+ pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(addr);
p4d = p4d_offset(pgd, addr);
/* the pgt_lock only for Xen */
- pgt_lock = &pgd_page_get_mm(page)->page_table_lock;
+ pgt_lock = &pgd_page_get_mm(ptdesc)->page_table_lock;
spin_lock(pgt_lock);
if (!p4d_none(*p4d_ref) && !p4d_none(*p4d))
diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c
--- a/arch/x86/mm/pat/set_memory.c
+++ b/arch/x86/mm/pat/set_memory.c
@@ -62,18 +62,18 @@ enum cpa_warn {
static const int cpa_warn_level = CPA_PROTECT;
/*
- * Serialize cpa() (for !DEBUG_PAGEALLOC which uses large identity mappings)
- * using cpa_lock. So that we don't allow any other cpu, with stale large tlb
- * entries change the page attribute in parallel to some other cpu
- * splitting a large page entry along with changing the attribute.
+ * Serialize cpa() using cpa_lock so that we don't allow any other cpu, with
+ * stale large tlb entries, to change the page attribute in parallel to some
+ * other cpu splitting a large page entry along with changing the attribute.
*/
static DEFINE_SPINLOCK(cpa_lock);
-#define CPA_FLUSHTLB 1
-#define CPA_ARRAY 2
-#define CPA_PAGES_ARRAY 4
-#define CPA_NO_CHECK_ALIAS 8 /* Do not search for aliases */
-#define CPA_COLLAPSE 16 /* try to collapse large pages */
+#define CPA_FLUSHTLB 0x01
+#define CPA_ARRAY 0x02
+#define CPA_PAGES_ARRAY 0x04
+#define CPA_NO_CHECK_ALIAS 0x08 /* Do not search for aliases */
+#define CPA_COLLAPSE 0x10 /* try to collapse large pages */
+#define CPA_DEBUG_PAGEALLOC 0x20
static inline pgprot_t cachemode2pgprot(enum page_cache_mode pcm)
{
@@ -86,9 +86,8 @@ static unsigned long direct_pages_count[PG_LEVEL_NUM];
void update_page_count(int level, unsigned long pages)
{
/* Protect against CPA */
- spin_lock(&pgd_lock);
+ guard(spinlock)(&pgd_lock);
direct_pages_count[level] += pages;
- spin_unlock(&pgd_lock);
}
static void split_page_count(int level)
@@ -418,6 +417,8 @@ static void cpa_collapse_large_pages(struct cpa_data *cpa)
int collapsed = 0;
int i;
+ guard(spinlock)(&cpa_lock);
+
if (cpa->flags & (CPA_PAGES_ARRAY | CPA_ARRAY)) {
for (i = 0; i < cpa->numpages; i++)
collapsed += collapse_large_pages(__cpa_addr(cpa, i),
@@ -888,24 +889,23 @@ static void __set_pmd_pte(pte_t *kpte, unsigned long address, pte_t pte)
{
/* change init_mm */
set_pte_atomic(kpte, pte);
-#ifdef CONFIG_X86_32
- {
- struct page *page;
- list_for_each_entry(page, &pgd_list, lru) {
+ if (IS_ENABLED(CONFIG_X86_32)) {
+ struct ptdesc *ptdesc;
+
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
pgd_t *pgd;
p4d_t *p4d;
pud_t *pud;
pmd_t *pmd;
- pgd = (pgd_t *)page_address(page) + pgd_index(address);
+ pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(address);
p4d = p4d_offset(pgd, address);
pud = pud_offset(p4d, address);
pmd = pmd_offset(pud, address);
set_pte_atomic((pte_t *)pmd, pte);
}
}
-#endif
}
static pgprot_t pgprot_clear_protnone_bits(pgprot_t prot)
@@ -1075,16 +1075,11 @@ static int __should_split_large_page(pte_t *kpte, unsigned long address,
static int should_split_large_page(pte_t *kpte, unsigned long address,
struct cpa_data *cpa)
{
- int do_split;
-
if (cpa->force_split)
return 1;
- spin_lock(&pgd_lock);
- do_split = __should_split_large_page(kpte, address, cpa);
- spin_unlock(&pgd_lock);
-
- return do_split;
+ guard(spinlock)(&pgd_lock);
+ return __should_split_large_page(kpte, address, cpa);
}
static void split_set_pte(struct cpa_data *cpa, pte_t *pte, unsigned long pfn,
@@ -1135,16 +1130,14 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address,
bool nx, rw;
pte_t *tmp;
- spin_lock(&pgd_lock);
+ guard(spinlock)(&pgd_lock);
/*
* Check for races, another CPU might have split this page
* up for us already:
*/
tmp = _lookup_address_cpa(cpa, address, &level, &nx, &rw);
- if (tmp != kpte) {
- spin_unlock(&pgd_lock);
+ if (tmp != kpte)
return 1;
- }
paravirt_alloc_pte(&init_mm, page_to_pfn(base));
@@ -1177,7 +1170,6 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address,
break;
default:
- spin_unlock(&pgd_lock);
return 1;
}
@@ -1225,7 +1217,6 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address,
* just split large page entry.
*/
flush_tlb_all();
- spin_unlock(&pgd_lock);
return 0;
}
@@ -1235,11 +1226,9 @@ static int split_large_page(struct cpa_data *cpa, pte_t *kpte,
{
struct ptdesc *ptdesc;
- if (!debug_pagealloc_enabled())
- spin_unlock(&cpa_lock);
+ spin_unlock(&cpa_lock);
ptdesc = pagetable_alloc(GFP_KERNEL, 0);
- if (!debug_pagealloc_enabled())
- spin_lock(&cpa_lock);
+ spin_lock(&cpa_lock);
if (!ptdesc)
return -ENOMEM;
@@ -1298,11 +1287,11 @@ static int collapse_pmd_page(pmd_t *pmd, unsigned long addr,
list_add(&page_ptdesc(pmd_page(old_pmd))->pt_list, pgtables);
if (IS_ENABLED(CONFIG_X86_32)) {
- struct page *page;
+ struct ptdesc *ptdesc;
/* Update all PGD tables to use the same large page */
- list_for_each_entry(page, &pgd_list, lru) {
- pgd_t *pgd = (pgd_t *)page_address(page) + pgd_index(addr);
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
+ pgd_t *pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(addr);
p4d_t *p4d = p4d_offset(pgd, addr);
pud_t *pud = pud_offset(p4d, addr);
pmd_t *pmd = pmd_offset(pud, addr);
@@ -1376,7 +1365,7 @@ static int collapse_pud_page(pud_t *pud, unsigned long addr,
*/
static int collapse_large_pages(unsigned long addr, struct list_head *pgtables)
{
- int collapsed = 0;
+ int collapsed;
pgd_t *pgd;
p4d_t *p4d;
pud_t *pud;
@@ -1384,26 +1373,24 @@ static int collapse_large_pages(unsigned long addr, struct list_head *pgtables)
addr &= PMD_MASK;
- spin_lock(&pgd_lock);
+ guard(spinlock)(&pgd_lock);
pgd = pgd_offset_k(addr);
if (pgd_none(*pgd))
- goto out;
+ return 0;
p4d = p4d_offset(pgd, addr);
if (p4d_none(*p4d))
- goto out;
+ return 0;
pud = pud_offset(p4d, addr);
if (!pud_present(*pud) || pud_leaf(*pud))
- goto out;
+ return 0;
pmd = pmd_offset(pud, addr);
if (!pmd_present(*pmd) || pmd_leaf(*pmd))
- goto out;
+ return 0;
collapsed = collapse_pmd_page(pmd, addr, pgtables);
if (collapsed)
collapsed += collapse_pud_page(pud, addr, pgtables);
-out:
- spin_unlock(&pgd_lock);
return collapsed;
}
@@ -2004,6 +1991,7 @@ static int __change_page_attr_set_clr(struct cpa_data *cpa, int primary)
{
unsigned long numpages = cpa->numpages;
unsigned long rempages = numpages;
+ bool lock = true;
int ret = 0;
/*
@@ -2013,6 +2001,29 @@ static int __change_page_attr_set_clr(struct cpa_data *cpa, int primary)
!cpa->force_split)
return ret;
+ /*
+ * DEBUG_PAGEALLOC is special; it is called from any context the
+ * page-allocator is, which violates the normal cpa_lock locking
+ * rules.
+ *
+ * However, since it is part of the page-allocator, things are still
+ * properly serialized by the page-allocator locking and the fact that
+ * when a page is owned by the page-allocator, it isn't owned by
+ * anybody else. That is, you *SHOULD NOT* be calling cpa() on memory
+ * that isn't allocated.
+ *
+ * Additionally, DEBUG_PAGEALLOC ensures (per probe_page_size_mask())
+ * that the kernel mapping is 4k pages, therefore there are no large
+ * pages to split/collapse.
+ *
+ * Furthermore, the page-allocator strictly manages pages that
+ * *exist*, avoiding pgd_lock.
+ *
+ * Therefore, it is safe to not take cpa_lock.
+ */
+ if (debug_pagealloc_enabled() && (cpa->flags & CPA_DEBUG_PAGEALLOC))
+ lock = false;
+
while (rempages) {
/*
* Store the remaining nr of pages for the large page
@@ -2023,11 +2034,12 @@ static int __change_page_attr_set_clr(struct cpa_data *cpa, int primary)
if (cpa->flags & (CPA_ARRAY | CPA_PAGES_ARRAY))
cpa->numpages = 1;
- if (!debug_pagealloc_enabled())
- spin_lock(&cpa_lock);
- ret = __change_page_attr(cpa, primary);
- if (!debug_pagealloc_enabled())
- spin_unlock(&cpa_lock);
+ if (lock) {
+ guard(spinlock)(&cpa_lock);
+ ret = __change_page_attr(cpa, primary);
+ } else {
+ ret = __change_page_attr(cpa, primary);
+ }
if (ret)
goto out;
@@ -2606,7 +2618,7 @@ int set_pages_rw(struct page *page, int numpages)
return set_memory_rw(addr, numpages);
}
-static int __set_pages_p(struct page *page, int numpages)
+static int __set_pages_p(struct page *page, int numpages, unsigned int cpa_flags)
{
unsigned long tempaddr = (unsigned long) page_address(page);
struct cpa_data cpa = { .vaddr = &tempaddr,
@@ -2614,7 +2626,7 @@ static int __set_pages_p(struct page *page, int numpages)
.numpages = numpages,
.mask_set = __pgprot(_PAGE_PRESENT | _PAGE_RW),
.mask_clr = __pgprot(0),
- .flags = CPA_NO_CHECK_ALIAS };
+ .flags = CPA_NO_CHECK_ALIAS | cpa_flags };
/*
* No alias checking needed for setting present flag. otherwise,
@@ -2625,7 +2637,7 @@ static int __set_pages_p(struct page *page, int numpages)
return __change_page_attr_set_clr(&cpa, 1);
}
-static int __set_pages_np(struct page *page, int numpages)
+static int __set_pages_np(struct page *page, int numpages, unsigned int cpa_flags)
{
unsigned long tempaddr = (unsigned long) page_address(page);
struct cpa_data cpa = { .vaddr = &tempaddr,
@@ -2633,7 +2645,7 @@ static int __set_pages_np(struct page *page, int numpages)
.numpages = numpages,
.mask_set = __pgprot(0),
.mask_clr = __pgprot(_PAGE_PRESENT | _PAGE_RW | _PAGE_DIRTY),
- .flags = CPA_NO_CHECK_ALIAS };
+ .flags = CPA_NO_CHECK_ALIAS | cpa_flags };
/*
* No alias checking needed for setting not present flag. otherwise,
@@ -2646,20 +2658,20 @@ static int __set_pages_np(struct page *page, int numpages)
int set_direct_map_invalid_noflush(struct page *page)
{
- return __set_pages_np(page, 1);
+ return __set_pages_np(page, 1, 0);
}
int set_direct_map_default_noflush(struct page *page)
{
- return __set_pages_p(page, 1);
+ return __set_pages_p(page, 1, 0);
}
int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid)
{
if (valid)
- return __set_pages_p(page, nr);
+ return __set_pages_p(page, nr, 0);
- return __set_pages_np(page, nr);
+ return __set_pages_np(page, nr, 0);
}
#ifdef CONFIG_DEBUG_PAGEALLOC
@@ -2678,15 +2690,23 @@ void __kernel_map_pages(struct page *page, int numpages, int enable)
* and hence no memory allocations during large page split.
*/
if (enable)
- __set_pages_p(page, numpages);
+ __set_pages_p(page, numpages, CPA_DEBUG_PAGEALLOC);
else
- __set_pages_np(page, numpages);
+ __set_pages_np(page, numpages, CPA_DEBUG_PAGEALLOC);
/*
- * We should perform an IPI and flush all tlbs,
- * but that can deadlock->flush only current cpu.
- * Preemption needs to be disabled around __flush_tlb_all() due to
- * CR3 reload in __native_flush_tlb().
+ * We should perform an IPI and flush all tlbs, but that can
+ * deadlock, settle for a local flush.
+ *
+ * Not doing a global TLB flush means that remote CPUs will retain
+ * stale TLB entries. In case of P->NP (on free) this means the remote
+ * CPUs will not take the faults, making the debug scheme less
+ * reliable. On the NP->P (on alloc) this means the remote CPUs can
+ * take a spurious fault. However spurious_kernel_fault() will observe
+ * *_present() and fix it up.
+ *
+ * Preemption needs to be disabled around __flush_tlb_all() due to CR3
+ * reload in __native_flush_tlb().
*/
preempt_disable();
__flush_tlb_all();
diff --git a/arch/x86/mm/pgtable.c b/arch/x86/mm/pgtable.c
--- a/arch/x86/mm/pgtable.c
+++ b/arch/x86/mm/pgtable.c
@@ -74,9 +74,9 @@ static void pgd_set_mm(pgd_t *pgd, struct mm_struct *mm)
virt_to_ptdesc(pgd)->pt_mm = mm;
}
-struct mm_struct *pgd_page_get_mm(struct page *page)
+struct mm_struct *pgd_page_get_mm(struct ptdesc *pt)
{
- return page_ptdesc(page)->pt_mm;
+ return pt->pt_mm;
}
static void pgd_ctor(struct mm_struct *mm, pgd_t *pgd)
diff --git a/arch/x86/mm/tlb.c b/arch/x86/mm/tlb.c
--- a/arch/x86/mm/tlb.c
+++ b/arch/x86/mm/tlb.c
@@ -1373,35 +1373,19 @@ void flush_tlb_multi(const struct cpumask *cpumask,
*/
unsigned long tlb_single_page_flush_ceiling __read_mostly = 33;
-static DEFINE_PER_CPU_SHARED_ALIGNED(struct flush_tlb_info, flush_tlb_info);
-
-#ifdef CONFIG_DEBUG_VM
-static DEFINE_PER_CPU(unsigned int, flush_tlb_info_idx);
-#endif
-
-static struct flush_tlb_info *get_flush_tlb_info(struct mm_struct *mm,
- unsigned long start, unsigned long end,
- unsigned int stride_shift, bool freed_tables,
- u64 new_tlb_gen)
+static void init_flush_tlb_info(struct flush_tlb_info *info,
+ struct mm_struct *mm,
+ unsigned long start, unsigned long end,
+ unsigned int stride_shift, bool freed_tables,
+ u64 new_tlb_gen)
{
- struct flush_tlb_info *info = this_cpu_ptr(&flush_tlb_info);
-
-#ifdef CONFIG_DEBUG_VM
- /*
- * Ensure that the following code is non-reentrant and flush_tlb_info
- * is not overwritten. This means no TLB flushing is initiated by
- * interrupt handlers and machine-check exception handlers.
- */
- BUG_ON(this_cpu_inc_return(flush_tlb_info_idx) != 1);
-#endif
-
/*
* If the number of flushes is so large that a full flush
* would be faster, do a full flush.
*/
if ((end - start) >> stride_shift > tlb_single_page_flush_ceiling) {
- start = 0;
- end = TLB_FLUSH_ALL;
+ start = 0;
+ end = TLB_FLUSH_ALL;
}
info->start = start;
@@ -1412,32 +1396,21 @@ static struct flush_tlb_info *get_flush_tlb_info(struct mm_struct *mm,
info->new_tlb_gen = new_tlb_gen;
info->initiating_cpu = smp_processor_id();
info->trim_cpumask = 0;
-
- return info;
-}
-
-static void put_flush_tlb_info(void)
-{
-#ifdef CONFIG_DEBUG_VM
- /* Complete reentrancy prevention checks */
- barrier();
- this_cpu_dec(flush_tlb_info_idx);
-#endif
}
void flush_tlb_mm_range(struct mm_struct *mm, unsigned long start,
unsigned long end, unsigned int stride_shift,
bool freed_tables)
{
- struct flush_tlb_info *info;
+ struct flush_tlb_info info;
+ bool remote_flush = false;
int cpu = get_cpu();
u64 new_tlb_gen;
/* This is also a barrier that synchronizes with switch_mm(). */
new_tlb_gen = inc_mm_tlb_gen(mm);
- info = get_flush_tlb_info(mm, start, end, stride_shift, freed_tables,
- new_tlb_gen);
+ init_flush_tlb_info(&info, mm, start, end, stride_shift, freed_tables, new_tlb_gen);
/*
* flush_tlb_multi() is not optimized for the common case in which only
@@ -1445,20 +1418,24 @@ void flush_tlb_mm_range(struct mm_struct *mm, unsigned long start,
* flush_tlb_func_local() directly in this case.
*/
if (mm_global_asid(mm)) {
- broadcast_tlb_flush(info);
+ broadcast_tlb_flush(&info);
} else if (cpumask_any_but(mm_cpumask(mm), cpu) < nr_cpu_ids) {
- info->trim_cpumask = should_trim_cpumask(mm);
- flush_tlb_multi(mm_cpumask(mm), info);
- consider_global_asid(mm);
+ remote_flush = true;
} else if (mm == this_cpu_read(cpu_tlbstate.loaded_mm)) {
lockdep_assert_irqs_enabled();
local_irq_disable();
- flush_tlb_func(info);
+ flush_tlb_func(&info);
local_irq_enable();
}
- put_flush_tlb_info();
put_cpu();
+
+ if (remote_flush) {
+ info.trim_cpumask = should_trim_cpumask(mm);
+ flush_tlb_multi(mm_cpumask(mm), &info);
+ consider_global_asid(mm);
+ }
+
mmu_notifier_arch_invalidate_secondary_tlbs(mm, start, end);
}
@@ -1527,19 +1504,16 @@ static void kernel_tlb_flush_range(struct flush_tlb_info *info)
void flush_tlb_kernel_range(unsigned long start, unsigned long end)
{
- struct flush_tlb_info *info;
+ struct flush_tlb_info info;
guard(preempt)();
+ init_flush_tlb_info(&info, NULL, start, end, PAGE_SHIFT, false,
+ TLB_GENERATION_INVALID);
- info = get_flush_tlb_info(NULL, start, end, PAGE_SHIFT, false,
- TLB_GENERATION_INVALID);
-
- if (info->end == TLB_FLUSH_ALL)
- kernel_tlb_flush_all(info);
+ if (info.end == TLB_FLUSH_ALL)
+ kernel_tlb_flush_all(&info);
else
- kernel_tlb_flush_range(info);
-
- put_flush_tlb_info();
+ kernel_tlb_flush_range(&info);
}
/*
@@ -1707,12 +1681,12 @@ EXPORT_SYMBOL_FOR_KVM(__flush_tlb_all);
void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch)
{
- struct flush_tlb_info *info;
-
+ struct flush_tlb_info info;
+ bool remote_flush = false;
int cpu = get_cpu();
- info = get_flush_tlb_info(NULL, 0, TLB_FLUSH_ALL, 0, false,
- TLB_GENERATION_INVALID);
+ init_flush_tlb_info(&info, NULL, 0, TLB_FLUSH_ALL, 0, false,
+ TLB_GENERATION_INVALID);
/*
* flush_tlb_multi() is not optimized for the common case in which only
* a local TLB flush is needed. Optimize this use-case by calling
@@ -1722,18 +1696,20 @@ void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch)
invlpgb_flush_all_nonglobals();
batch->unmapped_pages = false;
} else if (cpumask_any_but(&batch->cpumask, cpu) < nr_cpu_ids) {
- flush_tlb_multi(&batch->cpumask, info);
+ remote_flush = true;
} else if (cpumask_test_cpu(cpu, &batch->cpumask)) {
lockdep_assert_irqs_enabled();
local_irq_disable();
- flush_tlb_func(info);
+ flush_tlb_func(&info);
local_irq_enable();
}
- cpumask_clear(&batch->cpumask);
-
- put_flush_tlb_info();
put_cpu();
+
+ if (remote_flush)
+ flush_tlb_multi(&batch->cpumask, &info);
+
+ cpumask_clear(&batch->cpumask);
}
/*
diff --git a/arch/x86/xen/mmu_pv.c b/arch/x86/xen/mmu_pv.c
--- a/arch/x86/xen/mmu_pv.c
+++ b/arch/x86/xen/mmu_pv.c
@@ -836,15 +836,15 @@ static void xen_pgd_pin(struct mm_struct *mm)
*/
void xen_mm_pin_all(void)
{
- struct page *page;
+ struct ptdesc *ptdesc;
spin_lock(&init_mm.page_table_lock);
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
- if (!PagePinned(page)) {
- __xen_pgd_pin(&init_mm, (pgd_t *)page_address(page));
- SetPageSavePinned(page);
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
+ if (!PagePinned(ptdesc_page(ptdesc))) {
+ __xen_pgd_pin(&init_mm, (pgd_t *)ptdesc_address(ptdesc));
+ SetPageSavePinned(ptdesc_page(ptdesc));
}
}
@@ -947,16 +947,16 @@ static void xen_pgd_unpin(struct mm_struct *mm)
*/
void xen_mm_unpin_all(void)
{
- struct page *page;
+ struct ptdesc *ptdesc;
spin_lock(&init_mm.page_table_lock);
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
- if (PageSavePinned(page)) {
- BUG_ON(!PagePinned(page));
- __xen_pgd_unpin(&init_mm, (pgd_t *)page_address(page));
- ClearPageSavePinned(page);
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
+ if (PageSavePinned(ptdesc_page(ptdesc))) {
+ BUG_ON(!PagePinned(ptdesc_page(ptdesc)));
+ __xen_pgd_unpin(&init_mm, (pgd_t *)ptdesc_address(ptdesc));
+ ClearPageSavePinned(ptdesc_page(ptdesc));
}
}
@@ -0,0 +1,514 @@
diff --git a/include/linux/sched.h b/include/linux/sched.h
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -848,6 +848,17 @@ struct bore_ctx {
};
#endif /* CONFIG_SCHED_BORE */
+#if defined(CONFIG_SMP) && defined(CONFIG_PREEMPTION)
+struct task_ipi_mask {
+ union {
+ cpumask_t *ipi_mask_ptr;
+ unsigned long ipi_mask_val;
+ };
+};
+#else
+struct task_ipi_mask { };
+#endif
+
struct task_struct {
#ifdef CONFIG_THREAD_INFO_IN_TASK
/*
@@ -1387,6 +1398,7 @@ struct task_struct {
struct list_head perf_event_list;
struct perf_ctx_data __rcu *perf_ctx_data;
#endif
+ struct task_ipi_mask __private ipi_mask;
#ifdef CONFIG_DEBUG_PREEMPT
unsigned long preempt_disable_ip;
#endif
diff --git a/include/linux/smp.h b/include/linux/smp.h
--- a/include/linux/smp.h
+++ b/include/linux/smp.h
@@ -47,8 +47,7 @@ extern void __smp_call_single_queue(int cpu, struct llist_node *node);
/* total number of cpus in this system (may exceed NR_CPUS) */
extern unsigned int total_cpus;
-int smp_call_function_single(int cpuid, smp_call_func_t func, void *info,
- int wait);
+int smp_call_function_single(int cpuid, smp_call_func_t func, void *info, bool wait);
void on_each_cpu_cond_mask(smp_cond_func_t cond_func, smp_call_func_t func,
void *info, bool wait, const struct cpumask *mask);
@@ -239,6 +238,18 @@ static inline int get_boot_cpu_id(void)
#endif /* !SMP */
+#if defined(CONFIG_PREEMPTION) && defined(CONFIG_SMP)
+int smp_task_ipi_mask_alloc(struct task_struct *task);
+void smp_task_ipi_mask_free(struct task_struct *task);
+#else
+static inline int smp_task_ipi_mask_alloc(struct task_struct *task)
+{
+ return 0;
+}
+
+static inline void smp_task_ipi_mask_free(struct task_struct *task) { }
+#endif
+
/*
* raw_smp_processor_id() - get the current (unstable) CPU id
*
diff --git a/kernel/fork.c b/kernel/fork.c
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -547,6 +547,7 @@ void free_task(struct task_struct *tsk)
#endif
release_user_cpus_ptr(tsk);
scs_release(tsk);
+ smp_task_ipi_mask_free(tsk);
#ifndef CONFIG_THREAD_INFO_IN_TASK
/*
@@ -945,10 +946,14 @@ static struct task_struct *dup_task_struct(struct task_struct *orig, int node)
#endif
account_kernel_stack(tsk, 1);
- err = scs_prepare(tsk, node);
+ err = smp_task_ipi_mask_alloc(tsk);
if (err)
goto free_stack;
+ err = scs_prepare(tsk, node);
+ if (err)
+ goto free_ipi_mask;
+
#ifdef CONFIG_SECCOMP
/*
* We must handle setting up seccomp filters once we're under
@@ -1026,6 +1031,8 @@ static struct task_struct *dup_task_struct(struct task_struct *orig, int node)
#endif
return tsk;
+free_ipi_mask:
+ smp_task_ipi_mask_free(tsk);
free_stack:
exit_task_stack_account(tsk);
free_thread_stack(tsk);
diff --git a/kernel/scftorture.c b/kernel/scftorture.c
--- a/kernel/scftorture.c
+++ b/kernel/scftorture.c
@@ -348,6 +348,8 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
int ret = 0;
struct scf_check *scfcp = NULL;
struct scf_selector *scfsp = scf_sel_rand(trsp);
+ bool is_single = (scfsp->scfs_prim == SCF_PRIM_SINGLE ||
+ scfsp->scfs_prim == SCF_PRIM_SINGLE_RPC);
if (scfsp->scfs_prim == SCF_PRIM_SINGLE || scfsp->scfs_wait) {
scfcp = kmalloc_obj(*scfcp, GFP_ATOMIC);
@@ -364,8 +366,6 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
}
if (use_cpus_read_lock)
cpus_read_lock();
- else
- preempt_disable();
switch (scfsp->scfs_prim) {
case SCF_PRIM_RESCHED:
if (IS_BUILTIN(CONFIG_SCF_TORTURE_TEST)) {
@@ -411,13 +411,10 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
if (!ret) {
if (use_cpus_read_lock)
cpus_read_unlock();
- else
- preempt_enable();
+
wait_for_completion(&scfcp->scfc_completion);
if (use_cpus_read_lock)
cpus_read_lock();
- else
- preempt_disable();
} else {
scfp->n_single_rpc_ofl++;
scf_add_to_free_list(scfcp);
@@ -452,7 +449,7 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
scfcp->scfc_out = true;
}
if (scfcp && scfsp->scfs_wait) {
- if (WARN_ON_ONCE((num_online_cpus() > 1 || scfsp->scfs_prim == SCF_PRIM_SINGLE) &&
+ if (WARN_ON_ONCE(((use_cpus_read_lock && num_online_cpus() > 1) || is_single) &&
!scfcp->scfc_out)) {
pr_warn("%s: Memory-ordering failure, scfs_prim: %d.\n", __func__, scfsp->scfs_prim);
atomic_inc(&n_mb_out_errs); // Leak rather than trash!
@@ -463,8 +460,6 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
}
if (use_cpus_read_lock)
cpus_read_unlock();
- else
- preempt_enable();
if (allocfail)
schedule_timeout_idle((1 + longwait) * HZ); // Let no-wait handlers complete.
else if (!(torture_random(trsp) & 0xfff))
diff --git a/kernel/smp.c b/kernel/smp.c
--- a/kernel/smp.c
+++ b/kernel/smp.c
@@ -16,6 +16,7 @@
#include <linux/init.h>
#include <linux/interrupt.h>
#include <linux/gfp.h>
+#include <linux/slab.h>
#include <linux/smp.h>
#include <linux/cpu.h>
#include <linux/sched.h>
@@ -63,7 +64,14 @@ int smpcfd_prepare_cpu(unsigned int cpu)
free_cpumask_var(cfd->cpumask);
return -ENOMEM;
}
- cfd->csd = alloc_percpu(call_single_data_t);
+
+ /*
+ * Allocate the per-CPU CSD the first time a CPU comes up. It is
+ * not freed when the CPU is offlined, so csd_lock_wait() can access
+ * it even when the CPU was offlined after preemption was re-enabled.
+ */
+ if (!cfd->csd)
+ cfd->csd = alloc_percpu(call_single_data_t);
if (!cfd->csd) {
free_cpumask_var(cfd->cpumask);
free_cpumask_var(cfd->cpumask_ipi);
@@ -79,7 +87,6 @@ int smpcfd_dead_cpu(unsigned int cpu)
free_cpumask_var(cfd->cpumask);
free_cpumask_var(cfd->cpumask_ipi);
- free_percpu(cfd->csd);
return 0;
}
@@ -323,6 +330,8 @@ static void __csd_lock_wait(call_single_data_t *csd)
int bug_id = 0;
u64 ts0, ts1;
+ guard(preempt)();
+
ts1 = ts0 = ktime_get_mono_fast_ns();
for (;;) {
if (csd_lock_wait_toolong(csd, ts0, &ts1, &bug_id, &nmessages))
@@ -659,17 +668,9 @@ void flush_smp_call_function_queue(void)
local_irq_restore(flags);
}
-/**
- * smp_call_function_single - Run a function on a specific CPU
- * @cpu: Specific target CPU for this function.
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait until function has completed on other CPUs.
- *
- * Returns: %0 on success, else a negative status code.
- */
-int smp_call_function_single(int cpu, smp_call_func_t func, void *info,
- int wait)
+static int __smp_call_function_single(int cpu, smp_call_func_t func,
+ void *info, const struct cpumask *mask,
+ bool wait)
{
call_single_data_t *csd;
call_single_data_t csd_stack = {
@@ -686,6 +687,14 @@ int smp_call_function_single(int cpu, smp_call_func_t func, void *info,
*/
this_cpu = get_cpu();
+ if (mask) {
+ /* Try for same CPU (cheapest) */
+ if (!cpumask_test_cpu(this_cpu, mask))
+ cpu = sched_numa_find_nth_cpu(mask, 0, cpu_to_node(this_cpu));
+ else
+ cpu = this_cpu;
+ }
+
/*
* Can deadlock when called with interrupts disabled.
* We allow cpu's that are not yet online though, as no one else can
@@ -718,13 +727,32 @@ int smp_call_function_single(int cpu, smp_call_func_t func, void *info,
err = generic_exec_single(cpu, csd);
+ /*
+ * @csd is stack-allocated when @wait is true. No concurrent access
+ * except from the IPI completion path, so we can re-enable preemption
+ * early to reduce latency.
+ */
+ put_cpu();
+
if (wait)
csd_lock_wait(csd);
- put_cpu();
-
return err;
}
+
+/**
+ * smp_call_function_single - Run a function on a specific CPU
+ * @cpu: Specific target CPU for this function.
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait until function has completed on other CPUs.
+ *
+ * Returns: %0 on success, else a negative status code.
+ */
+int smp_call_function_single(int cpu, smp_call_func_t func, void *info, bool wait)
+{
+ return __smp_call_function_single(cpu, func, info, NULL, wait);
+}
EXPORT_SYMBOL(smp_call_function_single);
/**
@@ -775,10 +803,10 @@ EXPORT_SYMBOL_GPL(smp_call_function_single_async);
/**
* smp_call_function_any - Run a function on any of the given cpus
- * @mask: The mask of cpus it can run on.
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait until function has completed.
+ * @mask: The mask of cpus it can run on.
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait until function has completed.
*
* Selection preference:
* 1) current cpu if in @mask
@@ -789,20 +817,54 @@ EXPORT_SYMBOL_GPL(smp_call_function_single_async);
int smp_call_function_any(const struct cpumask *mask,
smp_call_func_t func, void *info, int wait)
{
- unsigned int cpu;
- int ret;
-
- /* Try for same CPU (cheapest) */
- cpu = get_cpu();
- if (!cpumask_test_cpu(cpu, mask))
- cpu = sched_numa_find_nth_cpu(mask, 0, cpu_to_node(cpu));
-
- ret = smp_call_function_single(cpu, func, info, wait);
- put_cpu();
- return ret;
+ return __smp_call_function_single(-1, func, info, mask, wait);
}
EXPORT_SYMBOL_GPL(smp_call_function_any);
+static DEFINE_STATIC_KEY_FALSE(ipi_mask_inlined);
+
+#ifdef CONFIG_PREEMPTION
+
+int smp_task_ipi_mask_alloc(struct task_struct *task)
+{
+ if (static_branch_unlikely(&ipi_mask_inlined))
+ return 0;
+
+ ACCESS_PRIVATE(task, ipi_mask).ipi_mask_ptr =
+ kmalloc(cpumask_size(), GFP_KERNEL);
+ if (!ACCESS_PRIVATE(task, ipi_mask).ipi_mask_ptr)
+ return -ENOMEM;
+
+ return 0;
+}
+
+void smp_task_ipi_mask_free(struct task_struct *task)
+{
+ if (static_branch_unlikely(&ipi_mask_inlined))
+ return;
+
+ kfree(ACCESS_PRIVATE(task, ipi_mask).ipi_mask_ptr);
+}
+
+static cpumask_t *smp_task_ipi_mask(struct task_struct *cur)
+{
+ /*
+ * If cpumask_size() is smaller than or equal to the pointer
+ * size, it stashes the cpumask in the pointer itself to
+ * avoid extra memory allocations.
+ */
+ if (static_branch_unlikely(&ipi_mask_inlined))
+ return (cpumask_t *)&ACCESS_PRIVATE(cur, ipi_mask).ipi_mask_val;
+
+ return ACCESS_PRIVATE(cur, ipi_mask).ipi_mask_ptr;
+}
+#else
+static cpumask_t *smp_task_ipi_mask(struct task_struct *cur)
+{
+ return NULL;
+}
+#endif
+
/*
* Flags to be used as scf_flags argument of smp_call_function_many_cond().
*
@@ -817,13 +879,20 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
unsigned int scf_flags,
smp_cond_func_t cond_func)
{
- int cpu, last_cpu, this_cpu = smp_processor_id();
- struct call_function_data *cfd;
+ struct cpumask *cpumask, *task_mask;
bool wait = scf_flags & SCF_WAIT;
- int nr_cpus = 0;
+ struct call_function_data *cfd;
+ int cpu, last_cpu, this_cpu;
bool run_remote = false;
+ int nr_cpus = 0;
- lockdep_assert_preemption_disabled();
+ this_cpu = get_cpu();
+ cfd = this_cpu_ptr(&cfd_data);
+ task_mask = smp_task_ipi_mask(current);
+ if (task_mask)
+ cpumask = task_mask;
+ else
+ cpumask = cfd->cpumask;
/*
* Can deadlock when called with interrupts disabled.
@@ -845,16 +914,15 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
/* Check if we need remote execution, i.e., any CPU excluding this one. */
if (cpumask_any_and_but(mask, cpu_online_mask, this_cpu) < nr_cpu_ids) {
- cfd = this_cpu_ptr(&cfd_data);
- cpumask_and(cfd->cpumask, mask, cpu_online_mask);
- __cpumask_clear_cpu(this_cpu, cfd->cpumask);
+ cpumask_and(cpumask, mask, cpu_online_mask);
+ __cpumask_clear_cpu(this_cpu, cpumask);
cpumask_clear(cfd->cpumask_ipi);
- for_each_cpu(cpu, cfd->cpumask) {
+ for_each_cpu(cpu, cpumask) {
call_single_data_t *csd = per_cpu_ptr(cfd->csd, cpu);
if (cond_func && !cond_func(cpu, info)) {
- __cpumask_clear_cpu(cpu, cfd->cpumask);
+ __cpumask_clear_cpu(cpu, cpumask);
continue;
}
@@ -904,8 +972,18 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
local_irq_restore(flags);
}
+ /*
+ * The IPI work has been queued and dispatched. On PREEMPT kernels,
+ * tasks created through dup_task_struct() have task-local wait masks.
+ * The boot init_task can fall back to cfd->cpumask when the mask is
+ * not inlined, but other tasks still use task-local masks and cannot
+ * overwrite it. On !PREEMPT kernels, preempt_enable() cannot schedule
+ * another task, so the per-CPU mask remains protected.
+ */
+ put_cpu();
+
if (run_remote && wait) {
- for_each_cpu(cpu, cfd->cpumask) {
+ for_each_cpu(cpu, cpumask) {
call_single_data_t *csd;
csd = per_cpu_ptr(cfd->csd, cpu);
@@ -916,15 +994,14 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
/**
* smp_call_function_many() - Run a function on a set of CPUs.
- * @mask: The set of cpus to run on (only runs on online subset).
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait (atomically) until function has completed
- * on other CPUs.
+ * @mask: The set of cpus to run on (only runs on online subset).
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait (atomically) until function has completed
+ * on other CPUs.
*
* You must not call this function with disabled interrupts or from a
- * hardware interrupt handler or from a bottom half handler. Preemption
- * must be disabled when calling this function.
+ * hardware interrupt handler or from a bottom half handler.
*
* @func is not called on the local CPU even if @mask contains it. Consider
* using on_each_cpu_cond_mask() instead if this is not desirable.
@@ -938,10 +1015,10 @@ EXPORT_SYMBOL(smp_call_function_many);
/**
* smp_call_function() - Run a function on all other CPUs.
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait (atomically) until function has completed
- * on other CPUs.
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait (atomically) until function has completed
+ * on other CPUs.
*
* If @wait is true, then returns once @func has returned; otherwise
* it returns just before the target cpu calls @func.
@@ -951,9 +1028,8 @@ EXPORT_SYMBOL(smp_call_function_many);
*/
void smp_call_function(smp_call_func_t func, void *info, int wait)
{
- preempt_disable();
- smp_call_function_many(cpu_online_mask, func, info, wait);
- preempt_enable();
+ smp_call_function_many_cond(cpu_online_mask, func, info,
+ wait ? SCF_WAIT : 0, NULL);
}
EXPORT_SYMBOL(smp_call_function);
@@ -1019,6 +1095,9 @@ EXPORT_SYMBOL(nr_cpu_ids);
void __init setup_nr_cpu_ids(void)
{
set_nr_cpu_ids(find_last_bit(cpumask_bits(cpu_possible_mask), NR_CPUS) + 1);
+
+ if (IS_ENABLED(CONFIG_PREEMPTION) && cpumask_size() <= sizeof(unsigned long))
+ static_branch_enable(&ipi_mask_inlined);
}
/* Called by boot processor to activate the rest. */
@@ -1055,12 +1134,14 @@ void __init smp_init(void)
* @func: The function to run on all applicable CPUs.
* This must be fast and non-blocking.
* @info: An arbitrary pointer to pass to both functions.
- * @wait: If true, wait (atomically) until function has
- * completed on other CPUs.
+ * @wait: If true, wait until function has completed on other CPUs.
* @mask: The set of cpus to run on (only runs on online subset).
*
- * Preemption is disabled to protect against CPUs going offline but not online.
- * CPUs going online during the call will not be seen or sent an IPI.
+ * Target CPU selection and work queueing are done with preemption
+ * disabled. This protects against CPUs going offline, but not against
+ * CPUs coming online concurrently; newly online CPUs are not guaranteed
+ * to be seen or sent an IPI. If @wait is true, the final wait for remote
+ * completion happens after that preemption-disabled section.
*
* You must not call this function with disabled interrupts or
* from a hardware interrupt handler or from a bottom half handler.
@@ -1073,9 +1154,7 @@ void on_each_cpu_cond_mask(smp_cond_func_t cond_func, smp_call_func_t func,
if (wait)
scf_flags |= SCF_WAIT;
- preempt_disable();
smp_call_function_many_cond(mask, func, info, scf_flags, cond_func);
- preempt_enable();
}
EXPORT_SYMBOL(on_each_cpu_cond_mask);
diff --git a/kernel/up.c b/kernel/up.c
--- a/kernel/up.c
+++ b/kernel/up.c
@@ -9,8 +9,7 @@
#include <linux/smp.h>
#include <linux/hypervisor.h>
-int smp_call_function_single(int cpu, void (*func) (void *info), void *info,
- int wait)
+int smp_call_function_single(int cpu, void (*func)(void *info), void *info, bool wait)
{
unsigned long flags;
@@ -0,0 +1,21 @@
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -13464,12 +13464,15 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
* still unbalanced. ld_moved simply stays zero, so it is
* correctly treated as an imbalance.
*/
- env.loop_max = min(sysctl_sched_nr_migrate, busiest->nr_running);
-
more_balance:
rq_lock_irqsave(busiest, &rf);
update_rq_clock(busiest);
+ if (!env.loop_max)
+ env.loop_max = min(sysctl_sched_nr_migrate, busiest->cfs.h_nr_queued);
+ else
+ env.loop_max = min(env.loop_max, busiest->cfs.h_nr_queued);
+
/*
* cur_ld_moved - load moved in current iteration
* ld_moved - cumulative load moved across iterations
@@ -0,0 +1,23 @@
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -3751,11 +3751,17 @@ static inline void ttwu_do_wakeup(struct task_struct *p)
void update_rq_avg_idle(struct rq *rq)
{
- u64 delta = rq_clock(rq) - rq->idle_stamp;
- u64 max = 2*rq->max_idle_balance_cost;
+ u64 idle_stamp = rq->idle_stamp;
+ u64 delta, max;
+
+ if (!idle_stamp)
+ return;
+
+ delta = rq_clock(rq) - idle_stamp;
update_avg(&rq->avg_idle, delta);
+ max = 2 * rq->max_idle_balance_cost;
if (rq->avg_idle > max)
rq->avg_idle = max;
rq->idle_stamp = 0;
@@ -0,0 +1,367 @@
diff --git a/arch/arm/include/asm/mmu_context.h b/arch/arm/include/asm/mmu_context.h
--- a/arch/arm/include/asm/mmu_context.h
+++ b/arch/arm/include/asm/mmu_context.h
@@ -80,7 +80,7 @@ static inline void check_and_switch_context(struct mm_struct *mm,
#ifndef MODULE
#define finish_arch_post_lock_switch \
finish_arch_post_lock_switch
-static inline void finish_arch_post_lock_switch(void)
+static __always_inline void finish_arch_post_lock_switch(void)
{
struct mm_struct *mm = current->mm;
diff --git a/arch/riscv/include/asm/sync_core.h b/arch/riscv/include/asm/sync_core.h
--- a/arch/riscv/include/asm/sync_core.h
+++ b/arch/riscv/include/asm/sync_core.h
@@ -6,7 +6,7 @@
* RISC-V implements return to user-space through an xRET instruction,
* which is not core serializing.
*/
-static inline void sync_core_before_usermode(void)
+static __always_inline void sync_core_before_usermode(void)
{
asm volatile ("fence.i" ::: "memory");
}
diff --git a/arch/s390/include/asm/mmu_context.h b/arch/s390/include/asm/mmu_context.h
--- a/arch/s390/include/asm/mmu_context.h
+++ b/arch/s390/include/asm/mmu_context.h
@@ -93,7 +93,7 @@ static inline void switch_mm(struct mm_struct *prev, struct mm_struct *next,
}
#define finish_arch_post_lock_switch finish_arch_post_lock_switch
-static inline void finish_arch_post_lock_switch(void)
+static __always_inline void finish_arch_post_lock_switch(void)
{
struct task_struct *tsk = current;
struct mm_struct *mm = tsk->mm;
diff --git a/arch/sparc/include/asm/mmu_context_64.h b/arch/sparc/include/asm/mmu_context_64.h
--- a/arch/sparc/include/asm/mmu_context_64.h
+++ b/arch/sparc/include/asm/mmu_context_64.h
@@ -160,7 +160,7 @@ static inline void arch_start_context_switch(struct task_struct *prev)
}
#define finish_arch_post_lock_switch finish_arch_post_lock_switch
-static inline void finish_arch_post_lock_switch(void)
+static __always_inline void finish_arch_post_lock_switch(void)
{
/* Restore the state of MCDPER register for the new process
* just switched to.
diff --git a/arch/x86/include/asm/sync_core.h b/arch/x86/include/asm/sync_core.h
--- a/arch/x86/include/asm/sync_core.h
+++ b/arch/x86/include/asm/sync_core.h
@@ -93,7 +93,7 @@ static __always_inline void sync_core(void)
* to user-mode. x86 implements return to user-space through sysexit,
* sysrel, and sysretq, which are not core serializing.
*/
-static inline void sync_core_before_usermode(void)
+static __always_inline void sync_core_before_usermode(void)
{
/* With PTI, we unconditionally serialize before running user code. */
if (static_cpu_has(X86_FEATURE_PTI))
diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h
--- a/include/linux/perf_event.h
+++ b/include/linux/perf_event.h
@@ -1632,7 +1632,7 @@ static inline void perf_event_task_migrate(struct task_struct *task)
task->sched_migrated = 1;
}
-static inline void perf_event_task_sched_in(struct task_struct *prev,
+static __always_inline void perf_event_task_sched_in(struct task_struct *prev,
struct task_struct *task)
{
if (static_branch_unlikely(&perf_sched_events))
diff --git a/include/linux/sched/mm.h b/include/linux/sched/mm.h
--- a/include/linux/sched/mm.h
+++ b/include/linux/sched/mm.h
@@ -32,7 +32,7 @@ extern struct mm_struct *mm_alloc(void);
* See also <Documentation/mm/active_mm.rst> for an in-depth explanation
* of &mm_struct.mm_count vs &mm_struct.mm_users.
*/
-static inline void mmgrab(struct mm_struct *mm)
+static __always_inline void mmgrab(struct mm_struct *mm)
{
atomic_inc(&mm->mm_count);
}
@@ -44,7 +44,7 @@ static inline void smp_mb__after_mmgrab(void)
extern void __mmdrop(struct mm_struct *mm);
-static inline void mmdrop(struct mm_struct *mm)
+static __always_inline void mmdrop(struct mm_struct *mm)
{
/*
* The implicit full barrier implied by atomic_dec_and_test() is
@@ -71,27 +71,27 @@ static inline void __mmdrop_delayed(struct rcu_head *rhp)
* Invoked from finish_task_switch(). Delegates the heavy lifting on RT
* kernels via RCU.
*/
-static inline void mmdrop_sched(struct mm_struct *mm)
+static __always_inline void mmdrop_sched(struct mm_struct *mm)
{
/* Provides a full memory barrier. See mmdrop() */
if (atomic_dec_and_test(&mm->mm_count))
call_rcu(&mm->delayed_drop, __mmdrop_delayed);
}
#else
-static inline void mmdrop_sched(struct mm_struct *mm)
+static __always_inline void mmdrop_sched(struct mm_struct *mm)
{
mmdrop(mm);
}
#endif
/* Helpers for lazy TLB mm refcounting */
-static inline void mmgrab_lazy_tlb(struct mm_struct *mm)
+static __always_inline void mmgrab_lazy_tlb(struct mm_struct *mm)
{
if (IS_ENABLED(CONFIG_MMU_LAZY_TLB_REFCOUNT))
mmgrab(mm);
}
-static inline void mmdrop_lazy_tlb(struct mm_struct *mm)
+static __always_inline void mmdrop_lazy_tlb(struct mm_struct *mm)
{
if (IS_ENABLED(CONFIG_MMU_LAZY_TLB_REFCOUNT)) {
mmdrop(mm);
@@ -104,7 +104,7 @@ static inline void mmdrop_lazy_tlb(struct mm_struct *mm)
}
}
-static inline void mmdrop_lazy_tlb_sched(struct mm_struct *mm)
+static __always_inline void mmdrop_lazy_tlb_sched(struct mm_struct *mm)
{
if (IS_ENABLED(CONFIG_MMU_LAZY_TLB_REFCOUNT))
mmdrop_sched(mm);
@@ -128,12 +128,12 @@ static inline void mmdrop_lazy_tlb_sched(struct mm_struct *mm)
* See also <Documentation/mm/active_mm.rst> for an in-depth explanation
* of &mm_struct.mm_count vs &mm_struct.mm_users.
*/
-static inline void mmget(struct mm_struct *mm)
+static __always_inline void mmget(struct mm_struct *mm)
{
atomic_inc(&mm->mm_users);
}
-static inline bool mmget_not_zero(struct mm_struct *mm)
+static __always_inline bool mmget_not_zero(struct mm_struct *mm)
{
return atomic_inc_not_zero(&mm->mm_users);
}
@@ -532,7 +532,7 @@ enum {
#include <asm/membarrier.h>
#endif
-static inline void membarrier_mm_sync_core_before_usermode(struct mm_struct *mm)
+static __always_inline void membarrier_mm_sync_core_before_usermode(struct mm_struct *mm)
{
/*
* The atomic_read() below prevents CSE. The following should
diff --git a/include/linux/tick.h b/include/linux/tick.h
--- a/include/linux/tick.h
+++ b/include/linux/tick.h
@@ -175,7 +175,7 @@ extern cpumask_var_t tick_nohz_full_mask;
#ifdef CONFIG_NO_HZ_FULL
extern bool tick_nohz_full_running;
-static inline bool tick_nohz_full_enabled(void)
+static __always_inline bool tick_nohz_full_enabled(void)
{
if (!context_tracking_enabled())
return false;
@@ -299,7 +299,7 @@ static inline void __tick_nohz_task_switch(void) { }
static inline void tick_nohz_full_setup(cpumask_var_t cpumask) { }
#endif
-static inline void tick_nohz_task_switch(void)
+static __always_inline void tick_nohz_task_switch(void)
{
if (tick_nohz_full_enabled())
__tick_nohz_task_switch();
diff --git a/include/linux/vtime.h b/include/linux/vtime.h
--- a/include/linux/vtime.h
+++ b/include/linux/vtime.h
@@ -89,24 +89,24 @@ static __always_inline void vtime_account_guest_exit(void)
* For now vtime state is tied to context tracking. We might want to decouple
* those later if necessary.
*/
-static inline bool vtime_accounting_enabled(void)
+static __always_inline bool vtime_accounting_enabled(void)
{
return context_tracking_enabled();
}
-static inline bool vtime_accounting_enabled_cpu(int cpu)
+static __always_inline bool vtime_accounting_enabled_cpu(int cpu)
{
return vtime_generic_enabled_cpu(cpu);
}
-static inline bool vtime_accounting_enabled_this_cpu(void)
+static __always_inline bool vtime_accounting_enabled_this_cpu(void)
{
return vtime_generic_enabled_this_cpu();
}
extern void vtime_task_switch_generic(struct task_struct *prev);
-static inline void vtime_task_switch(struct task_struct *prev)
+static __always_inline void vtime_task_switch(struct task_struct *prev)
{
if (vtime_accounting_enabled_this_cpu())
vtime_task_switch_generic(prev);
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -5099,7 +5099,7 @@ static inline void prepare_task(struct task_struct *next)
WRITE_ONCE(next->on_cpu, 1);
}
-static inline void finish_task(struct task_struct *prev)
+static __always_inline void finish_task(struct task_struct *prev)
{
/*
* This must be the very last reference to @prev from this CPU. After
@@ -5143,7 +5143,7 @@ static void zap_balance_callbacks(struct rq *rq)
rq->balance_callback = found ? &balance_push_callback : NULL;
}
-static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
+static __always_inline void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
{
void (*func)(struct rq *rq);
struct balance_callback *next;
@@ -5178,7 +5178,7 @@ struct balance_callback balance_push_callback = {
.func = balance_push,
};
-static inline struct balance_callback *
+static __always_inline struct balance_callback *
__splice_balance_callbacks(struct rq *rq, bool split)
{
struct balance_callback *head = rq->balance_callback;
@@ -5252,7 +5252,7 @@ prepare_lock_switch(struct rq *rq, struct task_struct *next, struct rq_flags *rf
__acquire(__rq_lockp(this_rq()));
}
-static inline void finish_lock_switch(struct rq *rq)
+static __always_inline void finish_lock_switch(struct rq *rq)
__releases(__rq_lockp(rq))
{
/*
@@ -5286,7 +5286,7 @@ static inline void kmap_local_sched_out(void)
#endif
}
-static inline void kmap_local_sched_in(void)
+static __always_inline void kmap_local_sched_in(void)
{
#ifdef CONFIG_KMAP_LOCAL
if (unlikely(current->kmap_ctrl.idx))
@@ -5340,7 +5340,7 @@ prepare_task_switch(struct rq *rq, struct task_struct *prev,
* past. 'prev == current' is still correct but we need to recalculate this_rq
* because prev may have moved to another CPU.
*/
-static struct rq *finish_task_switch(struct task_struct *prev)
+static __always_inline struct rq *finish_task_switch(struct task_struct *prev)
__releases(__rq_lockp(this_rq()))
{
struct rq *rq = this_rq();
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1460,12 +1460,12 @@ static inline struct cpumask *sched_group_span(struct sched_group *sg);
DECLARE_STATIC_KEY_FALSE(__sched_core_enabled);
-static inline bool sched_core_enabled(struct rq *rq)
+static __always_inline bool sched_core_enabled(struct rq *rq)
{
return static_branch_unlikely(&__sched_core_enabled) && rq->core_enabled;
}
-static inline bool sched_core_disabled(void)
+static __always_inline bool sched_core_disabled(void)
{
return !static_branch_unlikely(&__sched_core_enabled);
}
@@ -1474,7 +1474,7 @@ static inline bool sched_core_disabled(void)
* Be careful with this function; not for general use. The return value isn't
* stable unless you actually hold a relevant rq->__lock.
*/
-static inline raw_spinlock_t *rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *rq_lockp(struct rq *rq)
{
if (sched_core_enabled(rq))
return &rq->core->__lock;
@@ -1482,7 +1482,7 @@ static inline raw_spinlock_t *rq_lockp(struct rq *rq)
return &rq->__lock;
}
-static inline raw_spinlock_t *__rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *__rq_lockp(struct rq *rq)
__returns_ctx_lock(rq_lockp(rq)) /* alias them */
{
if (rq->core_enabled)
@@ -1585,12 +1585,12 @@ static inline bool sched_core_disabled(void)
return true;
}
-static inline raw_spinlock_t *rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *rq_lockp(struct rq *rq)
{
return &rq->__lock;
}
-static inline raw_spinlock_t *__rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *__rq_lockp(struct rq *rq)
__returns_ctx_lock(rq_lockp(rq)) /* alias them */
{
return &rq->__lock;
@@ -1650,33 +1650,33 @@ extern void raw_spin_rq_lock_nested(struct rq *rq, int subclass)
extern bool raw_spin_rq_trylock(struct rq *rq)
__cond_acquires(true, __rq_lockp(rq));
-static inline void raw_spin_rq_lock(struct rq *rq)
+static __always_inline void raw_spin_rq_lock(struct rq *rq)
__acquires(__rq_lockp(rq))
{
raw_spin_rq_lock_nested(rq, 0);
}
-static inline void raw_spin_rq_unlock(struct rq *rq)
+static __always_inline void raw_spin_rq_unlock(struct rq *rq)
__releases(__rq_lockp(rq))
{
raw_spin_unlock(rq_lockp(rq));
}
-static inline void raw_spin_rq_lock_irq(struct rq *rq)
+static __always_inline void raw_spin_rq_lock_irq(struct rq *rq)
__acquires(__rq_lockp(rq))
{
local_irq_disable();
raw_spin_rq_lock(rq);
}
-static inline void raw_spin_rq_unlock_irq(struct rq *rq)
+static __always_inline void raw_spin_rq_unlock_irq(struct rq *rq)
__releases(__rq_lockp(rq))
{
raw_spin_rq_unlock(rq);
local_irq_enable();
}
-static inline unsigned long _raw_spin_rq_lock_irqsave(struct rq *rq)
+static __always_inline unsigned long _raw_spin_rq_lock_irqsave(struct rq *rq)
__acquires(__rq_lockp(rq))
{
unsigned long flags;
@@ -1687,7 +1687,7 @@ static inline unsigned long _raw_spin_rq_lock_irqsave(struct rq *rq)
return flags;
}
-static inline void raw_spin_rq_unlock_irqrestore(struct rq *rq, unsigned long flags)
+static __always_inline void raw_spin_rq_unlock_irqrestore(struct rq *rq, unsigned long flags)
__releases(__rq_lockp(rq))
{
raw_spin_rq_unlock(rq);
@@ -0,0 +1,125 @@
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -3026,8 +3026,8 @@ static struct task_struct *pick_next_pushable_dl_task(struct rq *rq)
next_node = rb_first_cached(&rq->dl.pushable_dl_tasks_root);
while (next_node) {
i = __node_2_pdl(next_node);
- /* make sure task isn't on_cpu (possible with proxy-exec) */
- if (!task_on_cpu(rq, i)) {
+ /* skip tasks that cannot be migrated */
+ if (!task_on_cpu(rq, i) && !is_migration_disabled(i)) {
p = i;
break;
}
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -10691,17 +10691,40 @@ static enum llc_mig can_migrate_llc(int src_cpu, int dst_cpu,
return mig_llc;
}
+static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct *p)
+{
+ /*
+ * On asymmetric CPU capacity domains, do not let cache-aware
+ * balancing pull the task onto a destination CPU that cannot
+ * accommodate it. Doing so would turn the task into a misfit on
+ * the destination, trading a cache-locality gain for a capacity
+ * loss. If the task already does not fit its source CPU, the move
+ * cannot make things worse, so let the LLC preference decide.
+ */
+ if ((env->sd->flags & SD_ASYM_CPUCAPACITY) && p &&
+ !task_fits_cpu(p, env->dst_cpu) &&
+ task_fits_cpu(p, env->src_cpu))
+ return true;
+
+ return false;
+}
+
/*
* Check if task p can migrate from source LLC to
* destination LLC in terms of cache aware load balance.
*/
-static enum llc_mig can_migrate_llc_task(int src_cpu, int dst_cpu,
+static enum llc_mig can_migrate_llc_task(struct lb_env *env,
struct task_struct *p)
{
struct mm_struct *mm;
bool to_pref;
- int cpu;
+ int cpu, src_cpu, dst_cpu;
+ if (task_misfits_asym_cpu(env, p))
+ return mig_forbid;
+
+ src_cpu = env->src_cpu;
+ dst_cpu = env->dst_cpu;
mm = p->mm;
if (!mm)
return mig_unrestricted;
@@ -10758,6 +10781,14 @@ alb_break_llc(struct lb_env *env)
unsigned long util = 0;
struct task_struct *cur;
+ /*
+ * Migrating misfit tasks from current CPU
+ * to CPU with a better fit.
+ * Prioritize that over LLC preference.
+ */
+ if (env->migration_type == migrate_misfit)
+ return false;
+
if (env->src_rq->nr_running <= 1)
return true;
@@ -10765,7 +10796,8 @@ alb_break_llc(struct lb_env *env)
if (cur && cur->sched_class == &fair_sched_class)
util = task_util(cur);
- if (can_migrate_llc(env->src_cpu, env->dst_cpu,
+ if (task_misfits_asym_cpu(env, cur) ||
+ can_migrate_llc(env->src_cpu, env->dst_cpu,
util, false) == mig_forbid)
return true;
}
@@ -10805,8 +10837,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu))
return true;
- if (can_migrate_llc_task(env->src_cpu,
- env->dst_cpu, p) != mig_forbid)
+ if (can_migrate_llc_task(env, p) != mig_forbid)
return false;
return true;
@@ -11869,6 +11900,15 @@ static inline bool llc_balance(struct lb_env *env, struct sg_lb_stats *sgs,
if (env->sd->flags & SD_SHARE_LLC)
return false;
+ /*
+ * On asymmetric domains, group_misfit_task_load
+ * should be prioritized to move tasks to CPU that fit them
+ * over aggregating tasks to their preferred LLC.
+ */
+ if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
+ sgs->group_misfit_task_load)
+ return false;
+
/*
* Skip cache aware tagging if nr_balanced_failed is sufficiently high.
* Threshold of cache_nice_tries is set to 1 higher than nr_balance_failed
diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c
--- a/kernel/sched/rt.c
+++ b/kernel/sched/rt.c
@@ -1871,8 +1871,8 @@ static struct task_struct *pick_next_pushable_task(struct rq *rq)
return NULL;
plist_for_each_entry(i, head, pushable_tasks) {
- /* make sure task isn't on_cpu (possible with proxy-exec) */
- if (!task_on_cpu(rq, i)) {
+ /* skip tasks that cannot be migrated */
+ if (!task_on_cpu(rq, i) && !is_migration_disabled(i)) {
p = i;
break;
}
@@ -0,0 +1,24 @@
diff --git a/arch/x86/kernel/itmt.c b/arch/x86/kernel/itmt.c
--- a/arch/x86/kernel/itmt.c
+++ b/arch/x86/kernel/itmt.c
@@ -110,18 +110,14 @@ int sched_set_itmt_support(void)
arch_debugfs_dir,
&sysctl_sched_itmt_enabled,
&dfs_sched_itmt_fops);
- if (IS_ERR_OR_NULL(dfs_sched_itmt)) {
+ if (IS_ERR(dfs_sched_itmt))
dfs_sched_itmt = NULL;
- return -ENOMEM;
- }
dfs_sched_core_prio = debugfs_create_file("sched_core_priority", 0644,
arch_debugfs_dir, NULL,
&sched_core_priority_fops);
- if (IS_ERR_OR_NULL(dfs_sched_core_prio)) {
+ if (IS_ERR(dfs_sched_core_prio))
dfs_sched_core_prio = NULL;
- return -ENOMEM;
- }
sched_itmt_capable = true;
@@ -0,0 +1,124 @@
diff --git a/include/linux/sched/sd_flags.h b/include/linux/sched/sd_flags.h
--- a/include/linux/sched/sd_flags.h
+++ b/include/linux/sched/sd_flags.h
@@ -146,8 +146,7 @@ SD_FLAG(SD_ASYM_PACKING, SDF_NEEDS_GROUPS)
/*
* Prefer to place tasks in a sibling domain
*
- * Set up until domains start spanning NUMA nodes. Close to being a SHARED_CHILD
- * flag, but cleared below domains with SD_ASYM_CPUCAPACITY.
+ * Set up until domains start spanning NUMA nodes.
*
* NEEDS_GROUPS: Load balancing flag.
*/
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -12028,10 +12028,25 @@ static inline void update_sg_lb_stats(struct lb_env *env,
continue;
if (sd_flags & SD_ASYM_CPUCAPACITY) {
- /* Check for a misfit task on the cpu */
- if (sgs->group_misfit_task_load < rq->misfit_task_load) {
- sgs->group_misfit_task_load = rq->misfit_task_load;
- *sg_overloaded = 1;
+ if (rq->misfit_task_load) {
+ /*
+ * Always mark the root domain overloaded so big
+ * CPUs can pick up misfit tasks via newly idle
+ * balance.
+ */
+ if (balancing_at_rd)
+ *sg_overloaded = 1;
+
+ /*
+ * Only account misfit load if @dst_cpu can
+ * help; otherwise, the group may be classified
+ * as misfit_task and update_sd_pick_busiest()
+ * will skip it.
+ */
+ if (capacity_greater(capacity_of(env->dst_cpu),
+ group->sgc->max_capacity) &&
+ (sgs->group_misfit_task_load < rq->misfit_task_load))
+ sgs->group_misfit_task_load = rq->misfit_task_load;
}
} else if (env->idle && sched_reduced_capacity(rq, env->sd)) {
/* Check for a task running on a CPU with reduced capacity */
@@ -12110,6 +12125,17 @@ static bool update_sd_pick_busiest(struct lb_env *env,
sds->local_stat.group_type != group_has_spare))
return false;
+ /*
+ * Candidate sg has no more than one task per CPU and has higher
+ * per-CPU capacity. Migrating tasks to less capable CPUs may harm
+ * throughput. Maximize throughput, power/energy consequences are not
+ * considered.
+ */
+ if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
+ (sgs->group_type <= group_fully_busy) &&
+ (capacity_greater(sg->sgc->min_capacity, capacity_of(env->dst_cpu))))
+ return false;
+
if (sgs->group_type > busiest->group_type)
return true;
@@ -12216,17 +12242,6 @@ static bool update_sd_pick_busiest(struct lb_env *env,
break;
}
- /*
- * Candidate sg has no more than one task per CPU and has higher
- * per-CPU capacity. Migrating tasks to less capable CPUs may harm
- * throughput. Maximize throughput, power/energy consequences are not
- * considered.
- */
- if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
- (sgs->group_type <= group_fully_busy) &&
- (capacity_greater(sg->sgc->min_capacity, capacity_of(env->dst_cpu))))
- return false;
-
return true;
}
@@ -13144,9 +13159,24 @@ static struct rq *sched_balance_find_src_rq(struct lb_env *env,
* average load.
*/
if (env->sd->flags & SD_ASYM_CPUCAPACITY &&
- !capacity_greater(capacity_of(env->dst_cpu), capacity) &&
- nr_running == 1)
- continue;
+ nr_running == 1) {
+ bool cluster_equal_cap = static_branch_unlikely(&sched_cluster_active) &&
+ (get_actual_cpu_capacity(env->dst_cpu) ==
+ get_actual_cpu_capacity(i));
+ bool smt_degraded_cap = sched_smt_active() && !is_core_idle(i);
+
+ /*
+ * Busy SMT siblings reduce the capacity of CPU @i. Do
+ * not skip it in this case.
+ *
+ * CONFIG_SCHED_CLUSTER requires balancing load across
+ * clusters of identical capacity, accounting for
+ * hardware and cpufreq pressure.
+ */
+ if (!smt_degraded_cap && !cluster_equal_cap &&
+ !capacity_greater(capacity_of(env->dst_cpu), capacity))
+ continue;
+ }
/*
* Make sure we only pull tasks from a CPU of lower priority
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -1995,10 +1995,6 @@ sd_init(struct sched_domain_topology_level *tl,
/*
* Convert topological properties into behaviour.
*/
- /* Don't attempt to spread across CPUs of different capacities. */
- if ((sd->flags & SD_ASYM_CPUCAPACITY) && sd->child)
- sd->child->flags &= ~SD_PREFER_SIBLING;
-
if (sd->flags & SD_SHARE_CPUCAPACITY) {
sd->imbalance_pct = 110;
@@ -0,0 +1,77 @@
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -14050,29 +14050,62 @@ static inline int on_null_domain(struct rq *rq)
*/
static inline int find_new_ilb(void)
{
- int this_cpu = smp_processor_id();
- const struct cpumask *hk_mask;
- int ilb_cpu;
+ struct cpumask *ilb_cpus;
+ int ilb_cpu, fallback = -1;
- hk_mask = housekeeping_cpumask(HK_TYPE_KERNEL_NOISE);
+ lockdep_assert_irqs_disabled();
- for_each_cpu_and(ilb_cpu, nohz.idle_cpus_mask, hk_mask) {
- if (ilb_cpu == this_cpu)
+ /*
+ * Reuse the per-CPU select_rq_mask, which is protected from concurrent
+ * use on this CPU by having interrupts disabled.
+ */
+ ilb_cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
+ cpumask_and(ilb_cpus, nohz.idle_cpus_mask,
+ housekeeping_cpumask(HK_TYPE_KERNEL_NOISE));
+
+ for_each_cpu(ilb_cpu, ilb_cpus) {
+ if (!idle_cpu(ilb_cpu)) {
+ /*
+ * Once an idle fallback exists, a busy CPU proves that
+ * this core cannot be fully idle. Skip its siblings.
+ */
+ if (sched_smt_active() && fallback >= 0)
+ cpumask_andnot(ilb_cpus, ilb_cpus, cpu_smt_mask(ilb_cpu));
continue;
+ }
- if (idle_cpu(ilb_cpu))
- return ilb_cpu;
+ /*
+ * Running the idle load balancer on an idle sibling of a busy
+ * SMT core can reduce the capacity available to its sibling. Prefer
+ * a CPU whose entire core is idle, but retain the first idle CPU as
+ * a fallback so idle balancing can still make progress when no fully
+ * idle core exists.
+ */
+ if (sched_smt_active() && !is_core_idle(ilb_cpu)) {
+ if (fallback < 0)
+ fallback = ilb_cpu;
+
+ /*
+ * The core is not idle, so there is no need to check
+ * any of its other SMT siblings.
+ */
+ cpumask_andnot(ilb_cpus, ilb_cpus,
+ cpu_smt_mask(ilb_cpu));
+ continue;
+ }
+
+ return ilb_cpu;
}
- return -1;
+ return fallback;
}
/*
* Kick a CPU to do the NOHZ balancing, if it is time for it, via a cross-CPU
* SMP function call (IPI).
*
- * We pick the first idle CPU in the HK_TYPE_KERNEL_NOISE housekeeping set
- * (if there is one).
+ * Prefer a CPU on a fully idle core in the HK_TYPE_KERNEL_NOISE housekeeping
+ * set. Fall back to the first idle CPU when no fully idle core exists.
*/
static void kick_ilb(unsigned int flags)
{
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,61 @@
diff --git a/drivers/idle/intel_idle.c b/drivers/idle/intel_idle.c
--- a/drivers/idle/intel_idle.c
+++ b/drivers/idle/intel_idle.c
@@ -53,6 +53,7 @@
#include <linux/notifier.h>
#include <linux/cpu.h>
#include <linux/moduleparam.h>
+#include <linux/pm_qos.h>
#include <linux/sysfs.h>
#include <asm/cpuid/api.h>
#include <asm/cpu_device_id.h>
@@ -2697,6 +2698,9 @@ static void __init cmdline_table_adjust(struct cpuidle_driver *drv)
pr_info("Failed to adjust C-states with data from 'intel_idle.table'\n");
}
+#define INTEL_IDLE_INIT_QOS 20
+static struct pm_qos_request qos_req __initdata;
+
static int __init intel_idle_init(void)
{
const struct x86_cpu_id *id;
@@ -2766,6 +2770,13 @@ static int __init intel_idle_init(void)
if (retval)
pr_warn("failed to initialized sysfs");
+ /*
+ * Some platforms, in particular the Intel S1200BTL motherboard, have a
+ * problem with using package idle states too early, so prevent that
+ * from taking place until the device_initcall() phase is over.
+ */
+ cpu_latency_qos_add_request(&qos_req, INTEL_IDLE_INIT_QOS);
+
retval = cpuidle_register_driver(&intel_idle_driver);
if (retval) {
struct cpuidle_driver *drv = cpuidle_get_driver();
@@ -2790,6 +2801,9 @@ static int __init intel_idle_init(void)
intel_idle_cpuidle_devices_uninit();
cpuidle_unregister_driver(&intel_idle_driver);
init_driver_fail:
+ if (cpu_latency_qos_request_active((&qos_req)))
+ cpu_latency_qos_remove_request(&qos_req);
+
intel_idle_sysfs_uninit();
free_percpu(intel_idle_cpuidle_devices);
return retval;
@@ -2797,6 +2811,15 @@ static int __init intel_idle_init(void)
}
subsys_initcall_sync(intel_idle_init);
+static int __init intel_idle_init_complete(void)
+{
+ if (cpu_latency_qos_request_active((&qos_req)))
+ cpu_latency_qos_remove_request(&qos_req);
+
+ return 0;
+}
+device_initcall_sync(intel_idle_init_complete);
+
/*
* We are not really modular, but we used to support that. Meaning we also
* support "intel_idle.max_cstate=..." at boot and also a read-only export of
@@ -0,0 +1,480 @@
diff --git a/drivers/cpufreq/amd-pstate.c b/drivers/cpufreq/amd-pstate.c
--- a/drivers/cpufreq/amd-pstate.c
+++ b/drivers/cpufreq/amd-pstate.c
@@ -782,6 +782,7 @@ static unsigned int amd_pstate_fast_switch(struct cpufreq_policy *policy,
static void amd_pstate_adjust_perf(struct cpufreq_policy *policy,
unsigned long _min_perf,
unsigned long target_perf,
+ unsigned long _max_perf,
unsigned long capacity)
{
u8 max_perf, min_perf, des_perf, cap_perf;
diff --git a/drivers/cpufreq/cpufreq.c b/drivers/cpufreq/cpufreq.c
--- a/drivers/cpufreq/cpufreq.c
+++ b/drivers/cpufreq/cpufreq.c
@@ -2225,14 +2225,17 @@ EXPORT_SYMBOL_GPL(cpufreq_driver_fast_switch);
* @policy: cpufreq policy object of the target CPU.
* @min_perf: Minimum (required) performance level (units of @capacity).
* @target_perf: Target (desired) performance level (units of @capacity).
+ * @max_perf: Maximum (allowed) performance level (units of @capacity).
* @capacity: Capacity of the target CPU.
*
- * Carry out a fast performance level switch of @cpu without sleeping.
+ * Carry out a fast performance level adjustment for the CPU represented by
+ * @policy without sleeping.
*
* The driver's ->adjust_perf() callback invoked by this function must be
- * suitable for being called from within RCU-sched read-side critical sections
- * and it is expected to select a suitable performance level equal to or above
- * @min_perf and preferably equal to or below @target_perf.
+ * suitable for calling from within RCU-sched read-side critical sections and
+ * it is expected to program the processor to select suitable performance
+ * levels between @min_perf and @max_perf inclusive and preferably close to
+ * @target_perf going forward for the CPU represented by @policy.
*
* This function must not be called if policy->fast_switch_enabled is unset.
*
@@ -2244,9 +2247,10 @@ EXPORT_SYMBOL_GPL(cpufreq_driver_fast_switch);
void cpufreq_driver_adjust_perf(struct cpufreq_policy *policy,
unsigned long min_perf,
unsigned long target_perf,
+ unsigned long max_perf,
unsigned long capacity)
{
- cpufreq_driver->adjust_perf(policy, min_perf, target_perf, capacity);
+ cpufreq_driver->adjust_perf(policy, min_perf, target_perf, max_perf, capacity);
}
/**
diff --git a/drivers/cpufreq/intel_pstate.c b/drivers/cpufreq/intel_pstate.c
--- a/drivers/cpufreq/intel_pstate.c
+++ b/drivers/cpufreq/intel_pstate.c
@@ -299,7 +299,6 @@ struct pstate_funcs {
static struct pstate_funcs pstate_funcs __read_mostly;
static bool hwp_active __ro_after_init;
-static int hwp_mode_bdw __ro_after_init;
static bool per_cpu_limits __ro_after_init;
static bool hwp_forced __ro_after_init;
static bool hwp_boost __read_mostly;
@@ -587,21 +586,14 @@ static void intel_pstate_hybrid_hwp_adjust(struct cpudata *cpu)
hwp_is_hybrid = true;
- cpu->pstate.turbo_freq = rounddown(cpu->pstate.turbo_pstate * scaling,
- perf_ctl_scaling);
- cpu->pstate.max_freq = rounddown(cpu->pstate.max_pstate * scaling,
- perf_ctl_scaling);
-
freq = perf_ctl_max_phys * perf_ctl_scaling;
cpu->pstate.max_pstate_physical = intel_pstate_freq_to_hwp(cpu, freq);
- freq = cpu->pstate.min_pstate * perf_ctl_scaling;
- cpu->pstate.min_freq = freq;
/*
* Cast the min P-state value retrieved via pstate_funcs.get_min() to
* the effective range of HWP performance levels.
*/
- cpu->pstate.min_pstate = intel_pstate_freq_to_hwp(cpu, freq);
+ cpu->pstate.min_pstate = intel_pstate_freq_to_hwp(cpu, cpu->pstate.min_freq);
}
static bool turbo_is_disabled(void)
@@ -979,12 +971,10 @@ static int hybrid_get_cost(struct device *dev, unsigned long freq,
* capacity. Similarly, P-cores start to be populated when E-cores are
* utilized above 60% of the capacity.
*/
- if (hybrid_get_cpu_type(dev->id) == INTEL_CPU_TYPE_ATOM) {
- if (hybrid_has_l3(dev->id)) /* E-core */
- *cost += 1;
- } else { /* P-core */
+ if (hybrid_get_cpu_type(dev->id) == INTEL_CPU_TYPE_CORE) /* P-core */
*cost += 2;
- }
+ else if (hybrid_has_l3(dev->id)) /* E-core */
+ *cost += 1;
return 0;
}
@@ -1185,6 +1175,22 @@ static bool hybrid_clear_max_perf_cpu(void)
return ret;
}
+static void intel_pstate_update_freq_limits(struct cpudata *cpu)
+{
+ int scaling = cpu->pstate.scaling;
+ unsigned int turbo_freq = cpu->pstate.turbo_pstate * scaling;
+ unsigned int max_freq = cpu->pstate.max_pstate * scaling;
+ int perf_ctl_scaling = cpu->pstate.perf_ctl_scaling;
+
+ if (scaling != perf_ctl_scaling) {
+ turbo_freq = rounddown(turbo_freq, perf_ctl_scaling);
+ max_freq = rounddown(max_freq, perf_ctl_scaling);
+ }
+
+ cpu->pstate.turbo_freq = turbo_freq;
+ cpu->pstate.max_freq = max_freq;
+}
+
static void __intel_pstate_get_hwp_cap(struct cpudata *cpu)
{
u64 cap;
@@ -1197,20 +1203,8 @@ static void __intel_pstate_get_hwp_cap(struct cpudata *cpu)
static void intel_pstate_get_hwp_cap(struct cpudata *cpu)
{
- int scaling = cpu->pstate.scaling;
-
__intel_pstate_get_hwp_cap(cpu);
-
- cpu->pstate.max_freq = cpu->pstate.max_pstate * scaling;
- cpu->pstate.turbo_freq = cpu->pstate.turbo_pstate * scaling;
- if (scaling != cpu->pstate.perf_ctl_scaling) {
- int perf_ctl_scaling = cpu->pstate.perf_ctl_scaling;
-
- cpu->pstate.max_freq = rounddown(cpu->pstate.max_freq,
- perf_ctl_scaling);
- cpu->pstate.turbo_freq = rounddown(cpu->pstate.turbo_freq,
- perf_ctl_scaling);
- }
+ intel_pstate_update_freq_limits(cpu);
}
static void hybrid_update_capacity(struct cpudata *cpu)
@@ -2299,33 +2293,16 @@ static int hwp_get_cpu_scaling(int cpu)
return intel_pstate_cppc_get_scaling(cpu);
}
-static void intel_pstate_set_pstate(struct cpudata *cpu, int pstate)
-{
- trace_cpu_frequency(pstate * cpu->pstate.scaling, cpu->cpu);
- cpu->pstate.current_pstate = pstate;
- /*
- * Generally, there is no guarantee that this code will always run on
- * the CPU being updated, so force the register update to run on the
- * right CPU.
- */
- wrmsrq_on_cpu(cpu->cpu, MSR_IA32_PERF_CTL,
- pstate_funcs.get_val(cpu, pstate));
-}
-
-static void intel_pstate_set_min_pstate(struct cpudata *cpu)
-{
- intel_pstate_set_pstate(cpu, cpu->pstate.min_pstate);
-}
-
static void intel_pstate_get_cpu_pstates(struct cpudata *cpu)
{
int perf_ctl_scaling = pstate_funcs.get_scaling();
cpu->pstate.max_pstate_physical = pstate_funcs.get_max_physical(cpu->cpu);
cpu->pstate.min_pstate = pstate_funcs.get_min(cpu->cpu);
+ cpu->pstate.min_freq = cpu->pstate.min_pstate * perf_ctl_scaling;
cpu->pstate.perf_ctl_scaling = perf_ctl_scaling;
- if (hwp_active && !hwp_mode_bdw) {
+ if (hwp_active) {
__intel_pstate_get_hwp_cap(cpu);
if (pstate_funcs.get_cpu_scaling) {
@@ -2345,19 +2322,13 @@ static void intel_pstate_get_cpu_pstates(struct cpudata *cpu)
cpu->pstate.turbo_pstate = pstate_funcs.get_turbo(cpu->cpu);
}
- if (cpu->pstate.scaling == perf_ctl_scaling) {
- cpu->pstate.min_freq = cpu->pstate.min_pstate * perf_ctl_scaling;
- cpu->pstate.max_freq = cpu->pstate.max_pstate * perf_ctl_scaling;
- cpu->pstate.turbo_freq = cpu->pstate.turbo_pstate * perf_ctl_scaling;
- }
+ intel_pstate_update_freq_limits(cpu);
if (pstate_funcs.get_aperf_mperf_shift)
cpu->aperf_mperf_shift = pstate_funcs.get_aperf_mperf_shift();
if (pstate_funcs.get_vid)
pstate_funcs.get_vid(cpu);
-
- intel_pstate_set_min_pstate(cpu);
}
/*
@@ -2884,8 +2855,22 @@ static void intel_pstate_update_perf_limits(struct cpudata *cpu,
cpu->min_perf_ratio);
}
+static void intel_pstate_set_pstate(struct cpudata *cpu, int pstate)
+{
+ trace_cpu_frequency(pstate * cpu->pstate.scaling, cpu->cpu);
+ cpu->pstate.current_pstate = pstate;
+ /*
+ * Generally, there is no guarantee that this code will always run on
+ * the CPU being updated, so force the register update to run on the
+ * right CPU.
+ */
+ wrmsrq_on_cpu(cpu->cpu, MSR_IA32_PERF_CTL,
+ pstate_funcs.get_val(cpu, pstate));
+}
+
static int intel_pstate_set_policy(struct cpufreq_policy *policy)
{
+ unsigned int freq = policy->min;
struct cpudata *cpu;
if (!policy->cpuinfo.max_freq)
@@ -2901,7 +2886,23 @@ static int intel_pstate_set_policy(struct cpufreq_policy *policy)
intel_pstate_update_perf_limits(cpu, policy->min, policy->max);
- if (cpu->policy == CPUFREQ_POLICY_PERFORMANCE) {
+ if (hwp_active) {
+ /*
+ * The active mode only requires an update util hook if HWP
+ * boost is used and the policy is not "performance".
+ */
+ if (hwp_boost && cpu->policy != CPUFREQ_POLICY_PERFORMANCE) {
+ intel_pstate_set_update_util_hook(policy->cpu);
+ } else {
+ intel_pstate_clear_update_util_hook(policy->cpu);
+ if (cpu->policy == CPUFREQ_POLICY_PERFORMANCE) {
+ freq = cpu->max_perf_ratio * cpu->pstate.scaling;
+ if (cpu->pstate.scaling != cpu->pstate.perf_ctl_scaling)
+ freq = rounddown(freq, cpu->pstate.perf_ctl_scaling);
+ }
+ }
+ intel_pstate_hwp_set(policy->cpu);
+ } else if (cpu->policy == CPUFREQ_POLICY_PERFORMANCE) {
int pstate = max(cpu->pstate.min_pstate, cpu->max_perf_ratio);
/*
@@ -2910,25 +2911,17 @@ static int intel_pstate_set_policy(struct cpufreq_policy *policy)
*/
intel_pstate_clear_update_util_hook(policy->cpu);
intel_pstate_set_pstate(cpu, pstate);
+ freq = pstate * cpu->pstate.scaling;
} else {
intel_pstate_set_update_util_hook(policy->cpu);
}
-
- if (hwp_active) {
- /*
- * When hwp_boost was active before and dynamically it
- * was turned off, in that case we need to clear the
- * update util hook.
- */
- if (!hwp_boost)
- intel_pstate_clear_update_util_hook(policy->cpu);
- intel_pstate_hwp_set(policy->cpu);
- }
/*
- * policy->cur is never updated with the intel_pstate driver, but it
- * is used as a stale frequency value. So, keep it within limits.
+ * policy->cur is never updated in the intel_pstate driver, but it is
+ * used as a stale frequency value, so set it to reflect the actual
+ * requested P-state in the "performance" policy case and to the min
+ * otherwise.
*/
- policy->cur = policy->min;
+ policy->cur = freq;
mutex_unlock(&intel_pstate_limits_lock);
@@ -2971,6 +2964,11 @@ static int intel_pstate_verify_policy(struct cpufreq_policy_data *policy)
return 0;
}
+static void intel_pstate_set_min_pstate(struct cpudata *cpu)
+{
+ intel_pstate_set_pstate(cpu, cpu->pstate.min_pstate);
+}
+
static int intel_cpufreq_cpu_offline(struct cpufreq_policy *policy)
{
struct cpudata *cpu = all_cpu_data[policy->cpu];
@@ -3063,6 +3061,7 @@ static int __intel_pstate_cpu_init(struct cpufreq_policy *policy)
static int intel_pstate_cpu_init(struct cpufreq_policy *policy)
{
int ret = __intel_pstate_cpu_init(policy);
+ struct cpudata *cpu;
if (ret)
return ret;
@@ -3073,11 +3072,11 @@ static int intel_pstate_cpu_init(struct cpufreq_policy *policy)
*/
policy->policy = CPUFREQ_POLICY_POWERSAVE;
- if (hwp_active) {
- struct cpudata *cpu = all_cpu_data[policy->cpu];
-
+ cpu = all_cpu_data[policy->cpu];
+ if (hwp_active)
cpu->epp_cached = intel_pstate_get_epp(cpu, 0);
- }
+ else
+ intel_pstate_set_min_pstate(cpu);
return 0;
}
@@ -3243,6 +3242,7 @@ static unsigned int intel_cpufreq_fast_switch(struct cpufreq_policy *policy,
static void intel_cpufreq_adjust_perf(struct cpufreq_policy *policy,
unsigned long min_perf,
unsigned long target_perf,
+ unsigned long max_perf,
unsigned long capacity)
{
struct cpudata *cpu = all_cpu_data[policy->cpu];
@@ -3273,7 +3273,13 @@ static void intel_cpufreq_adjust_perf(struct cpufreq_policy *policy,
if (min_pstate > cpu->max_perf_ratio)
min_pstate = cpu->max_perf_ratio;
- max_pstate = min(cap_pstate, cpu->max_perf_ratio);
+ max_pstate = cap_pstate;
+ if (max_perf < capacity)
+ max_pstate = DIV_ROUND_UP(cap_pstate * max_perf, capacity);
+
+ if (max_pstate > cpu->max_perf_ratio)
+ max_pstate = cpu->max_perf_ratio;
+
if (max_pstate < min_pstate)
max_pstate = min_pstate;
@@ -3301,8 +3307,6 @@ static int intel_cpufreq_cpu_init(struct cpufreq_policy *policy)
return ret;
policy->cpuinfo.transition_latency = INTEL_CPUFREQ_TRANSITION_LATENCY;
- /* This reflects the intel_pstate_get_cpu_pstates() setting. */
- policy->cur = policy->cpuinfo.min_freq;
req = kzalloc_objs(*req, 2);
if (!req) {
@@ -3323,9 +3327,15 @@ static int intel_cpufreq_cpu_init(struct cpufreq_policy *policy)
WRITE_ONCE(cpu->hwp_req_cached, value);
cpu->epp_cached = intel_pstate_get_epp(cpu, value);
+
+ intel_cpufreq_hwp_update(cpu, cpu->pstate.min_pstate,
+ cpu->pstate.max_pstate,
+ cpu->pstate.min_pstate, false);
} else {
policy->transition_delay_us = INTEL_CPUFREQ_TRANSITION_DELAY;
+ intel_pstate_set_min_pstate(cpu);
}
+ policy->cur = policy->cpuinfo.min_freq;
freq = DIV_ROUND_UP(cpu->pstate.turbo_freq * global.min_perf_pct, 100);
@@ -3676,14 +3686,14 @@ static inline bool intel_pstate_has_acpi_ppc(void) { return false; }
static inline void intel_pstate_request_control_from_smm(void) {}
#endif /* CONFIG_ACPI */
-#define INTEL_PSTATE_HWP_BROADWELL 0x01
+#define INTEL_PSTATE_HWP_NOT_HYBRID 0x01
#define X86_MATCH_HWP(vfm, hwp_mode) \
X86_MATCH_VFM_FEATURE(vfm, X86_FEATURE_HWP, hwp_mode)
static const struct x86_cpu_id hwp_support_ids[] __initconst = {
- X86_MATCH_HWP(INTEL_BROADWELL_X, INTEL_PSTATE_HWP_BROADWELL),
- X86_MATCH_HWP(INTEL_BROADWELL_D, INTEL_PSTATE_HWP_BROADWELL),
+ X86_MATCH_HWP(INTEL_BROADWELL_X, INTEL_PSTATE_HWP_NOT_HYBRID),
+ X86_MATCH_HWP(INTEL_BROADWELL_D, INTEL_PSTATE_HWP_NOT_HYBRID),
X86_MATCH_HWP(INTEL_ANY, 0),
{}
};
@@ -3808,7 +3818,6 @@ static int __init intel_pstate_init(void)
if (!no_hwp) {
hwp_active = true;
- hwp_mode_bdw = id->driver_data;
intel_pstate.attr = hwp_cpufreq_attrs;
intel_cpufreq.attr = hwp_cpufreq_attrs;
intel_cpufreq.flags |= CPUFREQ_NEED_UPDATE_LIMITS;
@@ -3816,7 +3825,8 @@ static int __init intel_pstate_init(void)
if (!default_driver)
default_driver = &intel_pstate;
- pstate_funcs.get_cpu_scaling = hwp_get_cpu_scaling;
+ if (!id->driver_data)
+ pstate_funcs.get_cpu_scaling = hwp_get_cpu_scaling;
goto hwp_cpu_matched;
}
diff --git a/include/linux/cpufreq.h b/include/linux/cpufreq.h
--- a/include/linux/cpufreq.h
+++ b/include/linux/cpufreq.h
@@ -379,6 +379,7 @@ struct cpufreq_driver {
void (*adjust_perf)(struct cpufreq_policy *policy,
unsigned long min_perf,
unsigned long target_perf,
+ unsigned long max_perf,
unsigned long capacity);
/*
@@ -624,6 +625,7 @@ unsigned int cpufreq_driver_fast_switch(struct cpufreq_policy *policy,
void cpufreq_driver_adjust_perf(struct cpufreq_policy *policy,
unsigned long min_perf,
unsigned long target_perf,
+ unsigned long max_perf,
unsigned long capacity);
bool cpufreq_driver_has_adjust_perf(void);
int cpufreq_driver_target(struct cpufreq_policy *policy,
diff --git a/kernel/sched/cpufreq_schedutil.c b/kernel/sched/cpufreq_schedutil.c
--- a/kernel/sched/cpufreq_schedutil.c
+++ b/kernel/sched/cpufreq_schedutil.c
@@ -50,6 +50,7 @@ struct sugov_cpu {
unsigned long util;
unsigned long bw_min;
+ unsigned long bw_max;
/* The field below is for single-CPU policies only: */
#ifdef CONFIG_NO_HZ_COMMON
@@ -243,6 +244,7 @@ static void sugov_get_util(struct sugov_cpu *sg_cpu, unsigned long boost)
util = effective_cpu_util(sg_cpu->cpu, util, &min, &max);
util = max(util, boost);
sg_cpu->bw_min = min;
+ sg_cpu->bw_max = max;
sg_cpu->util = sugov_effective_cpu_perf(sg_cpu->cpu, util, min, max);
}
@@ -495,7 +497,7 @@ static void sugov_update_single_perf(struct update_util_data *hook, u64 time,
sg_cpu->util = prev_util;
cpufreq_driver_adjust_perf(sg_policy->policy, sg_cpu->bw_min,
- sg_cpu->util, max_cap);
+ sg_cpu->util, sg_cpu->bw_max, max_cap);
sg_policy->need_freq_update = false;
sg_policy->last_freq_update_time = time;
diff --git a/rust/kernel/cpufreq.rs b/rust/kernel/cpufreq.rs
--- a/rust/kernel/cpufreq.rs
+++ b/rust/kernel/cpufreq.rs
@@ -792,7 +792,13 @@ fn fast_switch(_policy: &mut Policy, _target_freq: u32) -> u32 {
}
/// Driver's `adjust_perf` callback.
- fn adjust_perf(_policy: &mut Policy, _min_perf: usize, _target_perf: usize, _capacity: usize) {
+ fn adjust_perf(
+ _policy: &mut Policy,
+ _min_perf: usize,
+ _target_perf: usize,
+ _max_perf: usize,
+ _capacity: usize,
+ ) {
build_error!(VTABLE_DEFAULT_ERROR)
}
@@ -1263,12 +1269,13 @@ impl<T: Driver> Registration<T> {
ptr: *mut bindings::cpufreq_policy,
min_perf: c_ulong,
target_perf: c_ulong,
+ max_perf: c_ulong,
capacity: c_ulong,
) {
// SAFETY: The `ptr` is guaranteed to be valid by the contract with the C code for the
// lifetime of `policy`.
let policy = unsafe { Policy::from_raw_mut(ptr) };
- T::adjust_perf(policy, min_perf, target_perf, capacity);
+ T::adjust_perf(policy, min_perf, target_perf, max_perf, capacity);
}
/// Driver's `get_intermediate` callback.
@@ -0,0 +1,127 @@
diff --git a/drivers/cpufreq/amd-pstate-ut.c b/drivers/cpufreq/amd-pstate-ut.c
--- a/drivers/cpufreq/amd-pstate-ut.c
+++ b/drivers/cpufreq/amd-pstate-ut.c
@@ -560,6 +560,11 @@ static int amd_pstate_ut_check_freq_attrs(u32 index)
static int __init amd_pstate_ut_init(void)
{
u32 i = 0, arr_size = ARRAY_SIZE(amd_pstate_ut_cases);
+ enum amd_pstate_mode mode = amd_pstate_get_status();
+
+ /* don't test if no running amd-pstate driver */
+ if (mode == AMD_PSTATE_UNDEFINED || mode == AMD_PSTATE_DISABLE)
+ return -EOPNOTSUPP;
for (i = 0; i < arr_size; i++) {
int ret;
diff --git a/drivers/cpufreq/amd-pstate.c b/drivers/cpufreq/amd-pstate.c
--- a/drivers/cpufreq/amd-pstate.c
+++ b/drivers/cpufreq/amd-pstate.c
@@ -199,7 +199,7 @@ static inline int get_mode_idx_from_str(const char *str, size_t size)
static DEFINE_MUTEX(amd_pstate_driver_lock);
-static u8 msr_get_epp(struct amd_cpudata *cpudata)
+static int msr_get_epp(struct amd_cpudata *cpudata)
{
u64 value;
int ret;
@@ -215,12 +215,12 @@ static u8 msr_get_epp(struct amd_cpudata *cpudata)
DEFINE_STATIC_CALL(amd_pstate_get_epp, msr_get_epp);
-static inline s16 amd_pstate_get_epp(struct amd_cpudata *cpudata)
+static inline int amd_pstate_get_epp(struct amd_cpudata *cpudata)
{
return static_call(amd_pstate_get_epp)(cpudata);
}
-static u8 shmem_get_epp(struct amd_cpudata *cpudata)
+static int shmem_get_epp(struct amd_cpudata *cpudata)
{
u64 epp;
int ret;
@@ -526,9 +526,6 @@ static int shmem_init_perf(struct amd_cpudata *cpudata)
WRITE_ONCE(cpudata->perf, perf);
WRITE_ONCE(cpudata->prefcore_ranking, cppc_perf.highest_perf);
- if (cppc_state == AMD_PSTATE_ACTIVE)
- return 0;
-
ret = cppc_get_auto_sel(cpudata->cpu, &auto_sel);
if (ret) {
pr_warn("failed to get auto_sel, ret: %d\n", ret);
@@ -1174,6 +1171,9 @@ static int amd_pstate_power_supply_notifier(struct notifier_block *nb,
if (cpudata->current_profile != PLATFORM_PROFILE_BALANCED)
return 0;
+ if (!policy)
+ return NOTIFY_OK;
+
epp = amd_pstate_get_balanced_epp(policy);
ret = amd_pstate_set_epp(policy, epp);
@@ -1209,6 +1209,9 @@ static int amd_pstate_profile_set(struct device *dev,
struct cpufreq_policy *policy __free(put_cpufreq_policy) = cpufreq_cpu_get(cpudata->cpu);
int ret;
+ if (!policy)
+ return -ENODEV;
+
switch (profile) {
case PLATFORM_PROFILE_LOW_POWER:
ret = amd_pstate_set_epp(policy, AMD_CPPC_EPP_POWERSAVE);
@@ -1877,6 +1880,7 @@ static int amd_pstate_epp_cpu_init(struct cpufreq_policy *policy)
struct amd_cpudata *cpudata;
union perf_cached perf;
struct device *dev;
+ int default_epp;
int ret;
/*
@@ -1925,6 +1929,13 @@ static int amd_pstate_epp_cpu_init(struct cpufreq_policy *policy)
policy->boost_supported = READ_ONCE(cpudata->boost_supported);
+ /* Fetch the firmware programmed default EPP value */
+ default_epp = amd_pstate_get_epp(cpudata);
+ if (default_epp < 0) {
+ ret = default_epp;
+ goto free_cpudata1;
+ }
+
/*
* Set the policy to provide a valid fallback value in case
* the default cpufreq governor is neither powersave nor performance.
@@ -1932,7 +1943,7 @@ static int amd_pstate_epp_cpu_init(struct cpufreq_policy *policy)
if (amd_pstate_acpi_pm_profile_server() ||
amd_pstate_acpi_pm_profile_undefined()) {
policy->policy = CPUFREQ_POLICY_PERFORMANCE;
- cpudata->epp_default_ac = cpudata->epp_default_dc = amd_pstate_get_epp(cpudata);
+ cpudata->epp_default_ac = cpudata->epp_default_dc = default_epp;
cpudata->current_profile = PLATFORM_PROFILE_PERFORMANCE;
} else {
policy->policy = CPUFREQ_POLICY_POWERSAVE;
diff --git a/drivers/cpufreq/amd-pstate.h b/drivers/cpufreq/amd-pstate.h
--- a/drivers/cpufreq/amd-pstate.h
+++ b/drivers/cpufreq/amd-pstate.h
@@ -32,6 +32,7 @@
* @min_limit_perf: Cached value of the performance corresponding to policy->min
* @max_limit_perf: Cached value of the performance corresponding to policy->max
* @bios_min_perf: Cached perf value corresponding to the "Requested CPU Min Frequency" BIOS option
+ * @val: Raw 64-bit value for atomic access via READ_ONCE()/WRITE_ONCE()
*/
union perf_cached {
struct {
@@ -89,7 +90,12 @@ struct amd_aperf_mperf {
* @epp_default_ac: Default EPP value for AC power source
* @epp_default_dc: Default EPP value for DC power source
* @dynamic_epp: Whether dynamic EPP is enabled
+ * @raw_epp: Whether the last EPP write was a raw numeric value rather than a
+ * named preference
* @power_nb: Notifier block for power events
+ * @current_profile: Currently selected platform profile option
+ * @ppdev: Device registered with the platform profile handler
+ * @profile_name: Name under which @ppdev is registered
*
* The amd_cpudata is key private data for each CPU thread in AMD P-State, and
* represents all the attributes and goals that AMD P-State requests at runtime.
@@ -0,0 +1,18 @@
diff --git a/drivers/cpufreq/amd-pstate.c b/drivers/cpufreq/amd-pstate.c
--- a/drivers/cpufreq/amd-pstate.c
+++ b/drivers/cpufreq/amd-pstate.c
@@ -1929,12 +1929,13 @@ static int amd_pstate_epp_cpu_init(struct cpufreq_policy *policy)
policy->boost_supported = READ_ONCE(cpudata->boost_supported);
- /* Fetch the firmware programmed default EPP value */
+ /* Cache the firmware programmed EPP */
default_epp = amd_pstate_get_epp(cpudata);
if (default_epp < 0) {
ret = default_epp;
goto free_cpudata1;
}
+ FIELD_MODIFY(AMD_CPPC_EPP_PERF_MASK, &cpudata->cppc_req_cached, default_epp);
/*
* Set the policy to provide a valid fallback value in case
@@ -0,0 +1,396 @@
diff --git a/mm/zsmalloc.c b/mm/zsmalloc.c
--- a/mm/zsmalloc.c
+++ b/mm/zsmalloc.c
@@ -21,6 +21,10 @@
* pool->lock
* class->lock
* zspage->lock
+ *
+ * When ZS_OBJ_CLASS_BITS > 0, zs_free() skips pool->lock; it picks
+ * the size_class from obj's encoded class_idx and serializes against
+ * page migration via class->lock.
*/
#include <linux/module.h>
@@ -67,8 +71,8 @@
#define MAX_POSSIBLE_PHYSMEM_BITS MAX_PHYSMEM_BITS
#else
/*
- * If this definition of MAX_PHYSMEM_BITS is used, OBJ_INDEX_BITS will just
- * be PAGE_SHIFT
+ * If this definition of MAX_PHYSMEM_BITS is used, ZS_OBJ_PFN_SHIFT will
+ * just be PAGE_SHIFT
*/
#define MAX_POSSIBLE_PHYSMEM_BITS BITS_PER_LONG
#endif
@@ -88,8 +92,23 @@
#define OBJ_TAG_BITS 1
#define OBJ_TAG_MASK OBJ_ALLOCATED_TAG
-#define OBJ_INDEX_BITS (BITS_PER_LONG - _PFN_BITS)
-#define OBJ_INDEX_MASK ((_AC(1, UL) << OBJ_INDEX_BITS) - 1)
+/*
+ * obj is encoded as [PFN | class_idx | obj_idx] within an unsigned long:
+ *
+ * |<-- _PFN_BITS -->|<-- ZS_OBJ_CLASS_BITS -->|<-- ZS_OBJ_IDX_BITS -->|
+ * +-----------------+-------------------------+-----------------------+
+ * | PFN | class_idx | obj_idx |
+ * +-----------------+-------------------------+-----------------------+
+ * MSB ^ LSB
+ * |
+ * +-- ZS_OBJ_PFN_SHIFT
+ *
+ * Encoding class_idx into obj lets zs_free() locate the size_class
+ * without holding pool->lock; class_idx is invariant across page
+ * migration (only PFN changes), so a lockless read of the obj value
+ * always yields a valid class_idx.
+ */
+#define ZS_OBJ_PFN_SHIFT (BITS_PER_LONG - _PFN_BITS)
#define HUGE_BITS 1
#define FULLNESS_BITS 4
@@ -98,9 +117,61 @@
#define ZS_MAX_PAGES_PER_ZSPAGE (_AC(CONFIG_ZSMALLOC_CHAIN_SIZE, UL))
+/*
+ * Bits to index a page within a zspage = ceil(log2(ZS_MAX_PAGES_PER_ZSPAGE)).
+ * Computed at preprocessor time, for use in #if below. Kconfig
+ * restricts ZSMALLOC_CHAIN_SIZE to [4, 16].
+ */
+#if ZS_MAX_PAGES_PER_ZSPAGE <= 4
+#define ZS_PAGES_PER_ZSPAGE_BITS 2
+#elif ZS_MAX_PAGES_PER_ZSPAGE <= 8
+#define ZS_PAGES_PER_ZSPAGE_BITS 3
+#elif ZS_MAX_PAGES_PER_ZSPAGE <= 16
+#define ZS_PAGES_PER_ZSPAGE_BITS 4
+#else
+#error "ZSMALLOC_CHAIN_SIZE out of expected range [4,16]"
+#endif
+
+/*
+ * Bits to index an object within a single PAGE_SIZE at the smallest
+ * possible object size: log2(PAGE_SIZE / 32) = PAGE_SHIFT - 5.
+ * 32 is the hard floor of ZS_MIN_ALLOC_SIZE.
+ */
+#define ZS_OBJS_PER_PAGE_BITS (PAGE_SHIFT - 5)
+
+/*
+ * Bits to index any object in the densest possible zspage. Below this,
+ * ZS_MIN_ALLOC_SIZE is auto-raised by the MAX(32, ...) formula -- still
+ * correct, but objects are coarser.
+ */
+#define ZS_OBJS_PER_ZSPAGE_BITS \
+ (ZS_PAGES_PER_ZSPAGE_BITS + ZS_OBJS_PER_PAGE_BITS)
+
+/*
+ * Encode class_idx only when obj has spare bits; otherwise
+ * ZS_OBJ_CLASS_BITS folds to 0 (32-bit, or 64-bit UML/fallback).
+ */
+#if BITS_PER_LONG >= 64 && \
+ ZS_OBJ_PFN_SHIFT >= (CLASS_BITS + 1) + ZS_OBJS_PER_ZSPAGE_BITS
+#define ZS_OBJ_CLASS_BITS (CLASS_BITS + 1)
+#else
+#define ZS_OBJ_CLASS_BITS 0
+#endif
+#define ZS_OBJ_CLASS_MASK ((_AC(1, UL) << ZS_OBJ_CLASS_BITS) - 1)
+
+#define ZS_OBJ_IDX_BITS (ZS_OBJ_PFN_SHIFT - ZS_OBJ_CLASS_BITS)
+#define ZS_OBJ_IDX_MASK ((_AC(1, UL) << ZS_OBJ_IDX_BITS) - 1)
+
+/*
+ * Belt-and-suspenders: the #if above already guarantees this when
+ * class_idx is enabled. Catches future tweaks that bypass it.
+ */
+static_assert(ZS_OBJ_IDX_BITS >= ZS_PAGES_PER_ZSPAGE_BITS,
+ "zsmalloc: ZS_MIN_ALLOC_SIZE would exceed ZS_MAX_ALLOC_SIZE");
+
/* ZS_MIN_ALLOC_SIZE must be multiple of ZS_ALIGN */
#define ZS_MIN_ALLOC_SIZE \
- MAX(32, (ZS_MAX_PAGES_PER_ZSPAGE << PAGE_SHIFT >> OBJ_INDEX_BITS))
+ MAX(32, (ZS_MAX_PAGES_PER_ZSPAGE << PAGE_SHIFT >> ZS_OBJ_IDX_BITS))
/* each chunk includes extra space to keep handle */
#define ZS_MAX_ALLOC_SIZE PAGE_SIZE
@@ -396,10 +467,13 @@ static void cache_free_zspage(struct zspage *zspage)
kmem_cache_free(zspage_cachep, zspage);
}
-/* class->lock(which owns the handle) synchronizes races */
+/*
+ * Pairs with READ_ONCE() in handle_to_obj(): zs_free() may read the
+ * handle locklessly, so prevent store tearing here.
+ */
static void record_obj(unsigned long handle, unsigned long obj)
{
- *(unsigned long *)handle = obj;
+ WRITE_ONCE(*(unsigned long *)handle, obj);
}
static inline bool __maybe_unused is_first_zpdesc(struct zpdesc *zpdesc)
@@ -725,33 +799,36 @@ static struct zpdesc *get_next_zpdesc(struct zpdesc *zpdesc)
static void obj_to_location(unsigned long obj, struct zpdesc **zpdesc,
unsigned int *obj_idx)
{
- *zpdesc = pfn_zpdesc(obj >> OBJ_INDEX_BITS);
- *obj_idx = (obj & OBJ_INDEX_MASK);
+ *zpdesc = pfn_zpdesc(obj >> ZS_OBJ_PFN_SHIFT);
+ *obj_idx = (obj & ZS_OBJ_IDX_MASK);
}
static void obj_to_zpdesc(unsigned long obj, struct zpdesc **zpdesc)
{
- *zpdesc = pfn_zpdesc(obj >> OBJ_INDEX_BITS);
+ *zpdesc = pfn_zpdesc(obj >> ZS_OBJ_PFN_SHIFT);
}
/**
- * location_to_obj - get obj value encoded from (<zpdesc>, <obj_idx>)
+ * location_to_obj - encode (<zpdesc>, <obj_idx>, <class_idx>) into obj value
* @zpdesc: zpdesc object resides in zspage
* @obj_idx: object index
+ * @class_idx: size class index; ignored when ZS_OBJ_CLASS_BITS == 0
*/
-static unsigned long location_to_obj(struct zpdesc *zpdesc, unsigned int obj_idx)
+static unsigned long location_to_obj(struct zpdesc *zpdesc, unsigned int obj_idx,
+ unsigned int class_idx)
{
unsigned long obj;
- obj = zpdesc_pfn(zpdesc) << OBJ_INDEX_BITS;
- obj |= obj_idx & OBJ_INDEX_MASK;
+ obj = zpdesc_pfn(zpdesc) << ZS_OBJ_PFN_SHIFT;
+ obj |= (unsigned long)(class_idx & ZS_OBJ_CLASS_MASK) << ZS_OBJ_IDX_BITS;
+ obj |= obj_idx & ZS_OBJ_IDX_MASK;
return obj;
}
static unsigned long handle_to_obj(unsigned long handle)
{
- return *(unsigned long *)handle;
+ return READ_ONCE(*(unsigned long *)handle);
}
static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj,
@@ -805,13 +882,26 @@ static int trylock_zspage(struct zspage *zspage)
return 0;
}
-static void __free_zspage(struct zs_pool *pool, struct size_class *class,
- struct zspage *zspage)
+/*
+ * Three free helpers, kept apart here:
+ *
+ * __free_zspage_lockless(): bare core; walks zpdescs and returns pages
+ * to the buddy allocator. Caller owns all zpdesc locks and has
+ * removed the zspage from its class list. Used by zs_free() outside
+ * class->lock so the buddy-side work does not stall the class.
+ *
+ * __free_zspage(): __free_zspage_lockless() + per-class accounting,
+ * under class->lock. Used by async_free_zspage(), the worker for
+ * zspages whose trylock_zspage() failed.
+ *
+ * free_zspage(): full wrapper - trylock zpdescs, remove from class
+ * list, call __free_zspage(); kicks deferred free on contention.
+ * Used by compaction.
+ */
+static inline void __free_zspage_lockless(struct zspage *zspage)
{
struct zpdesc *zpdesc, *next;
- assert_spin_locked(&class->lock);
-
VM_BUG_ON(get_zspage_inuse(zspage));
VM_BUG_ON(zspage->fullness != ZS_INUSE_RATIO_0);
@@ -827,7 +917,13 @@ static void __free_zspage(struct zs_pool *pool, struct size_class *class,
} while (zpdesc != NULL);
cache_free_zspage(zspage);
+}
+static void __free_zspage(struct zs_pool *pool, struct size_class *class,
+ struct zspage *zspage)
+{
+ assert_spin_locked(&class->lock);
+ __free_zspage_lockless(zspage);
class_stat_sub(class, ZS_OBJS_ALLOCATED, class->objs_per_zspage);
atomic_long_sub(class->pages_per_zspage, &pool->pages_allocated);
}
@@ -1280,7 +1376,7 @@ static unsigned long obj_malloc(struct zs_pool *pool,
kunmap_local(vaddr);
mod_zspage_inuse(zspage, 1);
- obj = location_to_obj(m_zpdesc, obj);
+ obj = location_to_obj(m_zpdesc, obj, zspage->class);
record_obj(handle, obj);
return obj;
@@ -1383,37 +1479,97 @@ static void obj_free(int class_size, unsigned long obj)
mod_zspage_inuse(zspage, -1);
}
+#if (ZS_OBJ_CLASS_BITS > 0) || defined(CONFIG_COMPACTION)
+/* Folds to 0 when ZS_OBJ_CLASS_BITS == 0; no ifdef needed at callers. */
+static unsigned int obj_to_class_idx(unsigned long obj)
+{
+ return (obj >> ZS_OBJ_IDX_BITS) & ZS_OBJ_CLASS_MASK;
+}
+#endif
+
+/*
+ * Resolve @handle to its zspage / size_class and acquire class->lock.
+ *
+ * When class_idx is encoded in obj (ZS_OBJ_CLASS_BITS > 0), it is
+ * invariant under page migration, so the handle can be read locklessly
+ * to pick the size_class. Once class->lock is held migration is
+ * blocked and the handle is re-read to obtain a stable PFN.
+ *
+ * Otherwise (32-bit, or 64-bit fallback paths like UML where the
+ * encoding is disabled), fall back to pool->lock for the lookup.
+ */
+#if ZS_OBJ_CLASS_BITS > 0
+static inline void obj_class_get_and_lock(struct zs_pool *pool, unsigned long handle,
+ unsigned long *objp, struct zspage **zspagep,
+ struct size_class **classp)
+ __acquires(&(*classp)->lock)
+{
+ struct zpdesc *f_zpdesc;
+ unsigned long obj;
+
+ obj = handle_to_obj(handle);
+ *classp = pool->size_class[obj_to_class_idx(obj)];
+ spin_lock(&(*classp)->lock);
+ /* Re-read under class->lock: PFN is now stable vs migration. */
+ obj = handle_to_obj(handle);
+ obj_to_zpdesc(obj, &f_zpdesc);
+ *zspagep = get_zspage(f_zpdesc);
+ *objp = obj;
+}
+#else
+static inline void obj_class_get_and_lock(struct zs_pool *pool, unsigned long handle,
+ unsigned long *objp, struct zspage **zspagep,
+ struct size_class **classp)
+ __acquires(&(*classp)->lock)
+{
+ struct zpdesc *f_zpdesc;
+ unsigned long obj;
+
+ read_lock(&pool->lock);
+ obj = handle_to_obj(handle);
+ obj_to_zpdesc(obj, &f_zpdesc);
+ *zspagep = get_zspage(f_zpdesc);
+ *classp = zspage_class(pool, *zspagep);
+ spin_lock(&(*classp)->lock);
+ read_unlock(&pool->lock);
+ *objp = obj;
+}
+#endif
+
void zs_free(struct zs_pool *pool, unsigned long handle)
{
struct zspage *zspage;
- struct zpdesc *f_zpdesc;
unsigned long obj;
struct size_class *class;
int fullness;
+ struct zspage *zspage_to_free = NULL;
if (IS_ERR_OR_NULL((void *)handle))
return;
- /*
- * The pool->lock protects the race with zpage's migration
- * so it's safe to get the page from handle.
- */
- read_lock(&pool->lock);
- obj = handle_to_obj(handle);
- obj_to_zpdesc(obj, &f_zpdesc);
- zspage = get_zspage(f_zpdesc);
- class = zspage_class(pool, zspage);
- spin_lock(&class->lock);
- read_unlock(&pool->lock);
+ obj_class_get_and_lock(pool, handle, &obj, &zspage, &class);
class_stat_sub(class, ZS_OBJS_INUSE, 1);
obj_free(class->size, obj);
fullness = fix_fullness_group(class, zspage);
- if (fullness == ZS_INUSE_RATIO_0)
- free_zspage(pool, class, zspage);
+ if (fullness == ZS_INUSE_RATIO_0) {
+ if (trylock_zspage(zspage)) {
+ remove_zspage(class, zspage);
+ class_stat_sub(class, ZS_OBJS_ALLOCATED,
+ class->objs_per_zspage);
+ zspage_to_free = zspage;
+ } else {
+ kick_deferred_free(pool);
+ }
+ }
spin_unlock(&class->lock);
+
+ if (zspage_to_free) {
+ __free_zspage_lockless(zspage_to_free);
+ atomic_long_sub(class->pages_per_zspage, &pool->pages_allocated);
+ }
cache_free_handle(handle);
}
EXPORT_SYMBOL_GPL(zs_free);
@@ -1646,9 +1802,6 @@ static void lock_zspage(struct zspage *zspage)
}
zspage_read_unlock(zspage);
}
-#endif /* CONFIG_COMPACTION */
-
-#ifdef CONFIG_COMPACTION
static void replace_sub_page(struct size_class *class, struct zspage *zspage,
struct zpdesc *newzpdesc, struct zpdesc *oldzpdesc)
@@ -1715,8 +1868,8 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
pool = zspage->pool;
/*
- * The pool migrate_lock protects the race between zpage migration
- * and zs_free.
+ * The pool migrate_lock protects against races between zpage migration
+ * and zs_free(), but only when ZS_OBJ_CLASS_BITS does not apply.
*/
write_lock(&pool->lock);
class = zspage_class(pool, zspage);
@@ -1764,7 +1917,8 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
old_obj = handle_to_obj(handle);
obj_to_location(old_obj, &dummy, &obj_idx);
- new_obj = (unsigned long)location_to_obj(newzpdesc, obj_idx);
+ new_obj = location_to_obj(newzpdesc, obj_idx,
+ obj_to_class_idx(old_obj));
record_obj(handle, new_obj);
}
}
@@ -1775,9 +1929,9 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
* Since we complete the data copy and set up new zspage structure,
* it's okay to release migration_lock.
*/
- write_unlock(&pool->lock);
- spin_unlock(&class->lock);
zspage_write_unlock(zspage);
+ spin_unlock(&class->lock);
+ write_unlock(&pool->lock);
zpdesc_get(newzpdesc);
if (zpdesc_zone(newzpdesc) != zpdesc_zone(zpdesc)) {
@@ -1894,8 +2048,9 @@ static unsigned long __zs_compact(struct zs_pool *pool,
unsigned long pages_freed = 0;
/*
- * protect the race between zpage migration and zs_free
- * as well as zpage allocation/free
+ * Protect against races between zpage migration and zs_free()
+ * (only when ZS_OBJ_CLASS_BITS does not apply), as well as
+ * zpage allocation and free.
*/
write_lock(&pool->lock);
spin_lock(&class->lock);
@@ -0,0 +1,313 @@
diff --git a/include/linux/rmap.h b/include/linux/rmap.h
--- a/include/linux/rmap.h
+++ b/include/linux/rmap.h
@@ -843,7 +843,7 @@ static inline int folio_try_share_anon_rmap_pmd(struct folio *folio,
* Called from mm/vmscan.c to handle paging out
*/
int folio_referenced(struct folio *, int is_locked,
- struct mem_cgroup *memcg, vm_flags_t *vm_flags);
+ struct mem_cgroup *memcg, vma_flags_t *vma_flags);
void try_to_migrate(struct folio *folio, enum ttu_flags flags);
void try_to_unmap(struct folio *, enum ttu_flags flags);
@@ -975,10 +975,9 @@ struct anon_vma *folio_lock_anon_vma_read(const struct folio *folio,
#define anon_vma_prepare(vma) (0)
static inline int folio_referenced(struct folio *folio, int is_locked,
- struct mem_cgroup *memcg,
- vm_flags_t *vm_flags)
+ struct mem_cgroup *memcg, vma_flags_t *vma_flags)
{
- *vm_flags = 0;
+ vma_flags_clear_all(vma_flags);
return 0;
}
diff --git a/mm/rmap.c b/mm/rmap.c
--- a/mm/rmap.c
+++ b/mm/rmap.c
@@ -907,7 +907,7 @@ pmd_t *mm_find_pmd(struct mm_struct *mm, unsigned long address)
struct folio_referenced_arg {
int mapcount;
int referenced;
- vm_flags_t vm_flags;
+ vma_flags_t vma_flags;
struct mem_cgroup *memcg;
};
@@ -926,7 +926,7 @@ static bool folio_referenced_one(struct folio *folio,
address = pvmw.address;
nr = 1;
- if (vma->vm_flags & VM_LOCKED) {
+ if (vma_test(vma, VMA_LOCKED_BIT)) {
ptes++;
pra->mapcount--;
@@ -947,7 +947,7 @@ static bool folio_referenced_one(struct folio *folio,
/* Restore the mlock which got missed */
mlock_vma_folio(folio, vma);
page_vma_mapped_walk_done(&pvmw);
- pra->vm_flags |= VM_LOCKED;
+ vma_flags_set(&pra->vma_flags, VMA_LOCKED_BIT);
return false; /* To break the loop */
}
@@ -1015,8 +1015,11 @@ static bool folio_referenced_one(struct folio *folio,
referenced++;
if (referenced) {
+ vma_flags_t vma_flags = vma->flags;
+
pra->referenced++;
- pra->vm_flags |= vma->vm_flags & ~VM_LOCKED;
+ vma_flags_clear(&vma_flags, VMA_LOCKED_BIT);
+ vma_flags_set_mask(&pra->vma_flags, vma_flags);
}
if (!pra->mapcount)
@@ -1054,7 +1057,7 @@ static bool invalid_folio_referenced_vma(struct vm_area_struct *vma, void *arg)
* @folio: The folio to test.
* @is_locked: Caller holds lock on the folio.
* @memcg: target memory cgroup
- * @vm_flags: A combination of all the vma->vm_flags which referenced the folio.
+ * @vma_flags: A combination of all the vma->flags which referenced the folio.
*
* Quick test_and_clear_referenced for all mappings of a folio,
*
@@ -1062,7 +1065,7 @@ static bool invalid_folio_referenced_vma(struct vm_area_struct *vma, void *arg)
* the function bailed out due to rmap lock contention.
*/
int folio_referenced(struct folio *folio, int is_locked,
- struct mem_cgroup *memcg, vm_flags_t *vm_flags)
+ struct mem_cgroup *memcg, vma_flags_t *vma_flags)
{
bool we_locked = false;
struct folio_referenced_arg pra = {
@@ -1078,7 +1081,7 @@ int folio_referenced(struct folio *folio, int is_locked,
};
VM_WARN_ON_ONCE_FOLIO(folio_is_zone_device(folio), folio);
- *vm_flags = 0;
+ vma_flags_clear_all(vma_flags);
if (!pra.mapcount)
return 0;
@@ -1092,7 +1095,7 @@ int folio_referenced(struct folio *folio, int is_locked,
}
rmap_walk(folio, &rwc);
- *vm_flags = pra.vm_flags;
+ vma_flags_set_mask(vma_flags, pra.vma_flags);
if (we_locked)
folio_unlock(folio);
diff --git a/mm/vmscan.c b/mm/vmscan.c
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -262,6 +262,12 @@ static bool writeback_throttling_sane(struct scan_control *sc)
}
#endif
+static inline bool is_exec_file_folio(const struct folio *folio,
+ const vma_flags_t *vma_flags)
+{
+ return vma_flags_test(vma_flags, VMA_EXEC_BIT) && folio_is_file_lru(folio);
+}
+
static void set_task_reclaim_state(struct task_struct *task,
struct reclaim_state *rs)
{
@@ -830,10 +836,16 @@ enum folio_references {
* with PG_active set. In contrast, the aging (page table walk) path uses
* folio_update_gen().
*/
-static bool lru_gen_set_refs(struct folio *folio)
+static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags)
{
/* see the comment on LRU_REFS_FLAGS */
if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) {
+ /* Activate file-backed executable folios after first usage. */
+ if (is_exec_file_folio(folio, vma_flags)) {
+ set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset));
+ return true;
+ }
+
set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
return false;
}
@@ -846,7 +858,7 @@ static bool lru_gen_set_refs(struct folio *folio)
return true;
}
#else
-static bool lru_gen_set_refs(struct folio *folio)
+static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags)
{
return false;
}
@@ -856,16 +868,16 @@ static enum folio_references folio_check_references(struct folio *folio,
struct scan_control *sc)
{
int referenced_ptes, referenced_folio;
- vm_flags_t vm_flags;
+ vma_flags_t vma_flags;
referenced_ptes = folio_referenced(folio, 1, sc->target_mem_cgroup,
- &vm_flags);
+ &vma_flags);
/*
* The supposedly reclaimable folio was found to be in a VM_LOCKED vma.
* Let the folio, now marked Mlocked, be moved to the unevictable list.
*/
- if (vm_flags & VM_LOCKED)
+ if (vma_flags_test(&vma_flags, VMA_LOCKED_BIT))
return FOLIOREF_ACTIVATE;
/*
@@ -881,7 +893,7 @@ static enum folio_references folio_check_references(struct folio *folio,
if (!referenced_ptes)
return FOLIOREF_RECLAIM;
- return lru_gen_set_refs(folio) ? FOLIOREF_ACTIVATE : FOLIOREF_KEEP;
+ return lru_gen_set_refs(folio, &vma_flags) ? FOLIOREF_ACTIVATE : FOLIOREF_KEEP;
}
referenced_folio = folio_test_clear_referenced(folio);
@@ -909,7 +921,7 @@ static enum folio_references folio_check_references(struct folio *folio,
/*
* Activate file-backed executable folios after first usage.
*/
- if ((vm_flags & VM_EXEC) && folio_is_file_lru(folio))
+ if (is_exec_file_folio(folio, &vma_flags))
return FOLIOREF_ACTIVATE;
return FOLIOREF_KEEP;
@@ -2067,7 +2079,7 @@ static void shrink_active_list(unsigned long nr_to_scan,
{
unsigned long nr_taken;
unsigned long nr_scanned;
- vm_flags_t vm_flags;
+ vma_flags_t vma_flags;
LIST_HEAD(l_hold); /* The folios which were snipped off */
LIST_HEAD(l_active);
LIST_HEAD(l_inactive);
@@ -2111,7 +2123,7 @@ static void shrink_active_list(unsigned long nr_to_scan,
/* Referenced or rmap lock contention: rotate */
if (folio_referenced(folio, 0, sc->target_mem_cgroup,
- &vm_flags) != 0) {
+ &vma_flags) != 0) {
/*
* Identify referenced, file-backed active folios and
* give them one more trip around the active list. So
@@ -2121,7 +2133,7 @@ static void shrink_active_list(unsigned long nr_to_scan,
* IO, plus JVM can create lots of anon VM_EXEC folios,
* so we ignore them here.
*/
- if ((vm_flags & VM_EXEC) && folio_is_file_lru(folio)) {
+ if (is_exec_file_folio(folio, &vma_flags)) {
nr_rotated += folio_nr_pages(folio);
list_add(&folio->lru, &l_active);
continue;
@@ -3190,14 +3202,19 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv)
******************************************************************************/
/* promote pages accessed through page tables */
-static int folio_update_gen(struct folio *folio, int gen)
+static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma_flags)
{
unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f);
VM_WARN_ON_ONCE(gen >= MAX_NR_GENS);
- /* see the comment on LRU_REFS_FLAGS */
- if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) {
+ /*
+ * See the comment on LRU_REFS_FLAGS, and activate file-backed
+ * executable folios after first usage to avoid typical IO
+ * thrashing from reclaiming.
+ */
+ if (!folio_test_referenced(folio) && !folio_test_workingset(folio) &&
+ !is_exec_file_folio(folio, vma_flags)) {
set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
return -1;
}
@@ -3430,8 +3447,8 @@ static bool suitable_to_scan(int total, int young)
return young * n >= total;
}
-static void walk_update_folio(struct lru_gen_mm_walk *walk, struct folio *folio,
- int new_gen, bool dirty)
+static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma,
+ struct folio *folio, int new_gen, bool dirty)
{
int old_gen;
@@ -3444,10 +3461,10 @@ static void walk_update_folio(struct lru_gen_mm_walk *walk, struct folio *folio,
folio_mark_dirty(folio);
if (walk) {
- old_gen = folio_update_gen(folio, new_gen);
+ old_gen = folio_update_gen(folio, new_gen, &vma->flags);
if (old_gen >= 0 && old_gen != new_gen)
update_batch_size(walk, folio, old_gen, new_gen);
- } else if (lru_gen_set_refs(folio)) {
+ } else if (lru_gen_set_refs(folio, &vma->flags)) {
old_gen = folio_lru_gen(folio);
if (old_gen >= 0 && old_gen != new_gen)
folio_activate(folio);
@@ -3520,7 +3537,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
continue;
if (last != folio) {
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, args->vma, last, gen, dirty);
last = folio;
dirty = false;
@@ -3533,7 +3550,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
walk->mm_stats[MM_LEAF_YOUNG] += nr;
}
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, args->vma, last, gen, dirty);
last = NULL;
if (i < PTRS_PER_PTE && get_next_vma(PMD_MASK, PAGE_SIZE, args, &start, &end))
@@ -3611,7 +3628,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
goto next;
if (last != folio) {
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
last = folio;
dirty = false;
@@ -3625,7 +3642,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
i = i > MIN_LRU_BATCH ? 0 : find_next_bit(bitmap, MIN_LRU_BATCH, i) + 1;
} while (i <= MIN_LRU_BATCH);
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
lazy_mmu_mode_disable();
spin_unlock(ptl);
@@ -4260,7 +4277,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
continue;
if (last != folio) {
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
last = folio;
dirty = false;
@@ -4272,7 +4289,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
young += nr;
}
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
lazy_mmu_mode_disable();
@@ -0,0 +1,161 @@
diff --git a/mm/ksm.c b/mm/ksm.c
--- a/mm/ksm.c
+++ b/mm/ksm.c
@@ -195,22 +195,28 @@ struct ksm_stable_node {
* @node: rb node of this rmap_item in the unstable tree
* @head: pointer to stable_node heading this list in the stable tree
* @hlist: link into hlist of rmap_items hanging off that stable_node
- * @age: number of scan iterations since creation
- * @remaining_skips: how many scans to skip
+ * @age: number of scan iterations since creation (unstable node)
+ * @remaining_skips: how many scans to skip (unstable node)
+ * @linear_page_index: the original page's index before merged by KSM (stable node)
*/
struct ksm_rmap_item {
struct ksm_rmap_item *rmap_list;
union {
- struct anon_vma *anon_vma; /* when stable */
+ struct anon_vma *anon_vma; /* for reverse mapping, when stable */
#ifdef CONFIG_NUMA
int nid; /* when node of unstable tree */
#endif
};
struct mm_struct *mm;
unsigned long address; /* + low bits used for flags below */
- unsigned int oldchecksum; /* when unstable */
- rmap_age_t age;
- rmap_age_t remaining_skips;
+ union {
+ struct {
+ unsigned int oldchecksum;
+ rmap_age_t age;
+ rmap_age_t remaining_skips;
+ }; /* when unstable */
+ unsigned long linear_page_index; /* for reverse mapping, when stable */
+ };
union {
struct rb_node node; /* when node of unstable tree */
struct { /* when listed from stable tree */
@@ -776,6 +782,11 @@ static struct vm_area_struct *find_mergeable_vma(struct mm_struct *mm,
return vma;
}
+/*
+ * break_cow: actively break COW, replacing the KSM page by a fresh anonymous
+ * page. This is called when rmap_item has not yet become stable, but page
+ * has been merged.
+ */
static void break_cow(struct ksm_rmap_item *rmap_item)
{
struct mm_struct *mm = rmap_item->mm;
@@ -787,6 +798,11 @@ static void break_cow(struct ksm_rmap_item *rmap_item)
* to undo, we also need to drop a reference to the anon_vma.
*/
put_anon_vma(rmap_item->anon_vma);
+ /*
+ * Reset linear_page_index that might overlay age-related
+ * information. (it's still unstable node)
+ */
+ rmap_item->linear_page_index = 0;
mmap_read_lock(mm);
vma = find_mergeable_vma(mm, addr);
@@ -899,6 +915,8 @@ static void remove_node_from_stable_tree(struct ksm_stable_node *stable_node)
VM_BUG_ON(stable_node->rmap_hlist_len <= 0);
stable_node->rmap_hlist_len--;
put_anon_vma(rmap_item->anon_vma);
+ /* Reset linear_page_index that might overlay age-related information. */
+ rmap_item->linear_page_index = 0;
rmap_item->address &= PAGE_MASK;
cond_resched();
}
@@ -1052,6 +1070,8 @@ static void remove_rmap_item_from_tree(struct ksm_rmap_item *rmap_item)
stable_node->rmap_hlist_len--;
put_anon_vma(rmap_item->anon_vma);
+ /* Reset linear_page_index that might overlay age-related information. */
+ rmap_item->linear_page_index = 0;
rmap_item->head = NULL;
rmap_item->address &= PAGE_MASK;
@@ -1598,8 +1618,15 @@ static int try_to_merge_with_ksm_page(struct ksm_rmap_item *rmap_item,
/* Unstable nid is in union with stable anon_vma: remove first */
remove_rmap_item_from_tree(rmap_item);
- /* Must get reference to anon_vma while still holding mmap_lock */
+ /*
+ * We can consider the VMA only while still holding the mmap lock,
+ * so lock, so reference the anon_vma and calculate the linear
+ * page index early, before stable_tree_append(). If anything goes
+ * wrong that prevents the rmap_item from being added to the
+ * stable_tree, break_cow() will clean it up.
+ */
rmap_item->anon_vma = vma->anon_vma;
+ rmap_item->linear_page_index = linear_page_index(vma, rmap_item->address);
get_anon_vma(vma->anon_vma);
out:
mmap_read_unlock(mm);
@@ -2458,6 +2485,13 @@ static bool should_skip_rmap_item(struct folio *folio,
if (folio_test_ksm(folio))
return false;
+ /*
+ * There is no age information in stable-tree nodes. We might end up
+ * here without a KSM page for example after COW.
+ */
+ if (rmap_item->address & STABLE_FLAG)
+ return false;
+
age = rmap_item->age;
if (age != U8_MAX)
rmap_item->age++;
@@ -3173,6 +3207,7 @@ void rmap_walk_ksm(struct folio *folio, struct rmap_walk_control *rwc)
hlist_for_each_entry(rmap_item, &stable_node->hlist, hlist) {
/* Ignore the stable/unstable/sqnr flags */
const unsigned long addr = rmap_item->address & PAGE_MASK;
+ const unsigned long index = rmap_item->linear_page_index;
struct anon_vma *anon_vma = rmap_item->anon_vma;
struct anon_vma_chain *vmac;
struct vm_area_struct *vma;
@@ -3186,8 +3221,18 @@ void rmap_walk_ksm(struct folio *folio, struct rmap_walk_control *rwc)
anon_vma_lock_read(anon_vma);
}
+ /*
+ * Currently, KSM folios are always small folios, so it's
+ * sufficient to search for a single page. We can simply use
+ * the linear_page_index of the original de-duplicate
+ * anonymous page that we remembered in the rmap_item while
+ * de-duplicating. Note that mremap() always de-duplicates KSM
+ * folios: so if there was mremap() in our parent or our child,
+ * we wouldn't have the KSM folio mapped in these processes
+ * anymore.
+ */
anon_vma_interval_tree_foreach(vmac, &anon_vma->rb_root,
- 0, ULONG_MAX) {
+ index, index) {
cond_resched();
vma = vmac->vma;
@@ -3243,17 +3288,17 @@ void collect_procs_ksm(const struct folio *folio, const struct page *page,
rcu_read_lock();
for_each_process(tsk) {
struct anon_vma_chain *vmac;
- unsigned long addr;
+ const unsigned long addr = rmap_item->address & PAGE_MASK;
+ const unsigned long index = rmap_item->linear_page_index;
struct task_struct *t =
task_early_kill(tsk, force_early);
if (!t)
continue;
- anon_vma_interval_tree_foreach(vmac, &av->rb_root, 0,
- ULONG_MAX)
+ anon_vma_interval_tree_foreach(vmac, &av->rb_root, index,
+ index)
{
vma = vmac->vma;
if (vma->vm_mm == t->mm) {
- addr = rmap_item->address & PAGE_MASK;
add_to_kill_ksm(t, page, vma, to_kill,
addr);
}
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,97 @@
diff --git a/lib/zstd/common/entropy_common.c b/lib/zstd/common/entropy_common.c
--- a/lib/zstd/common/entropy_common.c
+++ b/lib/zstd/common/entropy_common.c
@@ -202,6 +202,8 @@ BMI2_TARGET_ATTRIBUTE static size_t FSE_readNCount_body_bmi2(
{
return FSE_readNCount_body(normalizedCounter, maxSVPtr, tableLogPtr, headerBuffer, hbSize);
}
+#else
+#define FSE_readNCount_body_bmi2 FSE_readNCount_body_default
#endif
size_t FSE_readNCount_bmi2(
@@ -323,6 +325,8 @@ static BMI2_TARGET_ATTRIBUTE size_t HUF_readStats_body_bmi2(BYTE* huffWeight, si
{
return HUF_readStats_body(huffWeight, hwSize, rankStats, nbSymbolsPtr, tableLogPtr, src, srcSize, workSpace, wkspSize, 1);
}
+#else
+#define HUF_readStats_body_bmi2 HUF_readStats_body_default
#endif
size_t HUF_readStats_wksp(BYTE* huffWeight, size_t hwSize, U32* rankStats,
diff --git a/lib/zstd/common/fse_decompress.c b/lib/zstd/common/fse_decompress.c
--- a/lib/zstd/common/fse_decompress.c
+++ b/lib/zstd/common/fse_decompress.c
@@ -300,6 +300,8 @@ BMI2_TARGET_ATTRIBUTE static size_t FSE_decompress_wksp_body_bmi2(void* dst, siz
{
return FSE_decompress_wksp_body(dst, dstCapacity, cSrc, cSrcSize, maxLog, workSpace, wkspSize, 1);
}
+#else
+#define FSE_decompress_wksp_body_bmi2 FSE_decompress_wksp_body_default
#endif
size_t FSE_decompress_wksp_bmi2(void* dst, size_t dstCapacity, const void* cSrc, size_t cSrcSize, unsigned maxLog, void* workSpace, size_t wkspSize, int bmi2)
diff --git a/lib/zstd/compress/zstd_compress_sequences.c b/lib/zstd/compress/zstd_compress_sequences.c
--- a/lib/zstd/compress/zstd_compress_sequences.c
+++ b/lib/zstd/compress/zstd_compress_sequences.c
@@ -415,6 +415,10 @@ ZSTD_encodeSequences_bmi2(
sequences, nbSeq, longOffsets);
}
+#else
+
+#define ZSTD_encodeSequences_bmi2 ZSTD_encodeSequences_default
+
#endif
size_t ZSTD_encodeSequences(
diff --git a/lib/zstd/decompress/huf_decompress.c b/lib/zstd/decompress/huf_decompress.c
--- a/lib/zstd/decompress/huf_decompress.c
+++ b/lib/zstd/decompress/huf_decompress.c
@@ -700,6 +700,8 @@ size_t HUF_decompress4X1_usingDTable_internal_bmi2(void* dst, size_t dstSize, vo
size_t cSrcSize, HUF_DTable const* DTable) {
return HUF_decompress4X1_usingDTable_internal_body(dst, dstSize, cSrc, cSrcSize, DTable);
}
+#else
+#define HUF_decompress4X1_usingDTable_internal_bmi2 HUF_decompress4X1_usingDTable_internal_default
#endif
static
@@ -1503,6 +1505,8 @@ size_t HUF_decompress4X2_usingDTable_internal_bmi2(void* dst, size_t dstSize, vo
size_t cSrcSize, HUF_DTable const* DTable) {
return HUF_decompress4X2_usingDTable_internal_body(dst, dstSize, cSrc, cSrcSize, DTable);
}
+#else
+#define HUF_decompress4X2_usingDTable_internal_bmi2 HUF_decompress4X2_usingDTable_internal_default
#endif
static
diff --git a/lib/zstd/decompress/zstd_decompress_block.c b/lib/zstd/decompress/zstd_decompress_block.c
--- a/lib/zstd/decompress/zstd_decompress_block.c
+++ b/lib/zstd/decompress/zstd_decompress_block.c
@@ -622,6 +622,8 @@ BMI2_TARGET_ATTRIBUTE static void ZSTD_buildFSETable_body_bmi2(ZSTD_seqSymbol* d
ZSTD_buildFSETable_body(dt, normalizedCounter, maxSymbolValue,
baseValue, nbAdditionalBits, tableLog, wksp, wkspSize);
}
+#else
+#define ZSTD_buildFSETable_body_bmi2 ZSTD_buildFSETable_body_default
#endif
void ZSTD_buildFSETable(ZSTD_seqSymbol* dt,
@@ -1934,6 +1936,16 @@ ZSTD_decompressSequencesLong_bmi2(ZSTD_DCtx* dctx,
}
#endif /* ZSTD_FORCE_DECOMPRESS_SEQUENCES_SHORT */
+#else
+
+#ifndef ZSTD_FORCE_DECOMPRESS_SEQUENCES_LONG
+#define ZSTD_decompressSequences_bmi2 ZSTD_decompressSequences_default
+#define ZSTD_decompressSequencesSplitLitBuffer_bmi2 ZSTD_decompressSequencesSplitLitBuffer_default
+#endif
+#ifndef ZSTD_FORCE_DECOMPRESS_SEQUENCES_SHORT
+#define ZSTD_decompressSequencesLong_bmi2 ZSTD_decompressSequencesLong_default
+#endif
+
#endif /* DYNAMIC_BMI2 */
#ifndef ZSTD_FORCE_DECOMPRESS_SEQUENCES_LONG
@@ -0,0 +1,413 @@
diff --git a/lib/zstd/common/compiler.h b/lib/zstd/common/compiler.h
--- a/lib/zstd/common/compiler.h
+++ b/lib/zstd/common/compiler.h
@@ -14,6 +14,7 @@
#include <linux/types.h>
+#include "zstd_deps.h"
#include "portability_macros.h"
/*-*******************************************************
@@ -96,6 +97,17 @@
*/
#define BMI2_TARGET_ATTRIBUTE TARGET_ATTRIBUTE("lzcnt,bmi,bmi2")
+#if !DYNAMIC_BMI2
+# define ZSTD_USE_BMI2(bmi2) 0
+# define ZSTD_SET_BMI2(state, value) do { } while (0)
+#elif defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+# define ZSTD_USE_BMI2(bmi2) cpu_feature_enabled(X86_FEATURE_BMI2)
+# define ZSTD_SET_BMI2(state, value) do { } while (0)
+#else
+# define ZSTD_USE_BMI2(bmi2) (bmi2)
+# define ZSTD_SET_BMI2(state, value) do { (state) = (value); } while (0)
+#endif
+
/* prefetch
* can be disabled, by declaring NO_PREFETCH build macro */
#if ( (__GNUC__ >= 4) || ( (__GNUC__ == 3) && (__GNUC_MINOR__ >= 1) ) )
diff --git a/lib/zstd/common/entropy_common.c b/lib/zstd/common/entropy_common.c
--- a/lib/zstd/common/entropy_common.c
+++ b/lib/zstd/common/entropy_common.c
@@ -16,6 +16,10 @@
/* *************************************
* Dependencies
***************************************/
+#include "zstd_deps.h"
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "mem.h"
#include "error_private.h" /* ERR_*, ERROR */
#define FSE_STATIC_LINKING_ONLY /* FSE_MIN_TABLELOG */
@@ -210,11 +214,9 @@ size_t FSE_readNCount_bmi2(
short* normalizedCounter, unsigned* maxSVPtr, unsigned* tableLogPtr,
const void* headerBuffer, size_t hbSize, int bmi2)
{
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
return FSE_readNCount_body_bmi2(normalizedCounter, maxSVPtr, tableLogPtr, headerBuffer, hbSize);
}
-#endif
(void)bmi2;
return FSE_readNCount_body_default(normalizedCounter, maxSVPtr, tableLogPtr, headerBuffer, hbSize);
}
@@ -335,11 +337,9 @@ size_t HUF_readStats_wksp(BYTE* huffWeight, size_t hwSize, U32* rankStats,
void* workSpace, size_t wkspSize,
int flags)
{
-#if DYNAMIC_BMI2
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
return HUF_readStats_body_bmi2(huffWeight, hwSize, rankStats, nbSymbolsPtr, tableLogPtr, src, srcSize, workSpace, wkspSize);
}
-#endif
(void)flags;
return HUF_readStats_body_default(huffWeight, hwSize, rankStats, nbSymbolsPtr, tableLogPtr, src, srcSize, workSpace, wkspSize);
}
diff --git a/lib/zstd/common/fse_decompress.c b/lib/zstd/common/fse_decompress.c
--- a/lib/zstd/common/fse_decompress.c
+++ b/lib/zstd/common/fse_decompress.c
@@ -24,6 +24,9 @@
#include "fse.h"
#include "error_private.h"
#include "zstd_deps.h" /* ZSTD_memcpy */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "bits.h" /* ZSTD_highbit32 */
@@ -306,11 +309,9 @@ BMI2_TARGET_ATTRIBUTE static size_t FSE_decompress_wksp_body_bmi2(void* dst, siz
size_t FSE_decompress_wksp_bmi2(void* dst, size_t dstCapacity, const void* cSrc, size_t cSrcSize, unsigned maxLog, void* workSpace, size_t wkspSize, int bmi2)
{
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
return FSE_decompress_wksp_body_bmi2(dst, dstCapacity, cSrc, cSrcSize, maxLog, workSpace, wkspSize);
}
-#endif
(void)bmi2;
return FSE_decompress_wksp_body_default(dst, dstCapacity, cSrc, cSrcSize, maxLog, workSpace, wkspSize);
}
diff --git a/lib/zstd/common/zstd_deps.h b/lib/zstd/common/zstd_deps.h
--- a/lib/zstd/common/zstd_deps.h
+++ b/lib/zstd/common/zstd_deps.h
@@ -26,6 +26,11 @@
#ifndef ZSTD_DEPS_COMMON
#define ZSTD_DEPS_COMMON
+#if defined(__KERNEL__) && defined(CONFIG_X86) && \
+ !defined(__DISABLE_EXPORTS)
+#define ZSTD_USE_KERNEL_CPU_FEATURES
+#endif
+
#include <linux/limits.h>
#include <linux/stddef.h>
diff --git a/lib/zstd/compress/huf_compress.c b/lib/zstd/compress/huf_compress.c
--- a/lib/zstd/compress/huf_compress.c
+++ b/lib/zstd/compress/huf_compress.c
@@ -22,6 +22,9 @@
* Includes
****************************************************************/
#include "../common/zstd_deps.h" /* ZSTD_memcpy, ZSTD_memset */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "../common/compiler.h"
#include "../common/bitstream.h"
#include "hist.h"
@@ -1138,9 +1141,10 @@ HUF_compress1X_usingCTable_internal(void* dst, size_t dstSize,
const void* src, size_t srcSize,
const HUF_CElt* CTable, const int flags)
{
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
return HUF_compress1X_usingCTable_internal_bmi2(dst, dstSize, src, srcSize, CTable);
}
+ (void)flags;
return HUF_compress1X_usingCTable_internal_default(dst, dstSize, src, srcSize, CTable);
}
diff --git a/lib/zstd/compress/zstd_compress.c b/lib/zstd/compress/zstd_compress.c
--- a/lib/zstd/compress/zstd_compress.c
+++ b/lib/zstd/compress/zstd_compress.c
@@ -102,7 +102,7 @@ static void ZSTD_initCCtx(ZSTD_CCtx* cctx, ZSTD_customMem memManager)
assert(cctx != NULL);
ZSTD_memset(cctx, 0, sizeof(*cctx));
cctx->customMem = memManager;
- cctx->bmi2 = ZSTD_cpuSupportsBmi2();
+ ZSTD_SET_BMI2(cctx->bmi2, ZSTD_cpuSupportsBmi2());
{ size_t const err = ZSTD_CCtx_reset(cctx, ZSTD_reset_parameters);
assert(!ZSTD_isError(err));
(void)err;
@@ -142,7 +142,7 @@ ZSTD_CCtx* ZSTD_initStaticCCtx(void* workspace, size_t workspaceSize)
cctx->blockState.nextCBlock = (ZSTD_compressedBlockState_t*)ZSTD_cwksp_reserve_object(&cctx->workspace, sizeof(ZSTD_compressedBlockState_t));
cctx->tmpWorkspace = ZSTD_cwksp_reserve_object(&cctx->workspace, TMP_WORKSPACE_SIZE);
cctx->tmpWkspSize = TMP_WORKSPACE_SIZE;
- cctx->bmi2 = ZSTD_cpuid_bmi2(ZSTD_cpuid());
+ ZSTD_SET_BMI2(cctx->bmi2, ZSTD_cpuid_bmi2(ZSTD_cpuid()));
return cctx;
}
@@ -4042,7 +4042,7 @@ ZSTD_compressSeqStore_singleBlock(ZSTD_CCtx* zc,
op + ZSTD_blockHeaderSize, dstCapacity - ZSTD_blockHeaderSize,
srcSize,
zc->tmpWorkspace, zc->tmpWkspSize /* statically allocated in resetCCtx */,
- zc->bmi2);
+ ZSTD_CCtx_get_bmi2(zc));
FORWARD_IF_ERROR(cSeqsSize, "ZSTD_entropyCompressSeqStore failed!");
if (!zc->isFirstBlock &&
@@ -4332,7 +4332,7 @@ ZSTD_compressBlock_internal(ZSTD_CCtx* zc,
dst, dstCapacity,
srcSize,
zc->tmpWorkspace, zc->tmpWkspSize /* statically allocated in resetCCtx */,
- zc->bmi2);
+ ZSTD_CCtx_get_bmi2(zc));
if (frame &&
/* We don't want to emit our first block as a RLE even if it qualifies because
@@ -6796,7 +6796,7 @@ ZSTD_compressSequences_internal(ZSTD_CCtx* cctx,
op + ZSTD_blockHeaderSize /* Leave space for block header */, dstCapacity - ZSTD_blockHeaderSize,
blockSize,
cctx->tmpWorkspace, cctx->tmpWkspSize /* statically allocated in resetCCtx */,
- cctx->bmi2);
+ ZSTD_CCtx_get_bmi2(cctx));
FORWARD_IF_ERROR(compressedSeqsSize, "Compressing sequences of block failed");
DEBUGLOG(5, "Compressed sequences size: %zu", compressedSeqsSize);
@@ -7321,7 +7321,7 @@ ZSTD_compressSequencesAndLiterals_internal(ZSTD_CCtx* cctx,
&cctx->blockState.prevCBlock->entropy, &cctx->blockState.nextCBlock->entropy,
&cctx->appliedParams,
cctx->tmpWorkspace, cctx->tmpWkspSize /* statically allocated in resetCCtx */,
- cctx->bmi2);
+ ZSTD_CCtx_get_bmi2(cctx));
FORWARD_IF_ERROR(compressedSeqsSize, "Compressing sequences of block failed");
/* note: the spec forbids for any compressed block to be larger than maximum block size */
if (compressedSeqsSize > cctx->blockSizeMax) compressedSeqsSize = 0;
diff --git a/lib/zstd/compress/zstd_compress_internal.h b/lib/zstd/compress/zstd_compress_internal.h
--- a/lib/zstd/compress/zstd_compress_internal.h
+++ b/lib/zstd/compress/zstd_compress_internal.h
@@ -537,6 +537,15 @@ struct ZSTD_CCtx_s {
size_t extSeqBufCapacity;
};
+MEM_STATIC int ZSTD_CCtx_get_bmi2(const struct ZSTD_CCtx_s *cctx) {
+#if DYNAMIC_BMI2 && !defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+ return cctx->bmi2;
+#else
+ (void)cctx;
+ return 0;
+#endif
+}
+
typedef enum { ZSTD_dtlm_fast, ZSTD_dtlm_full } ZSTD_dictTableLoadMethod_e;
typedef enum { ZSTD_tfp_forCCtx, ZSTD_tfp_forCDict } ZSTD_tableFillPurpose_e;
diff --git a/lib/zstd/compress/zstd_compress_sequences.c b/lib/zstd/compress/zstd_compress_sequences.c
--- a/lib/zstd/compress/zstd_compress_sequences.c
+++ b/lib/zstd/compress/zstd_compress_sequences.c
@@ -12,6 +12,10 @@
/*-*************************************
* Dependencies
***************************************/
+#include "../common/zstd_deps.h"
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "zstd_compress_sequences.h"
/*
@@ -429,15 +433,13 @@ size_t ZSTD_encodeSequences(
SeqDef const* sequences, size_t nbSeq, int longOffsets, int bmi2)
{
DEBUGLOG(5, "ZSTD_encodeSequences: dstCapacity = %u", (unsigned)dstCapacity);
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
return ZSTD_encodeSequences_bmi2(dst, dstCapacity,
CTable_MatchLength, mlCodeTable,
CTable_OffsetBits, ofCodeTable,
CTable_LitLength, llCodeTable,
sequences, nbSeq, longOffsets);
}
-#endif
(void)bmi2;
return ZSTD_encodeSequences_default(dst, dstCapacity,
CTable_MatchLength, mlCodeTable,
diff --git a/lib/zstd/compress/zstd_compress_superblock.c b/lib/zstd/compress/zstd_compress_superblock.c
--- a/lib/zstd/compress/zstd_compress_superblock.c
+++ b/lib/zstd/compress/zstd_compress_superblock.c
@@ -684,6 +684,6 @@ size_t ZSTD_compressSuperBlock(ZSTD_CCtx* zc,
&zc->appliedParams,
dst, dstCapacity,
src, srcSize,
- zc->bmi2, lastBlock,
+ ZSTD_CCtx_get_bmi2(zc), lastBlock,
zc->tmpWorkspace, zc->tmpWkspSize /* statically allocated in resetCCtx */);
}
diff --git a/lib/zstd/decompress/huf_decompress.c b/lib/zstd/decompress/huf_decompress.c
--- a/lib/zstd/decompress/huf_decompress.c
+++ b/lib/zstd/decompress/huf_decompress.c
@@ -17,6 +17,9 @@
* Dependencies
****************************************************************/
#include "../common/zstd_deps.h" /* ZSTD_memcpy, ZSTD_memset */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "../common/compiler.h"
#include "../common/bitstream.h" /* BIT_* */
#include "../common/fse.h" /* to compress headers */
@@ -113,9 +116,10 @@ typedef size_t (*HUF_DecompressUsingDTableFn)(void *dst, size_t dstSize,
static size_t fn(void* dst, size_t dstSize, void const* cSrc, \
size_t cSrcSize, HUF_DTable const* DTable, int flags) \
{ \
- if (flags & HUF_flags_bmi2) { \
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) { \
return fn##_bmi2(dst, dstSize, cSrc, cSrcSize, DTable); \
} \
+ (void)flags; \
return fn##_default(dst, dstSize, cSrc, cSrcSize, DTable); \
}
@@ -899,18 +903,16 @@ static size_t HUF_decompress4X1_usingDTable_internal(void* dst, size_t dstSize,
HUF_DecompressUsingDTableFn fallbackFn = HUF_decompress4X1_usingDTable_internal_default;
HUF_DecompressFastLoopFn loopFn = HUF_decompress4X1_usingDTable_internal_fast_c_loop;
-#if DYNAMIC_BMI2
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
fallbackFn = HUF_decompress4X1_usingDTable_internal_bmi2;
# if ZSTD_ENABLE_ASM_X86_64_BMI2
if (!(flags & HUF_flags_disableAsm)) {
loopFn = HUF_decompress4X1_usingDTable_internal_fast_asm_loop;
}
# endif
- } else {
+ } else if (DYNAMIC_BMI2) {
return fallbackFn(dst, dstSize, cSrc, cSrcSize, DTable);
}
-#endif
#if ZSTD_ENABLE_ASM_X86_64_BMI2 && defined(__BMI2__)
if (!(flags & HUF_flags_disableAsm)) {
@@ -1723,18 +1725,16 @@ static size_t HUF_decompress4X2_usingDTable_internal(void* dst, size_t dstSize,
HUF_DecompressUsingDTableFn fallbackFn = HUF_decompress4X2_usingDTable_internal_default;
HUF_DecompressFastLoopFn loopFn = HUF_decompress4X2_usingDTable_internal_fast_c_loop;
-#if DYNAMIC_BMI2
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
fallbackFn = HUF_decompress4X2_usingDTable_internal_bmi2;
# if ZSTD_ENABLE_ASM_X86_64_BMI2
if (!(flags & HUF_flags_disableAsm)) {
loopFn = HUF_decompress4X2_usingDTable_internal_fast_asm_loop;
}
# endif
- } else {
+ } else if (DYNAMIC_BMI2) {
return fallbackFn(dst, dstSize, cSrc, cSrcSize, DTable);
}
-#endif
#if ZSTD_ENABLE_ASM_X86_64_BMI2 && defined(__BMI2__)
if (!(flags & HUF_flags_disableAsm)) {
diff --git a/lib/zstd/decompress/zstd_decompress.c b/lib/zstd/decompress/zstd_decompress.c
--- a/lib/zstd/decompress/zstd_decompress.c
+++ b/lib/zstd/decompress/zstd_decompress.c
@@ -259,9 +259,7 @@ static void ZSTD_initDCtx_internal(ZSTD_DCtx* dctx)
dctx->noForwardProgress = 0;
dctx->oversizedDuration = 0;
dctx->isFrameDecompression = 1;
-#if DYNAMIC_BMI2
- dctx->bmi2 = ZSTD_cpuSupportsBmi2();
-#endif
+ ZSTD_SET_BMI2(dctx->bmi2, ZSTD_cpuSupportsBmi2());
dctx->ddictSet = NULL;
ZSTD_DCtx_resetParameters(dctx);
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
diff --git a/lib/zstd/decompress/zstd_decompress_block.c b/lib/zstd/decompress/zstd_decompress_block.c
--- a/lib/zstd/decompress/zstd_decompress_block.c
+++ b/lib/zstd/decompress/zstd_decompress_block.c
@@ -16,6 +16,9 @@
* Dependencies
*********************************************************/
#include "../common/zstd_deps.h" /* ZSTD_memcpy, ZSTD_memmove, ZSTD_memset */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "../common/compiler.h" /* prefetch */
#include "../common/cpu.h" /* bmi2 */
#include "../common/mem.h" /* low level memory routines */
@@ -631,13 +634,11 @@ void ZSTD_buildFSETable(ZSTD_seqSymbol* dt,
const U32* baseValue, const U8* nbAdditionalBits,
unsigned tableLog, void* wksp, size_t wkspSize, int bmi2)
{
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
ZSTD_buildFSETable_body_bmi2(dt, normalizedCounter, maxSymbolValue,
baseValue, nbAdditionalBits, tableLog, wksp, wkspSize);
return;
}
-#endif
(void)bmi2;
ZSTD_buildFSETable_body_default(dt, normalizedCounter, maxSymbolValue,
baseValue, nbAdditionalBits, tableLog, wksp, wkspSize);
@@ -1955,11 +1956,9 @@ ZSTD_decompressSequences(ZSTD_DCtx* dctx, void* dst, size_t maxDstSize,
const ZSTD_longOffset_e isLongOffset)
{
DEBUGLOG(5, "ZSTD_decompressSequences");
-#if DYNAMIC_BMI2
- if (ZSTD_DCtx_get_bmi2(dctx)) {
+ if (ZSTD_USE_BMI2(ZSTD_DCtx_get_bmi2(dctx))) {
return ZSTD_decompressSequences_bmi2(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
-#endif
return ZSTD_decompressSequences_default(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
static size_t
@@ -1968,11 +1967,9 @@ ZSTD_decompressSequencesSplitLitBuffer(ZSTD_DCtx* dctx, void* dst, size_t maxDst
const ZSTD_longOffset_e isLongOffset)
{
DEBUGLOG(5, "ZSTD_decompressSequencesSplitLitBuffer");
-#if DYNAMIC_BMI2
- if (ZSTD_DCtx_get_bmi2(dctx)) {
+ if (ZSTD_USE_BMI2(ZSTD_DCtx_get_bmi2(dctx))) {
return ZSTD_decompressSequencesSplitLitBuffer_bmi2(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
-#endif
return ZSTD_decompressSequencesSplitLitBuffer_default(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
#endif /* ZSTD_FORCE_DECOMPRESS_SEQUENCES_LONG */
@@ -1991,11 +1988,9 @@ ZSTD_decompressSequencesLong(ZSTD_DCtx* dctx,
const ZSTD_longOffset_e isLongOffset)
{
DEBUGLOG(5, "ZSTD_decompressSequencesLong");
-#if DYNAMIC_BMI2
- if (ZSTD_DCtx_get_bmi2(dctx)) {
+ if (ZSTD_USE_BMI2(ZSTD_DCtx_get_bmi2(dctx))) {
return ZSTD_decompressSequencesLong_bmi2(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
-#endif
return ZSTD_decompressSequencesLong_default(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
#endif /* ZSTD_FORCE_DECOMPRESS_SEQUENCES_SHORT */
diff --git a/lib/zstd/decompress/zstd_decompress_internal.h b/lib/zstd/decompress/zstd_decompress_internal.h
--- a/lib/zstd/decompress/zstd_decompress_internal.h
+++ b/lib/zstd/decompress/zstd_decompress_internal.h
@@ -204,7 +204,7 @@ struct ZSTD_DCtx_s
}; /* typedef'd to ZSTD_DCtx within "zstd.h" */
MEM_STATIC int ZSTD_DCtx_get_bmi2(const struct ZSTD_DCtx_s *dctx) {
-#if DYNAMIC_BMI2
+#if DYNAMIC_BMI2 && !defined(ZSTD_USE_KERNEL_CPU_FEATURES)
return dctx->bmi2;
#else
(void)dctx;
@@ -0,0 +1,44 @@
diff --git a/crypto/zstd.c b/crypto/zstd.c
--- a/crypto/zstd.c
+++ b/crypto/zstd.c
@@ -96,6 +96,7 @@ static int zstd_compress_one(struct acomp_req *req, struct zstd_ctx *ctx,
static int zstd_compress(struct acomp_req *req)
{
+ bool stream_initialized = false;
struct crypto_acomp_stream *s;
unsigned int pos, scur, dcur;
unsigned int total_out = 0;
@@ -115,12 +116,6 @@ static int zstd_compress(struct acomp_req *req)
if (ret)
goto out;
- ctx->cctx = zstd_init_cstream(&ctx->params, 0, ctx->wksp, ctx->wksp_size);
- if (!ctx->cctx) {
- ret = -EINVAL;
- goto out;
- }
-
do {
dcur = acomp_walk_next_dst(&walk);
if (!dcur) {
@@ -142,6 +137,19 @@ static int zstd_compress(struct acomp_req *req)
goto out;
}
+ if (!stream_initialized) {
+ ctx->cctx = zstd_init_cstream(&ctx->params, 0,
+ ctx->wksp, ctx->wksp_size);
+ if (!ctx->cctx) {
+ /* Release in the reverse of the map order. */
+ acomp_walk_done_src(&walk, 0);
+ acomp_walk_done_dst(&walk, 0);
+ ret = -EINVAL;
+ goto out;
+ }
+ stream_initialized = true;
+ }
+
if (scur) {
inbuf.pos = 0;
inbuf.src = walk.src.virt.addr;
@@ -0,0 +1,44 @@
diff --git a/crypto/zstd.c b/crypto/zstd.c
--- a/crypto/zstd.c
+++ b/crypto/zstd.c
@@ -215,6 +215,7 @@ static int zstd_decompress_one(struct acomp_req *req, struct zstd_ctx *ctx,
static int zstd_decompress(struct acomp_req *req)
{
+ bool stream_initialized = false;
struct crypto_acomp_stream *s;
unsigned int total_out = 0;
unsigned int scur, dcur;
@@ -232,12 +233,6 @@ static int zstd_decompress(struct acomp_req *req)
if (ret)
goto out;
- ctx->dctx = zstd_init_dstream(ZSTD_MAX_SIZE, ctx->wksp, ctx->wksp_size);
- if (!ctx->dctx) {
- ret = -EINVAL;
- goto out;
- }
-
do {
scur = acomp_walk_next_src(&walk);
if (scur) {
@@ -263,6 +258,19 @@ static int zstd_decompress(struct acomp_req *req)
goto out;
}
+ if (!stream_initialized) {
+ ctx->dctx = zstd_init_dstream(ZSTD_MAX_SIZE, ctx->wksp,
+ ctx->wksp_size);
+ if (!ctx->dctx) {
+ /* Release in the reverse of the map order. */
+ acomp_walk_done_dst(&walk, 0);
+ acomp_walk_done_src(&walk, 0);
+ ret = -EINVAL;
+ goto out;
+ }
+ stream_initialized = true;
+ }
+
outbuf.pos = 0;
outbuf.dst = (u8 *)walk.dst.virt.addr;
outbuf.size = dcur;
@@ -0,0 +1,363 @@
--- a/crypto/af_alg.c
+++ b/crypto/af_alg.c
@@ -8,6 +8,7 @@
*/
#include <linux/atomic.h>
+#include <linux/capability.h>
#include <crypto/if_alg.h>
#include <linux/crypto.h>
#include <linux/init.h>
@@ -22,10 +23,28 @@
#include <linux/sched/signal.h>
#include <linux/security.h>
#include <linux/string.h>
+#include <linux/sysctl.h>
+#include <linux/user_namespace.h>
#include <keys/user-type.h>
#include <keys/trusted-type.h>
#include <keys/encrypted-type.h>
+static int af_alg_restrict = 1;
+
+static const struct ctl_table af_alg_table[] = {
+ {
+ .procname = "af_alg_restrict",
+ .data = &af_alg_restrict,
+ .maxlen = sizeof(int),
+ .mode = 0644,
+ .proc_handler = proc_dointvec_minmax,
+ .extra1 = SYSCTL_ZERO,
+ .extra2 = SYSCTL_TWO,
+ },
+};
+
+static struct ctl_table_header *af_alg_header;
+
struct alg_type_list {
const struct af_alg_type *type;
struct list_head list;
@@ -110,6 +129,43 @@
}
EXPORT_SYMBOL_GPL(af_alg_unregister_type);
+static bool af_alg_capable(void)
+{
+ return ns_capable_noaudit(&init_user_ns, CAP_NET_ADMIN) ||
+ capable(CAP_SYS_ADMIN);
+}
+
+int af_alg_check_restriction(const char *name,
+ const struct af_alg_allowlist_entry allowlist[])
+{
+ int level = READ_ONCE(af_alg_restrict);
+
+ if (level == 0)
+ return 0;
+ if (level == 1) {
+ for (const struct af_alg_allowlist_entry *ent = allowlist;
+ ent->name; ent++) {
+ if (strcmp(name, ent->name) == 0) {
+ if ((ent->flags & AF_ALG_UNPRIVILEGED) ||
+ af_alg_capable())
+ return 0;
+ /* List contains at most one entry per name. */
+ break;
+ }
+ }
+ }
+ /*
+ * Use -ENOENT (the error code for "algorithm not found") instead of
+ * -EACCES or -EPERM, for the highest chance of correctly triggering
+ * fallback code paths in userspace programs.
+ *
+ * Don't log a warning, since it would be noisy. iwd tries to bind a
+ * bunch of algorithms that it never uses.
+ */
+ return -ENOENT;
+}
+EXPORT_SYMBOL_GPL(af_alg_check_restriction);
+
static void alg_do_release(const struct af_alg_type *type, void *private)
{
if (!type)
@@ -506,6 +562,9 @@
struct sock *sk;
int err;
+ if (READ_ONCE(af_alg_restrict) == 2)
+ return -EAFNOSUPPORT;
+
if (sock->type != SOCK_SEQPACKET)
return -ESOCKTNOSUPPORT;
if (protocol != 0)
@@ -1222,27 +1281,32 @@
static int __init af_alg_init(void)
{
- int err = proto_register(&alg_proto, 0);
+ int err;
+
+ af_alg_header = register_sysctl("crypto", af_alg_table);
+ err = proto_register(&alg_proto, 0);
if (err)
- goto out;
+ goto out_unregister_sysctl;
err = sock_register(&alg_family);
- if (err != 0)
+ if (err)
goto out_unregister_proto;
-out:
- return err;
+ return 0;
out_unregister_proto:
proto_unregister(&alg_proto);
- goto out;
+out_unregister_sysctl:
+ unregister_sysctl_table(af_alg_header);
+ return err;
}
static void __exit af_alg_exit(void)
{
sock_unregister(PF_ALG);
proto_unregister(&alg_proto);
+ unregister_sysctl_table(af_alg_header);
}
module_init(af_alg_init);
--- a/crypto/algif_aead.c
+++ b/crypto/algif_aead.c
@@ -34,6 +34,11 @@
#include <linux/net.h>
#include <net/sock.h>
+static const struct af_alg_allowlist_entry aead_allowlist[] = {
+ { "ccm(aes)" }, /* bluez */
+ {},
+};
+
static inline bool aead_sufficient_data(struct sock *sk)
{
struct alg_sock *ask = alg_sk(sk);
@@ -344,6 +349,12 @@
static void *aead_bind(const char *name)
{
+ int err;
+
+ err = af_alg_check_restriction(name, aead_allowlist);
+ if (err)
+ return ERR_PTR(err);
+
return crypto_alloc_aead(name, 0, AF_ALG_CRYPTOAPI_MASK);
}
--- a/crypto/algif_hash.c
+++ b/crypto/algif_hash.c
@@ -16,6 +16,24 @@
#include <linux/net.h>
#include <net/sock.h>
+static const struct af_alg_allowlist_entry hash_allowlist[] = {
+ { "cmac(aes)" }, /* iwd, bluez */
+ { "hmac(md5)" }, /* iwd */
+ { "hmac(sha1)" }, /* iwd */
+ { "hmac(sha224)" }, /* iwd */
+ { "hmac(sha256)" }, /* iwd */
+ { "hmac(sha384)" }, /* iwd */
+ { "hmac(sha512)" }, /* iwd, sha512hmac */
+ { "md4" }, /* iwd */
+ { "md5" }, /* iwd */
+ { "sha1", AF_ALG_UNPRIVILEGED }, /* iwd, iproute2 < 7.0 */
+ { "sha224" }, /* iwd */
+ { "sha256" }, /* iwd */
+ { "sha384" }, /* iwd */
+ { "sha512" }, /* iwd */
+ {},
+};
+
struct hash_ctx {
struct af_alg_sgl sgl;
@@ -382,6 +400,12 @@
static void *hash_bind(const char *name)
{
+ int err;
+
+ err = af_alg_check_restriction(name, hash_allowlist);
+ if (err)
+ return ERR_PTR(err);
+
return crypto_alloc_ahash(name, 0, AF_ALG_CRYPTOAPI_MASK);
}
--- a/crypto/algif_rng.c
+++ b/crypto/algif_rng.c
@@ -50,6 +50,10 @@
MODULE_AUTHOR("Stephan Mueller <smueller@chronox.de>");
MODULE_DESCRIPTION("User-space interface for random number generators");
+static const struct af_alg_allowlist_entry rng_allowlist[] = {
+ {},
+};
+
struct rng_ctx {
#define MAXSIZE 128
unsigned int len;
@@ -201,6 +205,11 @@
{
struct rng_parent_ctx *pctx;
struct crypto_rng *rng;
+ int err;
+
+ err = af_alg_check_restriction(name, rng_allowlist);
+ if (err)
+ return ERR_PTR(err);
pctx = kzalloc_obj(*pctx);
if (!pctx)
--- a/crypto/algif_skcipher.c
+++ b/crypto/algif_skcipher.c
@@ -35,6 +35,24 @@
#include <linux/string.h>
#include <net/sock.h>
+static const struct af_alg_allowlist_entry skcipher_allowlist[] = {
+ { "adiantum(xchacha12,aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "adiantum(xchacha20,aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "cbc(aes)" }, /* iwd */
+ { "cbc(des)" }, /* iwd */
+ { "cbc(des3_ede)" }, /* iwd */
+ { "cbc(paes)" }, /* caam and others */
+ { "ctr(aes)" }, /* iwd */
+ { "ecb(aes)" }, /* iwd, bluez */
+ { "ecb(des)" }, /* iwd */
+ { "hctr2(aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "xts(aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup benchmark */
+ { "xts(camellia)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "xts(serpent)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "xts(twofish)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ {},
+};
+
static int skcipher_sendmsg(struct socket *sock, struct msghdr *msg,
size_t size)
{
@@ -311,6 +329,11 @@
static void *skcipher_bind(const char *name)
{
u32 mask = AF_ALG_CRYPTOAPI_MASK;
+ int err;
+
+ err = af_alg_check_restriction(name, skcipher_allowlist);
+ if (err)
+ return ERR_PTR(err);
if (strcmp(name, "cbc(paes)") == 0)
mask = 0;
--- a/include/crypto/if_alg.h
+++ b/include/crypto/if_alg.h
@@ -8,6 +8,7 @@
#ifndef _CRYPTO_IF_ALG_H
#define _CRYPTO_IF_ALG_H
+#include <linux/bits.h>
#include <linux/compiler.h>
#include <linux/completion.h>
#include <linux/if_alg.h>
@@ -121,7 +122,7 @@
* @iv: IV for cipher operation
* @state: Existing state for continuing operation
* @aead_assoclen: Length of AAD for AEAD cipher operations
- * @completion: Work queue for synchronous operation
+ * @wait: For waiting for completion of async crypto ops
* @used: TX bytes sent to kernel. This variable is used to
* ensure that user space cannot cause the kernel
* to allocate too much memory in sendmsg operation.
@@ -161,9 +162,20 @@
unsigned int inflight;
};
+/* Flags for af_alg_allowlist_entry::flags: */
+#define AF_ALG_UNPRIVILEGED BIT(0) /* Unprivileged use is allowed */
+
+struct af_alg_allowlist_entry {
+ const char *name;
+ u32 flags;
+};
+
int af_alg_register_type(const struct af_alg_type *type);
int af_alg_unregister_type(const struct af_alg_type *type);
+int af_alg_check_restriction(const char *name,
+ const struct af_alg_allowlist_entry allowlist[]);
+
int af_alg_release(struct socket *sock);
void af_alg_release_parent(struct sock *sk);
int af_alg_accept(struct sock *sk, struct socket *newsock,
@@ -177,10 +189,11 @@
}
/**
- * Size of available buffer for sending data from user space to kernel.
+ * af_alg_sndbuf - Size of available buffer for sending data from user space to kernel.
*
- * @sk socket of connection to user space
- * @return number of bytes still available
+ * @sk: socket of connection to user space
+ *
+ * Returns: number of bytes still available
*/
static inline int af_alg_sndbuf(struct sock *sk)
{
@@ -192,10 +205,11 @@
}
/**
- * Can the send buffer still be written to?
+ * af_alg_writable - Can the send buffer still be written to?
+ *
+ * @sk: socket of connection to user space
*
- * @sk socket of connection to user space
- * @return true => writable, false => not writable
+ * Returns: true => writable, false => not writable
*/
static inline bool af_alg_writable(struct sock *sk)
{
@@ -203,10 +217,11 @@
}
/**
- * Size of available buffer used by kernel for the RX user space operation.
+ * af_alg_rcvbuf - Size of available buffer used by kernel for the RX user space operation.
*
- * @sk socket of connection to user space
- * @return number of bytes still available
+ * @sk: socket of connection to user space
+ *
+ * Returns: number of bytes still available
*/
static inline int af_alg_rcvbuf(struct sock *sk)
{
@@ -218,10 +233,11 @@
}
/**
- * Can the RX buffer still be written to?
+ * af_alg_readable - Can the RX buffer still be read from?
+ *
+ * @sk: socket of connection to user space
*
- * @sk socket of connection to user space
- * @return true => writable, false => not writable
+ * Returns: true => readable, false => not readable
*/
static inline bool af_alg_readable(struct sock *sk)
{
@@ -0,0 +1,12 @@
diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h
--- a/arch/x86/include/asm/pgtable.h
+++ b/arch/x86/include/asm/pgtable.h
@@ -806,7 +806,7 @@ static inline pmd_t pmd_modify(pmd_t pmd, pgprot_t newprot)
pmdval_t val = pmd_val(pmd), oldval = val;
pmd_t pmd_result;
- val &= (_HPAGE_CHG_MASK & ~_PAGE_DIRTY);
+ val &= _HPAGE_CHG_MASK;
val |= check_pgprot(newprot) & ~_HPAGE_CHG_MASK;
val = flip_protnone_guard(oldval, val, PHYSICAL_PMD_PAGE_MASK);
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,386 @@
diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c
--- a/fs/btrfs/dev-replace.c
+++ b/fs/btrfs/dev-replace.c
@@ -626,7 +626,7 @@ static int btrfs_dev_replace_start(struct btrfs_fs_info *fs_info,
ret = mark_block_group_to_copy(fs_info, src_device);
if (ret)
- return ret;
+ goto leave;
down_write(&dev_replace->rwsem);
dev_replace->replace_task = current;
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -3436,6 +3436,9 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent)
*/
btrfs_remove_ordered_extent(ordered_extent);
+ /* Cleanup any remaining biocs attached to the OE. */
+ btrfs_cleanup_ordered_bioc_list(ordered_extent);
+
/* once for us */
btrfs_put_ordered_extent(ordered_extent);
/* once for the tree */
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -384,6 +384,7 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags &= ~BTRFS_INODE_COMPRESS;
inode_flags |= BTRFS_INODE_NOCOMPRESS;
} else if (fsflags & FS_COMPR_FL) {
+ enum btrfs_compression_type comp_type;
if (IS_SWAPFILE(&inode->vfs_inode))
return -ETXTBSY;
@@ -391,9 +392,23 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags |= BTRFS_INODE_COMPRESS;
inode_flags &= ~BTRFS_INODE_NOCOMPRESS;
- comp = btrfs_compress_type2str(fs_info->compress_type);
- if (!comp || comp[0] == 0)
- comp = btrfs_compress_type2str(BTRFS_COMPRESS_ZLIB);
+ /*
+ * Keep the algorithm recorded in the compression property,
+ * otherwise changing an unrelated attribute would reset it to
+ * the mount default, since FS_IOC_SETFLAGS callers write back
+ * the whole flag set they got from FS_IOC_GETFLAGS and that
+ * includes FS_COMPR_FL for any inode carrying the property.
+ *
+ * Inodes with the compress flag set but no property keep using
+ * the mount default, so they behave as before.
+ */
+ if (inode->prop_compress)
+ comp_type = inode->prop_compress;
+ else if (fs_info->compress_type)
+ comp_type = fs_info->compress_type;
+ else
+ comp_type = BTRFS_COMPRESS_ZLIB;
+ comp = btrfs_compress_type2str(comp_type);
} else {
inode_flags &= ~(BTRFS_INODE_COMPRESS | BTRFS_INODE_NOCOMPRESS);
}
diff --git a/fs/btrfs/raid-stripe-tree.c b/fs/btrfs/raid-stripe-tree.c
--- a/fs/btrfs/raid-stripe-tree.c
+++ b/fs/btrfs/raid-stripe-tree.c
@@ -310,8 +310,10 @@ static int update_raid_extent_item(struct btrfs_trans_handle *trans,
ret = btrfs_search_slot(trans, trans->fs_info->stripe_root, key, path,
0, 1);
- if (ret)
- return (ret == 1 ? ret : -EINVAL);
+ if (ret > 0)
+ ret = -ENOENT;
+ if (ret < 0)
+ return ret;
leaf = path->nodes[0];
slot = path->slots[0];
@@ -337,7 +339,6 @@ int btrfs_insert_one_raid_extent(struct btrfs_trans_handle *trans,
stripe_extent = kzalloc(item_size, GFP_NOFS);
if (unlikely(!stripe_extent)) {
btrfs_abort_transaction(trans, -ENOMEM);
- btrfs_end_transaction(trans);
return -ENOMEM;
}
@@ -374,7 +375,7 @@ int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans,
struct btrfs_ordered_extent *ordered_extent)
{
struct btrfs_io_context *bioc;
- int ret;
+ int ret = 0;
if (!btrfs_fs_incompat(trans->fs_info, RAID_STRIPE_TREE))
return 0;
@@ -382,17 +383,23 @@ int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans,
list_for_each_entry(bioc, &ordered_extent->bioc_list, rst_ordered_entry) {
ret = btrfs_insert_one_raid_extent(trans, bioc);
if (ret)
- return ret;
+ break;
}
- while (!list_empty(&ordered_extent->bioc_list)) {
- bioc = list_first_entry(&ordered_extent->bioc_list,
+ btrfs_cleanup_ordered_bioc_list(ordered_extent);
+ return ret;
+}
+
+void btrfs_cleanup_ordered_bioc_list(struct btrfs_ordered_extent *ordered)
+{
+ while (!list_empty(&ordered->bioc_list)) {
+ struct btrfs_io_context *bioc;
+
+ bioc = list_first_entry(&ordered->bioc_list,
typeof(*bioc), rst_ordered_entry);
list_del(&bioc->rst_ordered_entry);
btrfs_put_bioc(bioc);
}
-
- return 0;
}
int btrfs_get_raid_extent_offset(struct btrfs_fs_info *fs_info,
diff --git a/fs/btrfs/raid-stripe-tree.h b/fs/btrfs/raid-stripe-tree.h
--- a/fs/btrfs/raid-stripe-tree.h
+++ b/fs/btrfs/raid-stripe-tree.h
@@ -28,6 +28,7 @@ int btrfs_get_raid_extent_offset(struct btrfs_fs_info *fs_info,
u32 stripe_index, struct btrfs_io_stripe *stripe);
int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans,
struct btrfs_ordered_extent *ordered_extent);
+void btrfs_cleanup_ordered_bioc_list(struct btrfs_ordered_extent *ordered);
#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS
int btrfs_insert_one_raid_extent(struct btrfs_trans_handle *trans,
diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c
--- a/fs/btrfs/scrub.c
+++ b/fs/btrfs/scrub.c
@@ -1023,6 +1023,10 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx,
skip:
for_each_set_bit(sector_nr, &extent_bitmap, stripe->nr_sectors) {
+ const u64 sector_logical = stripe->logical +
+ ((u64)sector_nr << fs_info->sectorsize_bits);
+ const u64 sector_physical = physical +
+ ((u64)sector_nr << fs_info->sectorsize_bits);
bool repaired = false;
if (scrub_bitmap_test_bit_is_metadata(stripe, sector_nr)) {
@@ -1051,12 +1055,12 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx,
if (dev) {
btrfs_err_rl(fs_info,
"scrub: fixed up error at logical %llu on dev %s physical %llu",
- stripe->logical, btrfs_dev_name(dev),
- physical);
+ sector_logical, btrfs_dev_name(dev),
+ sector_physical);
} else {
btrfs_err_rl(fs_info,
"scrub: fixed up error at logical %llu on mirror %u",
- stripe->logical, stripe->mirror_num);
+ sector_logical, stripe->mirror_num);
}
continue;
}
@@ -1065,30 +1069,30 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx,
if (dev) {
btrfs_err_rl(fs_info,
"scrub: unable to fixup (regular) error at logical %llu on dev %s physical %llu",
- stripe->logical, btrfs_dev_name(dev),
- physical);
+ sector_logical, btrfs_dev_name(dev),
+ sector_physical);
} else {
btrfs_err_rl(fs_info,
"scrub: unable to fixup (regular) error at logical %llu on mirror %u",
- stripe->logical, stripe->mirror_num);
+ sector_logical, stripe->mirror_num);
}
if (scrub_bitmap_test_bit_io_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("i/o error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
if (scrub_bitmap_test_bit_csum_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("checksum error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
if (scrub_bitmap_test_bit_meta_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("header error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
if (scrub_bitmap_test_bit_meta_gen_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("generation error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
}
/* Update the device stats. */
diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c
--- a/fs/btrfs/send.c
+++ b/fs/btrfs/send.c
@@ -2065,7 +2065,7 @@ static int will_overwrite_ref(struct send_ctx *sctx, u64 dir, u64 dir_gen,
ret = is_inode_existent(sctx, dir, dir_gen, NULL, &parent_root_dir_gen);
if (ret <= 0)
- return 0;
+ return ret;
/*
* If we have a parent root we need to verify that the parent dir was
@@ -6417,6 +6417,13 @@ static int process_extent(struct send_ctx *sctx,
if (S_ISLNK(sctx->cur_inode_mode))
return 0;
+ if (unlikely(!S_ISREG(sctx->cur_inode_mode))) {
+ btrfs_crit(sctx->send_root->fs_info,
+ "send: extent for non-regular inode %llu root %llu mode 0%llo",
+ key->objectid, btrfs_root_id(sctx->send_root),
+ sctx->cur_inode_mode & S_IFMT);
+ return -EUCLEAN;
+ }
if (sctx->parent_root && !sctx->cur_inode_new) {
ret = is_extent_unchanged(sctx, path, key);
diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c
--- a/fs/btrfs/tests/extent-io-tests.c
+++ b/fs/btrfs/tests/extent-io-tests.c
@@ -133,14 +133,14 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
if (IS_ERR(root)) {
test_std_err(TEST_ALLOC_ROOT);
ret = PTR_ERR(root);
- goto out;
+ goto out_root_info;
}
inode = btrfs_new_test_inode();
if (!inode) {
test_std_err(TEST_ALLOC_INODE);
ret = -ENOMEM;
- goto out;
+ goto out_root_info;
}
tmp = &BTRFS_I(inode)->io_tree;
BTRFS_I(inode)->root = root;
@@ -333,6 +333,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
process_page_range(inode, 0, total_dirty - 1,
PROCESS_UNLOCK | PROCESS_RELEASE);
iput(inode);
+out_root_info:
btrfs_free_dummy_root(root);
btrfs_free_dummy_fs_info(fs_info);
return ret;
diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c
--- a/fs/btrfs/transaction.c
+++ b/fs/btrfs/transaction.c
@@ -458,8 +458,19 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans,
* through btrfs_record_root_in_trans without having to take the
* lock. smp_wmb() makes sure that all the writes above are
* done before we pop in the zero below
+ *
+ * If @force is true, it means the call is from
+ * qgroup_account_snapshot(), which only requires radix tree
+ * tracking.
+ * We should not force reloc root creation here, as the root
+ * may have already been modified, and in that case
+ * root->commit_root has already been dropped.
+ *
+ * Using that commit root will cause the reloc root to refer
+ * to a deleted extent, causing extent tree corruption.
*/
- ret = btrfs_init_reloc_root(trans, root);
+ if (!force)
+ ret = btrfs_init_reloc_root(trans, root);
smp_mb__before_atomic();
clear_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state);
}
@@ -2583,6 +2594,12 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans)
ret = btrfs_write_and_wait_transaction(trans);
if (unlikely(ret)) {
btrfs_err(fs_info, "error while writing out transaction: %pe", ERR_PTR(ret));
+ /*
+ * Abort before releasing tree_log_mutex, so a log sync waiting
+ * on it sees the fs error and skips writing super_for_commit
+ * for this failed transaction. See btrfs_sync_log().
+ */
+ btrfs_abort_transaction(trans, ret);
mutex_unlock(&fs_info->tree_log_mutex);
goto scrub_continue;
}
diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c
--- a/fs/btrfs/volumes.c
+++ b/fs/btrfs/volumes.c
@@ -3035,7 +3035,11 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path
error_sysfs:
btrfs_sysfs_remove_device(device);
mutex_lock(&fs_info->fs_devices->device_list_mutex);
+ if (seeding_dev)
+ btrfs_assign_next_active_device(device, seed_devices->latest_dev);
mutex_lock(&fs_info->chunk_mutex);
+ if (!list_empty(&device->post_commit_list))
+ list_del_init(&device->post_commit_list);
list_del_rcu(&device->dev_list);
list_del(&device->dev_alloc_list);
fs_info->fs_devices->num_devices--;
diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c
--- a/fs/btrfs/zoned.c
+++ b/fs/btrfs/zoned.c
@@ -2626,16 +2626,13 @@ static int do_zone_finish(struct btrfs_block_group *block_group, bool fully_writ
down_read(&dev_replace->rwsem);
map = block_group->physical_map;
for (i = 0; i < map->num_stripes; i++) {
-
ret = call_zone_finish(block_group, &map->stripes[i]);
- if (ret) {
- up_read(&dev_replace->rwsem);
- return ret;
- }
+ if (ret)
+ break;
}
up_read(&dev_replace->rwsem);
- if (!fully_written)
+ if (!ret && !fully_written)
btrfs_dec_block_group_ro(block_group);
spin_lock(&fs_info->zone_active_bgs_lock);
@@ -2648,7 +2645,7 @@ static int do_zone_finish(struct btrfs_block_group *block_group, bool fully_writ
clear_and_wake_up_bit(BTRFS_FS_NEED_ZONE_FINISH, &fs_info->flags);
- return 0;
+ return ret;
}
int btrfs_zone_finish(struct btrfs_block_group *block_group)
@@ -2713,6 +2710,7 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng
{
struct btrfs_block_group *block_group;
u64 min_alloc_bytes;
+ int ret = 0;
if (!btrfs_is_zoned(fs_info))
return 0;
@@ -2732,11 +2730,11 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng
block_group->start + block_group->zone_capacity)
goto out;
- do_zone_finish(block_group, true);
+ ret = do_zone_finish(block_group, true);
out:
btrfs_put_block_group(block_group);
- return 0;
+ return ret;
}
static void btrfs_zone_finish_endio_workfn(struct work_struct *work)
diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c
--- a/fs/btrfs/zstd.c
+++ b/fs/btrfs/zstd.c
@@ -307,8 +307,17 @@ struct list_head *zstd_get_workspace(struct btrfs_fs_info *fs_info, int level)
DEFINE_WAIT(wait);
prepare_to_wait(&zwsm->wait, &wait, TASK_UNINTERRUPTIBLE);
- schedule();
+ /*
+ * Re-check after being queued: zstd_put_workspace() only wakes
+ * a queue that already has a sleeper, so a workspace returned
+ * since the failed allocation woke nobody.
+ */
+ ws = zstd_find_workspace(fs_info, level);
+ if (!ws)
+ schedule();
finish_wait(&zwsm->wait, &wait);
+ if (ws)
+ return ws;
goto again;
}
@@ -0,0 +1,109 @@
diff --git c/fs/btrfs/zstd.c i/fs/btrfs/zstd.c
--- c/fs/btrfs/zstd.c
+++ i/fs/btrfs/zstd.c
@@ -589,10 +589,48 @@ int zstd_compress_bio(struct list_head *ws, struct compressed_bio *cb)
return ret;
}
+/*
+ * Map the destination for the next chunk of output.
+ *
+ * @decompressed is the offset of the next output byte inside the fully
+ * decompressed extent. If that offset has reached the current destination
+ * segment, its page-bounded bio_vec is kmapped so that zstd can write into the
+ * page cache directly, and the number of bytes writable there is returned.
+ * Otherwise @kaddr_ret is set to NULL and the number of bytes to skip before
+ * that segment is returned. This covers both the initial prefix and gaps in
+ * the destination bio.
+ */
+static u32 zstd_map_dest(struct compressed_bio *cb, u32 decompressed,
+ void **kaddr_ret)
+{
+ struct bio *orig_bio = &cb->orig_bbio->bio;
+ struct bio_vec bvec;
+ u32 bvec_offset;
+ u32 off;
+
+ bvec = bio_iter_iovec(orig_bio, orig_bio->bi_iter);
+ /*
+ * cb->start may underflow, but subtracting that value can still give us
+ * the correct offset inside the full decompressed extent.
+ */
+ bvec_offset = page_offset(bvec.bv_page) + bvec.bv_offset - cb->start;
+
+ if (decompressed < bvec_offset) {
+ *kaddr_ret = NULL;
+ return bvec_offset - decompressed;
+ }
+
+ off = decompressed - bvec_offset;
+ ASSERT(off < bvec.bv_len);
+ *kaddr_ret = bvec_kmap_local(&bvec) + off;
+ return bvec.bv_len - off;
+}
+
int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
{
struct btrfs_fs_info *fs_info = cb_to_fs_info(cb);
struct workspace *workspace = list_entry(ws, struct workspace, list);
+ struct bio *orig_bio = &cb->orig_bbio->bio;
struct folio_iter fi;
size_t srclen = bio_get_size(&cb->bbio.bio);
zstd_dstream *stream;
@@ -600,7 +638,6 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
const unsigned int min_folio_size = btrfs_min_folio_size(fs_info);
unsigned long folio_in_index = 0;
unsigned long total_folios_in = DIV_ROUND_UP(srclen, min_folio_size);
- unsigned long buf_start;
unsigned long total_out = 0;
bio_first_folio(&fi, &cb->bbio.bio, 0);
@@ -624,15 +661,26 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
workspace->in_buf.pos = 0;
workspace->in_buf.size = min_t(size_t, srclen, min_folio_size);
- workspace->out_buf.dst = workspace->buf;
- workspace->out_buf.pos = 0;
- workspace->out_buf.size = fs_info->sectorsize;
-
- while (1) {
+ while (orig_bio->bi_iter.bi_size) {
size_t ret2;
+ void *kaddr;
+ u32 dstlen;
+
+ dstlen = zstd_map_dest(cb, total_out, &kaddr);
+ if (kaddr) {
+ workspace->out_buf.dst = kaddr;
+ workspace->out_buf.size = dstlen;
+ } else {
+ workspace->out_buf.dst = workspace->buf;
+ workspace->out_buf.size = min_t(u32, dstlen,
+ fs_info->sectorsize);
+ }
+ workspace->out_buf.pos = 0;
ret2 = zstd_decompress_stream(stream, &workspace->out_buf,
&workspace->in_buf);
+ if (kaddr)
+ kunmap_local(kaddr);
if (unlikely(zstd_is_error(ret2))) {
struct btrfs_inode *inode = cb->bbio.inode;
@@ -643,14 +691,9 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
ret = -EIO;
goto done;
}
- buf_start = total_out;
total_out += workspace->out_buf.pos;
- workspace->out_buf.pos = 0;
-
- ret = btrfs_decompress_buf2page(workspace->out_buf.dst,
- total_out - buf_start, cb, buf_start);
- if (ret == 0)
- break;
+ if (kaddr)
+ bio_advance(orig_bio, workspace->out_buf.pos);
if (workspace->in_buf.pos >= srclen)
break;
@@ -0,0 +1,42 @@
diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c
--- a/fs/fuse/dir.c
+++ b/fs/fuse/dir.c
@@ -2280,6 +2280,9 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
*/
if ((is_truncate || !is_wb) &&
S_ISREG(inode->i_mode) && oldsize != outarg.attr.size) {
+ if (outarg.attr.size > oldsize)
+ truncate_pagecache_range(inode, oldsize,
+ outarg.attr.size - 1);
truncate_pagecache(inode, outarg.attr.size);
invalidate_inode_pages2(mapping);
}
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -1360,9 +1360,13 @@ static ssize_t fuse_perform_write(struct kiocb *iocb, struct iov_iter *ii)
struct fuse_conn *fc = get_fuse_conn(inode);
struct fuse_inode *fi = get_fuse_inode(inode);
loff_t pos = iocb->ki_pos;
+ loff_t old_size = i_size_read(inode);
int err = 0;
ssize_t res = 0;
+ if (pos > old_size)
+ truncate_pagecache_range(inode, old_size, pos - 1);
+
if (inode->i_size < pos + iov_iter_count(ii))
set_bit(FUSE_I_SIZE_UNSTABLE, &fi->state);
@@ -2908,6 +2912,11 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset,
/* we could have extended the file */
if (!(mode & FALLOC_FL_KEEP_SIZE)) {
+ loff_t oldsize = i_size_read(inode);
+
+ if (offset + length > oldsize)
+ truncate_pagecache_range(inode, oldsize,
+ offset + length - 1);
if (fuse_write_update_attr(inode, offset + length, length))
file_update_time(file);
}
@@ -0,0 +1,104 @@
diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -216,10 +216,13 @@ EXPORT_SYMBOL_GPL(fuse_req_hash);
/*
* A new request is available, wake fiq->waitq
*/
-static void fuse_dev_wake_and_unlock(struct fuse_iqueue *fiq)
+static void fuse_dev_wake_and_unlock(struct fuse_iqueue *fiq, bool sync)
__releases(fiq->lock)
{
- wake_up(&fiq->waitq);
+ if (sync)
+ wake_up_sync(&fiq->waitq);
+ else
+ wake_up(&fiq->waitq);
kill_fasync(&fiq->fasync, SIGIO, POLL_IN);
spin_unlock(&fiq->lock);
}
@@ -236,7 +239,7 @@ void fuse_dev_queue_forget(struct fuse_iqueue *fiq,
if (fiq->connected) {
fiq->forget_list_tail->next = forget;
fiq->forget_list_tail = forget;
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, false);
} else {
kfree(forget);
spin_unlock(&fiq->lock);
@@ -258,7 +261,7 @@ void fuse_dev_queue_interrupt(struct fuse_iqueue *fiq, struct fuse_req *req)
list_del_init(&req->intr_entry);
spin_unlock(&fiq->lock);
} else {
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, false);
}
} else {
spin_unlock(&fiq->lock);
@@ -288,11 +291,13 @@ EXPORT_SYMBOL_GPL(fuse_request_assign_unique);
static void fuse_dev_queue_req(struct fuse_iqueue *fiq, struct fuse_req *req)
{
+ bool sync = test_and_clear_bit(FR_SYNC_WAKEUP, &req->flags);
+
spin_lock(&fiq->lock);
if (fiq->connected) {
fuse_request_assign_unique_locked(fiq, req);
list_add_tail(&req->list, &fiq->pending);
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, sync);
} else {
spin_unlock(&fiq->lock);
req->out.h.error = -ENOTCONN;
@@ -755,6 +760,11 @@ static void __fuse_request_send(struct fuse_req *req)
/* acquire extra reference, since request is still needed after
fuse_request_end() */
__fuse_get_request(req);
+ /*
+ * This is a synchronous request: the caller will block waiting for
+ * the answer. Hint the scheduler via wake_up_sync().
+ */
+ set_bit(FR_SYNC_WAKEUP, &req->flags);
fuse_send_one(fiq, req);
request_wait_answer(req);
@@ -1809,7 +1819,7 @@ void fuse_chan_resend(struct fuse_chan *fch)
}
/* iq and pq requests are both oldest to newest */
list_splice(&to_queue, &fiq->pending);
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, false);
}
/* Look up request on processing list by unique ID */
diff --git a/fs/fuse/fuse_dev_i.h b/fs/fuse/fuse_dev_i.h
--- a/fs/fuse/fuse_dev_i.h
+++ b/fs/fuse/fuse_dev_i.h
@@ -38,6 +38,8 @@ struct fuse_iqueue;
* @FR_PRIVATE: request is on private list
* @FR_ASYNC: request is asynchronous
* @FR_URING: request is handled through fuse-io-uring
+ * @FR_SYNC_WAKEUP: use synchronous wakeup when queueing this request to
+ * give the scheduler a hint about the waker task
*/
enum fuse_req_flag {
FR_ISREPLY,
@@ -53,6 +55,7 @@ enum fuse_req_flag {
FR_PRIVATE,
FR_ASYNC,
FR_URING,
+ FR_SYNC_WAKEUP,
};
/**
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -1417,6 +1417,7 @@ static void process_init_reply(struct fuse_args *args, int error)
fm->sb->s_bdi->ra_pages =
min(fm->sb->s_bdi->ra_pages, ra_pages);
+ fm->sb->s_bdi->io_pages = fc->max_pages;
fc->minor = arg->minor;
fc->max_write = arg->minor < 5 ? 4096 : arg->max_write;
fc->max_write = max_t(unsigned, 4096, fc->max_write);
@@ -0,0 +1,38 @@
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -1219,8 +1219,7 @@ static ssize_t fuse_send_write_pages(struct fuse_io_args *ia,
struct file *file = iocb->ki_filp;
struct fuse_file *ff = file->private_data;
struct fuse_mount *fm = ff->fm;
- unsigned int offset, i;
- bool short_write;
+ unsigned int i;
int err;
for (i = 0; i < ap->num_folios; i++)
@@ -1235,24 +1234,9 @@ static ssize_t fuse_send_write_pages(struct fuse_io_args *ia,
if (!err && ia->write.out.size > count)
err = -EIO;
- short_write = ia->write.out.size < count;
- offset = ap->descs[0].offset;
- count = ia->write.out.size;
for (i = 0; i < ap->num_folios; i++) {
struct folio *folio = ap->folios[i];
- if (err) {
- folio_clear_uptodate(folio);
- } else {
- if (count >= folio_size(folio) - offset)
- count -= folio_size(folio) - offset;
- else {
- if (short_write)
- folio_clear_uptodate(folio);
- count = 0;
- }
- offset = 0;
- }
if (ia->write.folio_locked && (i == ap->num_folios - 1))
folio_unlock(folio);
folio_put(folio);
@@ -0,0 +1,13 @@
diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -409,7 +409,8 @@ void fuse_chan_max_background_set(struct fuse_chan *fch, unsigned int val)
fch->max_background = val;
fch->blocked = fch->num_background >= fch->max_background;
if (!fch->blocked)
- wake_up(&fch->blocked_waitq);
+ wake_up_nr(&fch->blocked_waitq,
+ fch->max_background - fch->num_background);
spin_unlock(&fch->bg_lock);
}
@@ -0,0 +1,123 @@
diff --git a/drivers/gpu/drm/drm_displayid_internal.h b/drivers/gpu/drm/drm_displayid_internal.h
index 4590d6a3d821..dbdc2053332d 100644
--- a/drivers/gpu/drm/drm_displayid_internal.h
+++ b/drivers/gpu/drm/drm_displayid_internal.h
@@ -67,6 +67,7 @@ struct drm_edid;
#define DATA_BLOCK_2_TILED_DISPLAY_TOPOLOGY 0x28
#define DATA_BLOCK_2_CONTAINER_ID 0x29
#define DATA_BLOCK_2_TYPE_10_FORMULA_TIMING 0x2a
+#define DATA_BLOCK_2_ADAPTIVE_SYNC 0x2b
#define DATA_BLOCK_2_VENDOR_SPECIFIC 0x7e
#define DATA_BLOCK_2_CTA_DISPLAY_ID 0x81
diff --git a/drivers/gpu/drm/drm_edid.c b/drivers/gpu/drm/drm_edid.c
index df3c25bac761..354d4ff1aff7 100644
--- a/drivers/gpu/drm/drm_edid.c
+++ b/drivers/gpu/drm/drm_edid.c
@@ -6598,6 +6598,49 @@ static void drm_get_monitor_range(struct drm_connector *connector,
info->monitor_range.min_vfreq, info->monitor_range.max_vfreq);
}
+#define DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE 6
+
+static bool drm_update_displayid_adaptive_sync_range(struct drm_connector *connector,
+ const struct displayid_block *block)
+{
+ struct drm_monitor_range_info *range = &connector->display_info.monitor_range;
+ const u8 *data = (const u8 *)(block + 1);
+ u16 best_min = 0, best_max = 0;
+ unsigned int best_span = 0;
+ int i;
+
+ if (block->rev != 0 || block->num_bytes < DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE ||
+ block->num_bytes % DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE)
+ return false;
+
+ for (i = 0; i < block->num_bytes; i += DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE) {
+ const u8 *desc = &data[i];
+ u16 min_vfreq = desc[2];
+ u16 max_vfreq = ((((u16)desc[4]) & 0x3) << 8) | desc[3];
+ unsigned int span;
+
+ max_vfreq += 1;
+ if (!min_vfreq || max_vfreq <= min_vfreq)
+ continue;
+
+ span = max_vfreq - min_vfreq;
+ if (span <= best_span)
+ continue;
+
+ best_min = min_vfreq;
+ best_max = max_vfreq;
+ best_span = span;
+ }
+
+ if (best_span) {
+ range->min_vfreq = best_min;
+ range->max_vfreq = best_max;
+ return true;
+ }
+
+ return false;
+}
+
static void drm_parse_vesa_mso_data(struct drm_connector *connector,
const struct displayid_block *block)
{
@@ -6720,26 +6763,43 @@ static void update_displayid_info(struct drm_connector *connector,
{
struct drm_display_info *info = &connector->display_info;
const struct displayid_block *block;
+ bool adaptive_sync_range = false;
+ bool displayid_base_logged = false;
struct displayid_iter iter;
displayid_iter_edid_begin(drm_edid, &iter);
displayid_iter_for_each(block, &iter) {
- drm_dbg_kms(connector->dev,
- "[CONNECTOR:%d:%s] DisplayID extension version 0x%02x, primary use 0x%02x\n",
- connector->base.id, connector->name,
- displayid_version(&iter),
- displayid_primary_use(&iter));
- if (displayid_version(&iter) == DISPLAY_ID_STRUCTURE_VER_20 &&
- (displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_VR ||
- displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_AR))
- info->non_desktop = true;
-
/*
- * We're only interested in the base section here, no need to
- * iterate further.
+ * Primary use is a DisplayID base section property, but later
+ * blocks may still carry useful metadata like adaptive sync ranges.
*/
- break;
+ if (!displayid_base_logged) {
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] DisplayID extension version 0x%02x, primary use 0x%02x\n",
+ connector->base.id, connector->name,
+ displayid_version(&iter),
+ displayid_primary_use(&iter));
+ if (displayid_version(&iter) == DISPLAY_ID_STRUCTURE_VER_20 &&
+ (displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_VR ||
+ displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_AR))
+ info->non_desktop = true;
+
+ displayid_base_logged = true;
+ }
+
+ if (!info->monitor_range.min_vfreq && !info->monitor_range.max_vfreq &&
+ block->tag == DATA_BLOCK_2_ADAPTIVE_SYNC)
+ adaptive_sync_range =
+ drm_update_displayid_adaptive_sync_range(connector, block);
}
+
+ if (adaptive_sync_range)
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] DisplayID adaptive sync refresh rate range is %d Hz - %d Hz\n",
+ connector->base.id, connector->name,
+ info->monitor_range.min_vfreq,
+ info->monitor_range.max_vfreq);
+
displayid_iter_end(&iter);
}
@@ -0,0 +1,601 @@
diff --git a/drivers/gpu/drm/nouveau/nouveau_ttm.c b/drivers/gpu/drm/nouveau/nouveau_ttm.c
--- a/drivers/gpu/drm/nouveau/nouveau_ttm.c
+++ b/drivers/gpu/drm/nouveau/nouveau_ttm.c
@@ -188,6 +188,11 @@ nouveau_ttm_init_vram(struct nouveau_drm *drm)
man->func = &nouveau_vram_manager;
+ man->cg = drmm_cgroup_register_region(drm->dev, "vram",
+ drm->gem.vram_available);
+ if (IS_ERR(man->cg))
+ return PTR_ERR(man->cg);
+
ttm_resource_manager_init(man, &drm->ttm.bdev,
drm->gem.vram_available >> PAGE_SHIFT);
ttm_set_driver_manager(&drm->ttm.bdev, TTM_PL_VRAM, man);
diff --git a/drivers/gpu/drm/ttm/ttm_bo.c b/drivers/gpu/drm/ttm/ttm_bo.c
--- a/drivers/gpu/drm/ttm/ttm_bo.c
+++ b/drivers/gpu/drm/ttm/ttm_bo.c
@@ -488,6 +488,118 @@ int ttm_bo_evict_first(struct ttm_device *bdev, struct ttm_resource_manager *man
return ret;
}
+struct ttm_bo_alloc_state {
+ /** @charge_pool: The memory pool the resource is charged to */
+ struct dmem_cgroup_pool_state *charge_pool;
+ /** @limit_pool: Which pool limit we should test against */
+ struct dmem_cgroup_pool_state *limit_pool;
+ /** @in_evict: Whether we are currently evicting buffers */
+ bool in_evict;
+ /** @may_try_low: If only unprotected BOs, i.e. BOs whose cgroup
+ * is exceeding its dmem low/min protection, should be considered for eviction
+ */
+ bool may_try_low;
+};
+
+/**
+ * ttm_bo_alloc_at_place - Attempt allocating a BO's backing store in a place
+ *
+ * @bo: The buffer to allocate the backing store of
+ * @place: The place to attempt allocation in
+ * @ctx: ttm_operation_ctx associated with this allocation
+ * @force_space: If we should evict buffers to force space
+ * @res: On allocation success, the resulting struct ttm_resource.
+ * @alloc_state: Object holding allocation state such as charged cgroups.
+ *
+ * Returns:
+ * -EBUSY: No space available, but allocation should be retried with ttm_bo_evict_alloc.
+ * -ENOSPC: No space available, allocation should not be retried.
+ * -ERESTARTSYS: An interruptible sleep was interrupted by a signal.
+ *
+ */
+static int ttm_bo_alloc_at_place(struct ttm_buffer_object *bo,
+ const struct ttm_place *place,
+ bool force_space,
+ struct ttm_resource **res,
+ struct ttm_bo_alloc_state *alloc_state)
+{
+ bool may_evict;
+ int ret;
+
+ may_evict = !alloc_state->in_evict && force_space &&
+ place->mem_type != TTM_PL_SYSTEM;
+ if (!alloc_state->charge_pool) {
+ ret = ttm_resource_try_charge(bo, place, &alloc_state->charge_pool,
+ force_space ? &alloc_state->limit_pool
+ : NULL);
+ if (ret) {
+ /*
+ * -EAGAIN means the charge failed, which we treat
+ * like an allocation failure. Therefore, return an
+ * error code indicating the allocation failed -
+ * either -EBUSY if the allocation should be
+ * retried with eviction, or -ENOSPC if there should
+ * be no second attempt.
+ */
+ if (!alloc_state->in_evict)
+ alloc_state->may_try_low = may_evict;
+ if (ret == -EAGAIN)
+ ret = may_evict ? -EBUSY : -ENOSPC;
+ return ret;
+ }
+ }
+
+ /*
+ * cgroup protection plays a special role in eviction.
+ * Conceptually, protection of memory via the dmem cgroup controller
+ * entitles the protected cgroup to use a certain amount of memory.
+ * There are two types of protection - the 'low' limit is a
+ * "best-effort" protection, whereas the 'min' limit provides a hard
+ * guarantee that memory within the cgroup's allowance will not be
+ * evicted under any circumstance.
+ *
+ * To faithfully model this concept in TTM, we also need to take cgroup
+ * protection into account when allocating. When allocation in one
+ * place fails, TTM will default to trying other places first before
+ * evicting.
+ * If the allocation is covered by dmem cgroup protection, however,
+ * this prevents the allocation from using the memory it is "entitled"
+ * to. To make sure unprotected allocations cannot push new protected
+ * allocations out of places they are "entitled" to use, we should
+ * evict buffers not covered by any cgroup protection, if this
+ * allocation is covered by cgroup protection.
+ *
+ * Buffers covered by 'min' protection are a special case - the 'min'
+ * limit is a stronger guarantee than 'low', and thus buffers protected
+ * by 'low' but not 'min' should also be considered for eviction.
+ * Buffers protected by 'min' will never be considered for eviction
+ * anyway, so the regular eviction path should be triggered here.
+ * Buffers protected by 'low' but not 'min' will take a special
+ * eviction path that only evicts buffers covered by neither 'low' or
+ * 'min' protections.
+ */
+ if (!alloc_state->in_evict) {
+ may_evict |= dmem_cgroup_below_min(NULL, alloc_state->charge_pool);
+ alloc_state->may_try_low = may_evict;
+
+ may_evict |= dmem_cgroup_below_low(NULL, alloc_state->charge_pool);
+ }
+
+ ret = ttm_resource_alloc(bo, place, res, alloc_state->charge_pool);
+ if (ret) {
+ if (ret == -ENOSPC && may_evict)
+ ret = -EBUSY;
+ return ret;
+ }
+
+ /*
+ * Ownership of charge_pool has been transferred to the TTM resource,
+ * don't make the caller think we still hold a reference to it.
+ */
+ alloc_state->charge_pool = NULL;
+ return 0;
+}
+
/**
* struct ttm_bo_evict_walk - Parameters for the evict walk.
*/
@@ -503,22 +615,61 @@ struct ttm_bo_evict_walk {
/** @evicted: Number of successful evictions. */
unsigned long evicted;
- /** @limit_pool: Which pool limit we should test against */
- struct dmem_cgroup_pool_state *limit_pool;
/** @try_low: Whether we should attempt to evict BO's with low watermark threshold */
bool try_low;
/** @hit_low: If we cannot evict a bo when @try_low is false (first pass) */
bool hit_low;
+
+ /** @alloc_state: State associated with the allocation attempt. */
+ struct ttm_bo_alloc_state *alloc_state;
};
static s64 ttm_bo_evict_cb(struct ttm_lru_walk *walk, struct ttm_buffer_object *bo)
{
struct ttm_bo_evict_walk *evict_walk =
container_of(walk, typeof(*evict_walk), walk);
+ struct dmem_cgroup_pool_state *limit_pool, *ancestor = NULL;
+ bool evict_valuable;
s64 lret;
- if (!dmem_cgroup_state_evict_valuable(evict_walk->limit_pool, bo->resource->css,
- evict_walk->try_low, &evict_walk->hit_low))
+ /*
+ * If may_try_low is not set, then we're trying to evict unprotected
+ * buffers in favor of a protected allocation for charge_pool. Explicitly skip
+ * buffers belonging to the same cgroup here - that cgroup is definitely protected,
+ * even though dmem_cgroup_state_evict_valuable would allow the eviction because a
+ * cgroup is always allowed to evict from itself even if it is protected.
+ */
+ if (!evict_walk->alloc_state->may_try_low &&
+ bo->resource->css == evict_walk->alloc_state->charge_pool)
+ return 0;
+
+ limit_pool = evict_walk->alloc_state->limit_pool;
+ /*
+ * If there is no explicit limit pool, find the root of the shared subtree between
+ * evictor and evictee. This is important so that recursive protection rules can
+ * apply properly: Recursive protection distributes cgroup protection afforded
+ * to a parent cgroup but not used explicitly by a child cgroup between all child
+ * cgroups (see docs of effective_protection in mm/page_counter.c). However, when
+ * direct siblings compete for memory, siblings that were explicitly protected
+ * should get prioritized over siblings that weren't. This only happens correctly
+ * when the root of the shared subtree is passed to
+ * dmem_cgroup_state_evict_valuable. Otherwise, the effective-protection
+ * calculation cannot distinguish direct siblings from unrelated subtrees and the
+ * calculated protection ends up wrong.
+ */
+ if (!limit_pool) {
+ ancestor = dmem_cgroup_get_common_ancestor(bo->resource->css,
+ evict_walk->alloc_state->charge_pool);
+ limit_pool = ancestor;
+ }
+
+ evict_valuable = dmem_cgroup_state_evict_valuable(limit_pool, bo->resource->css,
+ evict_walk->try_low,
+ &evict_walk->hit_low);
+ if (ancestor)
+ dmem_cgroup_pool_state_put(ancestor);
+
+ if (!evict_valuable)
return 0;
if (bo->pin_count || !bo->bdev->funcs->eviction_valuable(bo, evict_walk->place))
@@ -537,8 +688,10 @@ static s64 ttm_bo_evict_cb(struct ttm_lru_walk *walk, struct ttm_buffer_object *
evict_walk->evicted++;
if (evict_walk->res)
- lret = ttm_resource_alloc(evict_walk->evictor, evict_walk->place,
- evict_walk->res, NULL);
+ lret = ttm_bo_alloc_at_place(evict_walk->evictor,
+ evict_walk->place, false,
+ evict_walk->res,
+ evict_walk->alloc_state);
if (lret == 0)
return 1;
out:
@@ -560,7 +713,7 @@ static int ttm_bo_evict_alloc(struct ttm_device *bdev,
struct ttm_operation_ctx *ctx,
struct ww_acquire_ctx *ticket,
struct ttm_resource **res,
- struct dmem_cgroup_pool_state *limit_pool)
+ struct ttm_bo_alloc_state *state)
{
struct ttm_bo_evict_walk evict_walk = {
.walk = {
@@ -573,15 +726,21 @@ static int ttm_bo_evict_alloc(struct ttm_device *bdev,
.place = place,
.evictor = evictor,
.res = res,
- .limit_pool = limit_pool,
+ .alloc_state = state,
};
s64 lret;
+ state->in_evict = true;
+
evict_walk.walk.arg.trylock_only = true;
lret = ttm_lru_walk_for_evict(&evict_walk.walk, bdev, man, 1);
- /* One more attempt if we hit low limit? */
- if (!lret && evict_walk.hit_low) {
+ /* If we failed to find enough BOs to evict, but we skipped over
+ * some BOs because they were covered by dmem low protection, retry
+ * evicting these protected BOs too, except if we're told not to
+ * consider protected BOs at all.
+ */
+ if (!lret && evict_walk.hit_low && state->may_try_low) {
evict_walk.try_low = true;
lret = ttm_lru_walk_for_evict(&evict_walk.walk, bdev, man, 1);
}
@@ -602,11 +761,13 @@ static int ttm_bo_evict_alloc(struct ttm_device *bdev,
} while (!lret && evict_walk.evicted);
/* We hit the low limit? Try once more */
- if (!lret && evict_walk.hit_low && !evict_walk.try_low) {
+ if (!lret && evict_walk.hit_low && !evict_walk.try_low &&
+ state->may_try_low) {
evict_walk.try_low = true;
goto retry;
}
out:
+ state->in_evict = false;
if (lret < 0)
return lret;
if (lret == 0)
@@ -724,9 +885,8 @@ static int ttm_bo_alloc_resource(struct ttm_buffer_object *bo,
for (i = 0; i < placement->num_placement; ++i) {
const struct ttm_place *place = &placement->placement[i];
- struct dmem_cgroup_pool_state *limit_pool = NULL;
+ struct ttm_bo_alloc_state alloc_state = {};
struct ttm_resource_manager *man;
- bool may_evict;
man = ttm_manager_type(bdev, place->mem_type);
if (!man || !ttm_resource_manager_used(man))
@@ -736,25 +896,30 @@ static int ttm_bo_alloc_resource(struct ttm_buffer_object *bo,
TTM_PL_FLAG_FALLBACK))
continue;
- may_evict = (force_space && place->mem_type != TTM_PL_SYSTEM);
- ret = ttm_resource_alloc(bo, place, res, force_space ? &limit_pool : NULL);
- if (ret) {
- if (ret != -ENOSPC) {
- dmem_cgroup_pool_state_put(limit_pool);
- return ret;
- }
- if (!may_evict) {
- dmem_cgroup_pool_state_put(limit_pool);
- continue;
- }
+ ret = ttm_bo_alloc_at_place(bo, place, force_space, res,
+ &alloc_state);
+ if (ret == -ENOSPC) {
+ dmem_cgroup_uncharge(alloc_state.charge_pool, bo->base.size);
+ dmem_cgroup_pool_state_put(alloc_state.limit_pool);
+ continue;
+ } else if (ret == -EBUSY) {
ret = ttm_bo_evict_alloc(bdev, man, place, bo, ctx,
- ticket, res, limit_pool);
- dmem_cgroup_pool_state_put(limit_pool);
- if (ret == -EBUSY)
- continue;
- if (ret)
+ ticket, res, &alloc_state);
+
+ dmem_cgroup_pool_state_put(alloc_state.limit_pool);
+
+ if (ret) {
+ dmem_cgroup_uncharge(alloc_state.charge_pool,
+ bo->base.size);
+ if (ret == -EBUSY)
+ continue;
return ret;
+ }
+ } else if (ret) {
+ dmem_cgroup_uncharge(alloc_state.charge_pool, bo->base.size);
+ dmem_cgroup_pool_state_put(alloc_state.limit_pool);
+ return ret;
}
ret = ttm_bo_add_pipelined_eviction_fences(bo, man, ctx->no_wait_gpu);
diff --git a/drivers/gpu/drm/ttm/ttm_resource.c b/drivers/gpu/drm/ttm/ttm_resource.c
--- a/drivers/gpu/drm/ttm/ttm_resource.c
+++ b/drivers/gpu/drm/ttm/ttm_resource.c
@@ -386,33 +386,52 @@ void ttm_resource_fini(struct ttm_resource_manager *man,
}
EXPORT_SYMBOL(ttm_resource_fini);
+/**
+ * ttm_resource_try_charge - charge a resource manager's cgroup pool
+ * @bo: buffer for which an allocation should be charged
+ * @place: where the allocation is attempted to be placed
+ * @ret_pool: on charge success, the pool that was charged
+ * @ret_limit_pool: on charge failure, the pool responsible for the failure
+ *
+ * Should be used to charge cgroups before attempting resource allocation.
+ * When charging succeeds, the value of ret_pool should be passed to
+ * ttm_resource_alloc.
+ *
+ * Returns: 0 on charge success, negative errno on failure.
+ */
+int ttm_resource_try_charge(struct ttm_buffer_object *bo,
+ const struct ttm_place *place,
+ struct dmem_cgroup_pool_state **ret_pool,
+ struct dmem_cgroup_pool_state **ret_limit_pool)
+{
+ struct ttm_resource_manager *man =
+ ttm_manager_type(bo->bdev, place->mem_type);
+
+ if (!man->cg) {
+ *ret_pool = NULL;
+ if (ret_limit_pool)
+ *ret_limit_pool = NULL;
+ return 0;
+ }
+
+ return dmem_cgroup_try_charge(man->cg, bo->base.size, ret_pool,
+ ret_limit_pool);
+}
+
int ttm_resource_alloc(struct ttm_buffer_object *bo,
const struct ttm_place *place,
struct ttm_resource **res_ptr,
- struct dmem_cgroup_pool_state **ret_limit_pool)
+ struct dmem_cgroup_pool_state *charge_pool)
{
struct ttm_resource_manager *man =
ttm_manager_type(bo->bdev, place->mem_type);
- struct dmem_cgroup_pool_state *pool = NULL;
int ret;
- if (man->cg) {
- ret = dmem_cgroup_try_charge(man->cg, bo->base.size, &pool, ret_limit_pool);
- if (ret) {
- if (ret == -EAGAIN)
- ret = -ENOSPC;
- return ret;
- }
- }
-
ret = man->func->alloc(man, bo, place, res_ptr);
- if (ret) {
- if (pool)
- dmem_cgroup_uncharge(pool, bo->base.size);
+ if (ret)
return ret;
- }
- (*res_ptr)->css = pool;
+ (*res_ptr)->css = charge_pool;
spin_lock(&bo->bdev->lru_lock);
ttm_resource_add_bulk_move(*res_ptr, bo);
diff --git a/include/drm/ttm/ttm_resource.h b/include/drm/ttm/ttm_resource.h
--- a/include/drm/ttm/ttm_resource.h
+++ b/include/drm/ttm/ttm_resource.h
@@ -458,10 +458,14 @@ void ttm_resource_init(struct ttm_buffer_object *bo,
void ttm_resource_fini(struct ttm_resource_manager *man,
struct ttm_resource *res);
+int ttm_resource_try_charge(struct ttm_buffer_object *bo,
+ const struct ttm_place *place,
+ struct dmem_cgroup_pool_state **ret_pool,
+ struct dmem_cgroup_pool_state **ret_limit_pool);
int ttm_resource_alloc(struct ttm_buffer_object *bo,
const struct ttm_place *place,
struct ttm_resource **res,
- struct dmem_cgroup_pool_state **ret_limit_pool);
+ struct dmem_cgroup_pool_state *charge_pool);
void ttm_resource_free(struct ttm_buffer_object *bo, struct ttm_resource **res);
bool ttm_resource_intersects(struct ttm_device *bdev,
struct ttm_resource *res,
diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h
--- a/include/linux/cgroup.h
+++ b/include/linux/cgroup.h
@@ -623,6 +623,27 @@ static inline struct cgroup *cgroup_ancestor(struct cgroup *cgrp,
return cgrp->ancestors[ancestor_level];
}
+/**
+ * cgroup_common_ancestor - find common ancestor of two cgroups
+ * @a: first cgroup to find common ancestor of
+ * @b: second cgroup to find common ancestor of
+ *
+ * Find the first cgroup that is an ancestor of both @a and @b, if it exists
+ * and return a pointer to it. If such a cgroup doesn't exist, return NULL.
+ *
+ * This function is safe to call as long as both @a and @b are accessible.
+ */
+static inline struct cgroup *cgroup_common_ancestor(struct cgroup *a,
+ struct cgroup *b)
+{
+ int level;
+
+ for (level = min(a->level, b->level); level >= 0; level--)
+ if (a->ancestors[level] == b->ancestors[level])
+ return a->ancestors[level];
+ return NULL;
+}
+
/**
* task_under_cgroup_hierarchy - test task's membership of cgroup ancestry
* @task: the task to be tested
diff --git a/include/linux/cgroup_dmem.h b/include/linux/cgroup_dmem.h
--- a/include/linux/cgroup_dmem.h
+++ b/include/linux/cgroup_dmem.h
@@ -24,6 +24,12 @@ void dmem_cgroup_uncharge(struct dmem_cgroup_pool_state *pool, u64 size);
bool dmem_cgroup_state_evict_valuable(struct dmem_cgroup_pool_state *limit_pool,
struct dmem_cgroup_pool_state *test_pool,
bool ignore_low, bool *ret_hit_low);
+bool dmem_cgroup_below_min(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test);
+bool dmem_cgroup_below_low(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test);
+struct dmem_cgroup_pool_state *dmem_cgroup_get_common_ancestor(struct dmem_cgroup_pool_state *a,
+ struct dmem_cgroup_pool_state *b);
void dmem_cgroup_pool_state_put(struct dmem_cgroup_pool_state *pool);
#else
@@ -59,6 +65,25 @@ bool dmem_cgroup_state_evict_valuable(struct dmem_cgroup_pool_state *limit_pool,
return true;
}
+static inline bool dmem_cgroup_below_min(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ return false;
+}
+
+static inline bool dmem_cgroup_below_low(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ return false;
+}
+
+static inline
+struct dmem_cgroup_pool_state *dmem_cgroup_get_common_ancestor(struct dmem_cgroup_pool_state *a,
+ struct dmem_cgroup_pool_state *b)
+{
+ return NULL;
+}
+
static inline void dmem_cgroup_pool_state_put(struct dmem_cgroup_pool_state *pool)
{ }
diff --git a/kernel/cgroup/dmem.c b/kernel/cgroup/dmem.c
--- a/kernel/cgroup/dmem.c
+++ b/kernel/cgroup/dmem.c
@@ -695,6 +695,110 @@ int dmem_cgroup_try_charge(struct dmem_cgroup_region *region, u64 size,
}
EXPORT_SYMBOL_GPL(dmem_cgroup_try_charge);
+/**
+ * dmem_cgroup_below_min() - Tests whether current usage is within min limit.
+ *
+ * @root: Root of the subtree to calculate protection for, or NULL to calculate global protection.
+ * @test: The pool to test the usage/min limit of.
+ *
+ * Return: true if usage is below min and the cgroup is protected, false otherwise.
+ */
+bool dmem_cgroup_below_min(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ if (root == test || !pool_parent(test))
+ return false;
+
+ if (!root) {
+ for (root = test; pool_parent(root); root = pool_parent(root))
+ {}
+ }
+
+ /*
+ * In mem_cgroup_below_min(), the memcg pendant, this call is missing.
+ * mem_cgroup_below_min() gets called during traversal of the cgroup tree, where
+ * protection is already calculated as part of the traversal. dmem cgroup eviction
+ * does not traverse the cgroup tree, so we need to recalculate effective protection
+ * here.
+ */
+ dmem_cgroup_calculate_protection(root, test);
+ return page_counter_read(&test->cnt) <= READ_ONCE(test->cnt.emin);
+}
+EXPORT_SYMBOL_GPL(dmem_cgroup_below_min);
+
+/**
+ * dmem_cgroup_below_low() - Tests whether current usage is within low limit.
+ *
+ * @root: Root of the subtree to calculate protection for, or NULL to calculate global protection.
+ * @test: The pool to test the usage/low limit of.
+ *
+ * Return: true if usage is below low and the cgroup is protected, false otherwise.
+ */
+bool dmem_cgroup_below_low(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ if (root == test || !pool_parent(test))
+ return false;
+
+ if (!root) {
+ for (root = test; pool_parent(root); root = pool_parent(root))
+ {}
+ }
+
+ /*
+ * In mem_cgroup_below_low(), the memcg pendant, this call is missing.
+ * mem_cgroup_below_low() gets called during traversal of the cgroup tree, where
+ * protection is already calculated as part of the traversal. dmem cgroup eviction
+ * does not traverse the cgroup tree, so we need to recalculate effective protection
+ * here.
+ */
+ dmem_cgroup_calculate_protection(root, test);
+ return page_counter_read(&test->cnt) <= READ_ONCE(test->cnt.elow);
+}
+EXPORT_SYMBOL_GPL(dmem_cgroup_below_low);
+
+/**
+ * dmem_cgroup_get_common_ancestor(): Find the first common ancestor of two pools.
+ * @a: First pool to find the common ancestor of.
+ * @b: First pool to find the common ancestor of.
+ *
+ * Return: The first pool that is a parent of both @a and @b, or NULL if either @a or @b are NULL,
+ * or if such a pool does not exist. A reference to the returned pool is grabbed and must be
+ * released by the caller when it is done using the pool.
+ */
+struct dmem_cgroup_pool_state *dmem_cgroup_get_common_ancestor(struct dmem_cgroup_pool_state *a,
+ struct dmem_cgroup_pool_state *b)
+{
+ struct cgroup *ancestor_cgroup;
+ struct cgroup_subsys_state *ancestor_css;
+ struct dmemcg_state *ancestor_dmemcs = NULL;
+ struct dmem_cgroup_pool_state *pool = NULL;
+
+ if (!a || !b)
+ return NULL;
+
+ ancestor_cgroup = cgroup_common_ancestor(a->cs->css.cgroup, b->cs->css.cgroup);
+ if (!ancestor_cgroup)
+ return NULL;
+
+ rcu_read_lock();
+ ancestor_css = cgroup_e_css(ancestor_cgroup, &dmem_cgrp_subsys);
+ if (css_tryget(ancestor_css))
+ ancestor_dmemcs = css_to_dmemcs(ancestor_css);
+ rcu_read_unlock();
+
+ if (ancestor_dmemcs) {
+ pool = get_cg_pool_unlocked(css_to_dmemcs(ancestor_css),
+ a->region);
+ if (WARN_ON(IS_ERR(pool))) {
+ pool = NULL;
+ css_put(ancestor_css);
+ }
+ }
+ return pool;
+}
+EXPORT_SYMBOL_GPL(dmem_cgroup_get_common_ancestor);
+
static int dmem_cgroup_region_capacity_show(struct seq_file *sf, void *v)
{
struct dmem_cgroup_region *region;
@@ -0,0 +1,13 @@
diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c
index c6963ea420cc..c38d613f15d6 100644
--- a/drivers/gpu/drm/i915/display/intel_alpm.c
+++ b/drivers/gpu/drm/i915/display/intel_alpm.c
@@ -399,7 +399,7 @@ static void lnl_alpm_configure(struct intel_dp *intel_dp,
ALPM_CTL_AUX_LESS_SLEEP_HOLD_TIME_50_SYMBOLS |
ALPM_CTL_AUX_LESS_WAKE_TIME(crtc_state->alpm_state.aux_less_wake_lines);
- if (intel_dp->as_sdp_supported) {
+ if (intel_dp->as_sdp_supported && crtc_state->has_panel_replay) {
u32 pr_alpm_ctl = get_pr_alpm_as_sdp_transmission_time(crtc_state);
if (crtc_state->link_off_after_as_sdp_when_pr_active)
@@ -0,0 +1,196 @@
diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h
index c21e0c0ef0b1..e62788328bfd 100644
--- a/drivers/gpu/drm/i915/display/intel_display_types.h
+++ b/drivers/gpu/drm/i915/display/intel_display_types.h
@@ -1780,6 +1780,7 @@ struct intel_psr {
u32 dc3co_exitline;
u32 dc3co_exit_delay;
struct delayed_work dc3co_work;
+ struct delayed_work panel_replay_reenable_work;
u8 entry_setup_frames;
u8 io_wake_lines;
diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c
index beaa1d62613d..644ab9b1bee9 100644
--- a/drivers/gpu/drm/i915/display/intel_psr.c
+++ b/drivers/gpu/drm/i915/display/intel_psr.c
@@ -2475,6 +2475,7 @@ void intel_psr_disable(struct intel_dp *intel_dp,
mutex_unlock(&intel_dp->psr.lock);
cancel_work_sync(&intel_dp->psr.work);
cancel_delayed_work_sync(&intel_dp->psr.dc3co_work);
+ cancel_delayed_work_sync(&intel_dp->psr.panel_replay_reenable_work);
}
/**
@@ -2506,6 +2507,7 @@ void intel_psr_pause(struct intel_dp *intel_dp)
cancel_work_sync(&psr->work);
cancel_delayed_work_sync(&psr->dc3co_work);
+ cancel_delayed_work_sync(&psr->panel_replay_reenable_work);
}
/**
@@ -3539,6 +3541,17 @@ static void intel_psr_handle_irq(struct intel_dp *intel_dp)
drm_dp_dpcd_writeb(&intel_dp->aux, DP_SET_POWER, DP_SET_POWER_D0);
}
+#define PANEL_REPLAY_REENABLE_DELAY_MS 50
+
+static bool panel_replay_alpm_cursor_lag_workaround_enabled(struct intel_dp *intel_dp)
+{
+ return intel_dp_is_edp(intel_dp) &&
+ intel_dp->psr.panel_replay_enabled &&
+ intel_dp->psr.sel_update_enabled &&
+ intel_has_dpcd_quirk(intel_dp,
+ QUIRK_PANEL_REPLAY_ALPM_CURSOR_LAG);
+}
+
static void intel_psr_work(struct work_struct *work)
{
struct intel_dp *intel_dp =
@@ -3557,6 +3570,11 @@ static void intel_psr_work(struct work_struct *work)
if (intel_dp->psr.pause_counter)
goto unlock;
+ /* The dedicated delayed work owns re-entry while the workaround is armed. */
+ if (panel_replay_alpm_cursor_lag_workaround_enabled(intel_dp) &&
+ delayed_work_pending(&intel_dp->psr.panel_replay_reenable_work))
+ goto unlock;
+
/*
* We have to make sure PSR is ready for re-enable
* otherwise it keeps disabled until next full enable/disable cycle.
@@ -3566,6 +3584,10 @@ static void intel_psr_work(struct work_struct *work)
if (!__psr_wait_for_idle_locked(intel_dp))
goto unlock;
+ if (panel_replay_alpm_cursor_lag_workaround_enabled(intel_dp) &&
+ delayed_work_pending(&intel_dp->psr.panel_replay_reenable_work))
+ goto unlock;
+
/*
* The delayed work can race with an invalidate hence we need to
* recheck. Since psr_flush first clears this and then reschedules we
@@ -3579,6 +3601,33 @@ static void intel_psr_work(struct work_struct *work)
mutex_unlock(&intel_dp->psr.lock);
}
+static void panel_replay_reenable_work(struct work_struct *work)
+{
+ struct intel_dp *intel_dp =
+ container_of(work, typeof(*intel_dp),
+ psr.panel_replay_reenable_work.work);
+
+ mutex_lock(&intel_dp->psr.lock);
+
+ if (!intel_dp->psr.enabled ||
+ !panel_replay_alpm_cursor_lag_workaround_enabled(intel_dp) ||
+ intel_dp->psr.pause_counter ||
+ READ_ONCE(intel_dp->psr.irq_aux_error))
+ goto unlock;
+
+ if (!__psr_wait_for_idle_locked(intel_dp))
+ goto unlock;
+
+ /* Recheck for activity or a newer deadline after the unlocked wait. */
+ if (delayed_work_pending(&intel_dp->psr.panel_replay_reenable_work) ||
+ intel_dp->psr.busy_frontbuffer_bits || intel_dp->psr.active)
+ goto unlock;
+
+ intel_psr_activate(intel_dp);
+unlock:
+ mutex_unlock(&intel_dp->psr.lock);
+}
+
static void intel_psr_configure_full_frame_update(struct intel_dp *intel_dp)
{
struct intel_display *display = to_intel_display(intel_dp);
@@ -3650,8 +3699,11 @@ void intel_psr_invalidate(struct intel_display *display,
INTEL_FRONTBUFFER_ALL_MASK(intel_dp->psr.pipe);
intel_dp->psr.busy_frontbuffer_bits |= pipe_frontbuffer_bits;
- if (pipe_frontbuffer_bits)
+ if (pipe_frontbuffer_bits) {
+ if (panel_replay_alpm_cursor_lag_workaround_enabled(intel_dp))
+ cancel_delayed_work(&intel_dp->psr.panel_replay_reenable_work);
_psr_invalidate_handle(intel_dp);
+ }
mutex_unlock(&intel_dp->psr.lock);
}
@@ -3767,6 +3819,15 @@ void intel_psr_flush(struct intel_display *display,
if (intel_dp->psr.pause_counter)
goto unlock;
+ if (pipe_frontbuffer_bits &&
+ panel_replay_alpm_cursor_lag_workaround_enabled(intel_dp)) {
+ intel_psr_exit(intel_dp);
+ mod_delayed_work(display->wq.unordered,
+ &intel_dp->psr.panel_replay_reenable_work,
+ msecs_to_jiffies(PANEL_REPLAY_REENABLE_DELAY_MS));
+ goto unlock;
+ }
+
if (origin == ORIGIN_FLIP ||
(origin == ORIGIN_CURSOR_UPDATE &&
!intel_dp->psr.psr2_sel_fetch_enabled)) {
@@ -3830,6 +3891,8 @@ void intel_psr_init(struct intel_dp *intel_dp)
INIT_WORK(&intel_dp->psr.work, intel_psr_work);
INIT_DELAYED_WORK(&intel_dp->psr.dc3co_work, tgl_dc3co_disable_work);
+ INIT_DELAYED_WORK(&intel_dp->psr.panel_replay_reenable_work,
+ panel_replay_reenable_work);
mutex_init(&intel_dp->psr.lock);
}
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c
index 33245f44c0d5..dc6576016400 100644
--- a/drivers/gpu/drm/i915/display/intel_quirks.c
+++ b/drivers/gpu/drm/i915/display/intel_quirks.c
@@ -86,12 +86,13 @@ static void quirk_edp_limit_rate_hbr2(struct intel_display *display)
drm_info(display->drm, "Applying eDP Limit rate to HBR2 quirk\n");
}
-static void quirk_disable_edp_panel_replay(struct intel_dp *intel_dp)
+static void quirk_panel_replay_alpm_cursor_lag(struct intel_dp *intel_dp)
{
struct intel_display *display = to_intel_display(intel_dp);
- intel_set_dpcd_quirk(intel_dp, QUIRK_DISABLE_EDP_PANEL_REPLAY);
- drm_info(display->drm, "Applying disable Panel Replay quirk\n");
+ intel_set_dpcd_quirk(intel_dp, QUIRK_PANEL_REPLAY_ALPM_CURSOR_LAG);
+ drm_info(display->drm,
+ "Applying Panel Replay ALPM cursor lag workaround\n");
}
static void quirk_disable_psr2(struct intel_display *display)
@@ -276,7 +277,7 @@ static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
.subsystem_vendor = 0x1028,
.subsystem_device = 0x0db9,
.sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
- .hook = quirk_disable_edp_panel_replay,
+ .hook = quirk_panel_replay_alpm_cursor_lag,
},
/* Dell XPS 16 DA16260 */
{
@@ -284,7 +285,7 @@ static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
.subsystem_vendor = 0x1028,
.subsystem_device = 0x0dba,
.sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
- .hook = quirk_disable_edp_panel_replay,
+ .hook = quirk_panel_replay_alpm_cursor_lag,
},
};
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.h b/drivers/gpu/drm/i915/display/intel_quirks.h
index 970a4fe52faf..50ef333b87a9 100644
--- a/drivers/gpu/drm/i915/display/intel_quirks.h
+++ b/drivers/gpu/drm/i915/display/intel_quirks.h
@@ -23,6 +23,7 @@ enum intel_quirk_id {
QUIRK_EDP_LIMIT_RATE_HBR2,
QUIRK_DISABLE_EDP_PANEL_REPLAY,
QUIRK_DISABLE_PSR2,
+ QUIRK_PANEL_REPLAY_ALPM_CURSOR_LAG,
};
void intel_init_quirks(struct intel_display *display);
@@ -0,0 +1,20 @@
diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c
index beaa1d626..5fdf4594c 100644
--- a/drivers/gpu/drm/i915/display/intel_psr.c
+++ b/drivers/gpu/drm/i915/display/intel_psr.c
@@ -714,8 +714,14 @@ static void _psr_init_dpcd(struct intel_dp *intel_dp, struct intel_connector *co
* To support PSR version 02h and PSR version 03h without
* Y-coordinate requirement panels we would need to enable
* GTC first.
+ *
+ * Early Transport (version 04h) implies Y-coordinate
+ * support. Accept it without explicit Y-coordinate
+ * requirement bit.
*/
- connector->dp.psr_caps.su_support = y_req &&
+ connector->dp.psr_caps.su_support =
+ (y_req || connector->dp.psr_caps.dpcd[0] ==
+ DP_PSR2_WITH_Y_COORD_ET_SUPPORTED) &&
intel_alpm_aux_wake_supported(intel_dp);
drm_dbg_kms(display->drm, "PSR2 %ssupported\n",
connector->dp.psr_caps.su_support ? "" : "not ");
@@ -0,0 +1,43 @@
diff --git a/drivers/gpu/drm/xe/display/xe_display_bo.c b/drivers/gpu/drm/xe/display/xe_display_bo.c
--- a/drivers/gpu/drm/xe/display/xe_display_bo.c
+++ b/drivers/gpu/drm/xe/display/xe_display_bo.c
@@ -147,33 +147,12 @@ static struct drm_gem_object *xe_display_bo_fbdev_create(struct drm_device *drm,
struct xe_device *xe = to_xe_device(drm);
struct xe_bo *obj;
- obj = ERR_PTR(-ENODEV);
-
- if (xe_display_bo_fbdev_prefer_stolen(xe, size)) {
- obj = xe_bo_create_pin_map_novm(xe, xe_device_get_root_tile(xe),
- size,
- ttm_bo_type_kernel,
- XE_BO_FLAG_FORCE_WC |
- XE_BO_FLAG_STOLEN |
- XE_BO_FLAG_GGTT,
- false);
- if (!IS_ERR(obj))
- drm_info(&xe->drm, "Allocated fbdev into stolen\n");
- else
- drm_info(&xe->drm, "Allocated fbdev into stolen failed: %li\n", PTR_ERR(obj));
- } else {
- drm_info(&xe->drm, "Allocating fbdev: Stolen memory not preferred.\n");
- }
-
- if (IS_ERR(obj)) {
- obj = xe_bo_create_pin_map_novm(xe, xe_device_get_root_tile(xe), size,
- ttm_bo_type_kernel,
- XE_BO_FLAG_FORCE_WC |
- XE_BO_FLAG_VRAM_IF_DGFX(xe_device_get_root_tile(xe)) |
- XE_BO_FLAG_GGTT,
- false);
- }
-
+ obj = xe_bo_create_pin_map_novm(xe, xe_device_get_root_tile(xe), size,
+ ttm_bo_type_kernel,
+ XE_BO_FLAG_FORCE_WC |
+ XE_BO_FLAG_VRAM_IF_DGFX(xe_device_get_root_tile(xe)) |
+ XE_BO_FLAG_GGTT,
+ false);
if (IS_ERR(obj)) {
drm_err(&xe->drm, "failed to allocate framebuffer (%pe)\n", obj);
return ERR_PTR(-ENOMEM);
@@ -0,0 +1,23 @@
diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c
--- a/drivers/gpu/drm/i915/display/intel_display.c
+++ b/drivers/gpu/drm/i915/display/intel_display.c
@@ -2452,6 +2452,19 @@ static int intel_crtc_set_context_latency(struct intel_crtc_state *crtc_state)
set_context_latency = max(set_context_latency,
intel_psr_min_set_context_latency(crtc_state));
+ /*
+ * From PTL onwards, the set context latency can be in the vactive
+ * region, letting the safe window start some lines before the vblank
+ * start. With modes that have a smaller vblank region, the computed
+ * guardband is clamped to the vblank length, making the undelayed and
+ * delayed vblank coincide. If the SCL is also 0, the 'safe window'
+ * becomes effectively 0, and the DSB configured to wait for it gets
+ * stalled, since the hardware never signals the safe window. Keep the
+ * set context latency at a minimum of 1 to avoid this.
+ */
+ if (DISPLAY_VER(display) >= 30)
+ set_context_latency = max(1, set_context_latency);
+
return set_context_latency;
}
+112
View File
@@ -0,0 +1,112 @@
diff --git a/drivers/gpu/drm/i915/display/intel_fbc.c b/drivers/gpu/drm/i915/display/intel_fbc.c
--- a/drivers/gpu/drm/i915/display/intel_fbc.c
+++ b/drivers/gpu/drm/i915/display/intel_fbc.c
@@ -59,6 +59,7 @@
#include "intel_fbc_regs.h"
#include "intel_frontbuffer.h"
#include "intel_parent.h"
+#include "skl_universal_plane.h"
#define for_each_fbc_id(__display, __fbc_id) \
for ((__fbc_id) = INTEL_FBC_A; (__fbc_id) < I915_MAX_FBCS; (__fbc_id)++) \
@@ -1305,6 +1306,9 @@ static bool intel_fbc_surface_size_ok(const struct intel_plane_state *plane_stat
struct intel_display *display = to_intel_display(plane_state);
unsigned int effective_w, effective_h, max_w, max_h;
+ if (DISPLAY_VER(display) >= 20)
+ return true;
+
intel_fbc_max_surface_size(display, &max_w, &max_h);
effective_w = plane_state->view.color_plane[0].x +
@@ -1315,10 +1319,18 @@ static bool intel_fbc_surface_size_ok(const struct intel_plane_state *plane_stat
return effective_w <= max_w && effective_h <= max_h;
}
-static void intel_fbc_max_plane_size(struct intel_display *display,
+static void intel_fbc_max_plane_size(const struct intel_plane_state *plane_state,
unsigned int *w, unsigned int *h)
{
- if (DISPLAY_VER(display) >= 10) {
+ struct intel_display *display = to_intel_display(plane_state);
+ struct intel_plane *plane = to_intel_plane(plane_state->uapi.plane);
+ const struct drm_framebuffer *fb = plane_state->hw.fb;
+ unsigned int rotation = plane_state->hw.rotation;
+
+ if (DISPLAY_VER(display) >= 20) {
+ *w = intel_plane_max_width(plane, fb, 0, rotation);
+ *h = 4096;
+ } else if (DISPLAY_VER(display) >= 10) {
*w = 5120;
*h = 4096;
} else if (DISPLAY_VER(display) >= 8 || display->platform.haswell) {
@@ -1335,10 +1347,9 @@ static void intel_fbc_max_plane_size(struct intel_display *display,
static bool intel_fbc_plane_size_valid(const struct intel_plane_state *plane_state)
{
- struct intel_display *display = to_intel_display(plane_state);
unsigned int w, h, max_w, max_h;
- intel_fbc_max_plane_size(display, &max_w, &max_h);
+ intel_fbc_max_plane_size(plane_state, &max_w, &max_h);
w = drm_rect_width(&plane_state->uapi.src) >> 16;
h = drm_rect_height(&plane_state->uapi.src) >> 16;
diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c
--- a/drivers/gpu/drm/i915/display/skl_universal_plane.c
+++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c
@@ -1937,10 +1937,10 @@ static int intel_plane_min_height(struct intel_plane *plane,
return 1;
}
-static int intel_plane_max_width(struct intel_plane *plane,
- const struct drm_framebuffer *fb,
- int color_plane,
- unsigned int rotation)
+int intel_plane_max_width(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation)
{
if (plane->max_width)
return plane->max_width(fb, color_plane, rotation);
@@ -1948,10 +1948,10 @@ static int intel_plane_max_width(struct intel_plane *plane,
return INT_MAX;
}
-static int intel_plane_max_height(struct intel_plane *plane,
- const struct drm_framebuffer *fb,
- int color_plane,
- unsigned int rotation)
+int intel_plane_max_height(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation)
{
if (plane->max_height)
return plane->max_height(fb, color_plane, rotation);
diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.h b/drivers/gpu/drm/i915/display/skl_universal_plane.h
--- a/drivers/gpu/drm/i915/display/skl_universal_plane.h
+++ b/drivers/gpu/drm/i915/display/skl_universal_plane.h
@@ -8,6 +8,7 @@
#include <linux/types.h>
+struct drm_framebuffer;
struct intel_crtc;
struct intel_display;
struct intel_initial_plane_config;
@@ -42,5 +43,13 @@ bool icl_is_hdr_plane(struct intel_display *display, enum plane_id plane_id);
u32 skl_plane_aux_dist(const struct intel_plane_state *plane_state,
int color_plane);
+int intel_plane_max_width(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation);
+int intel_plane_max_height(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation);
#endif
@@ -0,0 +1,67 @@
diff --git a/drivers/gpu/drm/i915/display/intel_bw.c b/drivers/gpu/drm/i915/display/intel_bw.c
--- a/drivers/gpu/drm/i915/display/intel_bw.c
+++ b/drivers/gpu/drm/i915/display/intel_bw.c
@@ -52,6 +52,8 @@
#define DEPROGBWPCLIMIT 60
+#define PEAK_BW_THRESHOLD 20000
+
struct intel_psf_gv_point {
u8 clk; /* clock in multiples of 16.6666 MHz */
};
@@ -587,6 +589,33 @@
return num_channels * (channel_width / 8) * dclk;
}
+static void xe3_add_peakbw_threshold(struct intel_display *display)
+{
+ u8 qgv_points = display->bw.max[0].num_qgv_points;
+
+ if (!HAS_PEAK_BW_THRESHOLD(display))
+ return;
+
+ if (qgv_points >= I915_NUM_QGV_POINTS) {
+ drm_dbg_kms(display->drm, "QGV points maxed out; skipping peak bandwidth threshold.\n");
+ return;
+ }
+
+ if (qgv_points <= 1)
+ return;
+
+ for (int i = 0; i < ARRAY_SIZE(display->bw.max); i++) {
+ struct intel_bw_info *bi = &display->bw.max[i];
+
+ bi->num_qgv_points++;
+ bi->peakbw[qgv_points] = PEAK_BW_THRESHOLD;
+ bi->deratedbw[qgv_points] = PEAK_BW_THRESHOLD;
+ }
+
+ drm_dbg_kms(display->drm, "An extra QGV point %d added for Peak bw threshod of %d\n",
+ qgv_points, PEAK_BW_THRESHOLD);
+}
+
static int tgl_get_bw_info(struct intel_display *display,
const struct dram_info *dram_info,
const struct intel_soc_bw_params *soc_bw_params,
@@ -684,6 +713,9 @@
}
}
+ /* For xe3 cases add an extra qgv point for Peak bw threshold */
+ xe3_add_peakbw_threshold(display);
+
/*
* In case if SAGV is disabled in BIOS, we always get 1
* SAGV point, but we can't send PCode commands to restrict it
diff --git a/drivers/gpu/drm/i915/display/intel_display_device.h b/drivers/gpu/drm/i915/display/intel_display_device.h
--- a/drivers/gpu/drm/i915/display/intel_display_device.h
+++ b/drivers/gpu/drm/i915/display/intel_display_device.h
@@ -191,6 +191,7 @@
#define HAS_MBUS_JOINING(__display) ((__display)->platform.alderlake_p || DISPLAY_VER(__display) >= 14)
#define HAS_MSO(__display) (DISPLAY_VER(__display) >= 12)
#define HAS_OVERLAY(__display) (DISPLAY_INFO(__display)->has_overlay)
+#define HAS_PEAK_BW_THRESHOLD(__display) (DISPLAY_VER(__display) >= 30)
#define HAS_PIPEDMC(__display) (DISPLAY_VER(__display) >= 12)
#define HAS_PIXEL_NORMALIZER(__display) (DISPLAY_VER(__display) >= 35)
#define HAS_PSR(__display) (DISPLAY_INFO(__display)->has_psr)
@@ -0,0 +1,456 @@
diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
--- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
+++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
@@ -104,6 +104,7 @@
#include "ivsrcid/dcn/irqsrcs_dcn_1_0.h"
#include "modules/inc/mod_freesync.h"
+#include "modules/inc/mod_info_packet.h"
#include "modules/inc/mod_power.h"
#include "modules/power/power_helpers.h"
@@ -10068,6 +10069,9 @@ static void update_freesync_state_on_stream(
&vrr_infopacket,
pack_sdp_v1_3);
+ if (new_stream->sink->sink_signal == SIGNAL_TYPE_HDMI_FRL)
+ mod_build_infopacket_vtem(new_stream, &vrr_params, 0, &vrr_infopacket);
+
new_crtc_state->freesync_vrr_info_changed |=
(memcmp(&new_crtc_state->vrr_infopacket,
&vrr_infopacket,
@@ -10079,6 +10083,36 @@ static void update_freesync_state_on_stream(
new_stream->vrr_infopacket = vrr_infopacket;
new_stream->allow_freesync = mod_freesync_get_freesync_enabled(&vrr_params);
+ /*
+ * HDMI ALLM: when Gaming-VRR is active (VRR_EN=1) and the sink
+ * advertises ALLM in the SCDS, the Source shall transmit the HF-VSIF
+ * with ALLM_Mode=1 (HDMI 2.1 Section 7.6.6).
+ */
+ if (new_stream->signal == SIGNAL_TYPE_HDMI_TYPE_A ||
+ new_stream->signal == SIGNAL_TYPE_HDMI_FRL) {
+ struct dc_info_packet hf_vsif = {0};
+ bool sink_allm = aconn && aconn->base.display_info.hdmi.allm;
+ bool allm = sink_allm &&
+ (vrr_params.state == VRR_STATE_ACTIVE_VARIABLE ||
+ vrr_params.state == VRR_STATE_ACTIVE_FIXED);
+ bool allm_changed;
+
+ if (allm)
+ mod_build_hf_vsif_infopacket(new_stream, &hf_vsif, allm, allm);
+
+ allm_changed = memcmp(&new_stream->hfvsif_infopacket, &hf_vsif,
+ sizeof(hf_vsif)) != 0;
+ new_crtc_state->freesync_vrr_info_changed |= allm_changed;
+ new_stream->hfvsif_infopacket = hf_vsif;
+
+ if (allm_changed)
+ drm_dbg_driver(adev_to_drm(adev),
+ "ALLM: flip on crtc=%u: sink_allm=%d vrr_state=%d -> ALLM_Mode=%d\n",
+ new_crtc_state->base.crtc->base.id,
+ sink_allm,
+ vrr_params.state, allm);
+ }
+
if (new_crtc_state->freesync_vrr_info_changed)
drm_dbg_kms(adev_to_drm(adev), "VRR packet update: crtc=%u enabled=%d state=%d",
new_crtc_state->base.crtc->base.id,
@@ -10645,9 +10679,12 @@ static void amdgpu_dm_commit_planes(struct drm_atomic_commit *state,
}
if (acrtc_state->stream) {
- if (acrtc_state->freesync_vrr_info_changed)
+ if (acrtc_state->freesync_vrr_info_changed) {
bundle->stream_update.vrr_infopacket =
&acrtc_state->stream->vrr_infopacket;
+ bundle->stream_update.hfvsif_infopacket =
+ &acrtc_state->stream->hfvsif_infopacket;
+ }
}
}
@@ -12049,6 +12086,14 @@ static void get_freesync_config_for_crtc(
}
out:
new_crtc_state->freesync_config = config;
+
+ drm_dbg_driver(new_con_state->base.connector->dev,
+ "VRR: cfg vrr_enabled=%d vrr_supported=%d fs_capable=%d vrefresh=%d min=%d max=%d state=%d\n",
+ new_crtc_state->base.vrr_enabled,
+ new_crtc_state->vrr_supported,
+ new_con_state->freesync_capable, vrefresh,
+ aconnector->min_vfreq, aconnector->max_vfreq,
+ config.state);
}
static void reset_freesync_config_for_crtc(
@@ -14011,6 +14056,15 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector,
if (!adev->dm.freesync_module || !dc_supports_vrr(sink->ctx->dce_version))
goto update;
+ drm_dbg_driver(adev_to_drm(adev),
+ "VRR: enter signal=%d hdmi_vrr=%d mrange[%d-%d] hdmi.vrr_cap[sup=%d min=%d max=%d]\n",
+ sink->sink_signal, connector->display_info.hdmi.vrr_cap.supported,
+ connector->display_info.monitor_range.min_vfreq,
+ connector->display_info.monitor_range.max_vfreq,
+ connector->display_info.hdmi.vrr_cap.supported,
+ connector->display_info.hdmi.vrr_cap.vrr_min,
+ connector->display_info.hdmi.vrr_cap.vrr_max);
+
edid = drm_edid_raw(drm_edid); // FIXME: Get rid of drm_edid_raw()
/* Some eDP panels only have the refresh rate range info in DisplayID */
@@ -14036,7 +14090,9 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector,
amdgpu_dm_connector->as_type = ADAPTIVE_SYNC_TYPE_EDP;
}
- } else if (drm_edid && sink->sink_signal == SIGNAL_TYPE_HDMI_TYPE_A) {
+ } else if (drm_edid &&
+ (sink->sink_signal == SIGNAL_TYPE_HDMI_TYPE_A ||
+ sink->sink_signal == SIGNAL_TYPE_HDMI_FRL)) {
i = parse_hdmi_amd_vsdb(amdgpu_dm_connector, edid, &vsdb_info);
if (i >= 0) {
amdgpu_dm_connector->vsdb_info = vsdb_info;
@@ -14052,6 +14108,59 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector,
connector->display_info.monitor_range.max_vfreq = vsdb_info.max_refresh_rate_hz;
}
}
+
+ drm_dbg_driver(adev_to_drm(adev),
+ "VRR: amd_vsdb i=%d fs_sup=%d min=%d max=%d fs_capable=%d\n",
+ i, vsdb_info.freesync_supported,
+ vsdb_info.min_refresh_rate_hz,
+ vsdb_info.max_refresh_rate_hz, freesync_capable);
+
+ /*
+ * If AMD VSDB didn't provide a valid FreeSync range, fall back to
+ * the HDMI 2.1 VRR capability parsed from the HF-VSDB.
+ */
+ if (!freesync_capable && connector->display_info.hdmi.vrr_cap.supported) {
+ struct drm_hdmi_vrr_cap *vrr_cap =
+ &connector->display_info.hdmi.vrr_cap;
+
+ drm_dbg_driver(adev_to_drm(adev),
+ "VRR: HF-VSDB fallback: hdmi_vrr=1 vrr_cap[sup=%d min=%d max=%d] mrange_max=%d\n",
+ vrr_cap->supported, vrr_cap->vrr_min, vrr_cap->vrr_max,
+ connector->display_info.monitor_range.max_vfreq);
+
+ if (vrr_cap->supported && vrr_cap->vrr_min > 0) {
+ amdgpu_dm_connector->min_vfreq = vrr_cap->vrr_min;
+ amdgpu_dm_connector->max_vfreq = vrr_cap->vrr_max ?
+ vrr_cap->vrr_max :
+ connector->display_info.monitor_range.max_vfreq;
+
+ /*
+ * VRRMAX = 0 in the HF-VSDB means "up to the Base
+ * Refresh Rate". If the EDID also did not provide a
+ * monitor range max, fall back to the Base Refresh
+ * Rate (the highest refresh rate of the preferred
+ * timing) so a valid VRR range is still reported to
+ * userspace.
+ */
+ if (!amdgpu_dm_connector->max_vfreq) {
+ struct drm_display_mode *brr_mode =
+ get_highest_refresh_rate_mode(amdgpu_dm_connector, true);
+
+ if (brr_mode)
+ amdgpu_dm_connector->max_vfreq =
+ drm_mode_vrefresh(brr_mode);
+ }
+
+ if (amdgpu_dm_connector->max_vfreq -
+ amdgpu_dm_connector->min_vfreq > 10)
+ freesync_capable = true;
+
+ connector->display_info.monitor_range.min_vfreq =
+ amdgpu_dm_connector->min_vfreq;
+ connector->display_info.monitor_range.max_vfreq =
+ amdgpu_dm_connector->max_vfreq;
+ }
+ }
}
if (amdgpu_dm_connector->dc_link)
@@ -14093,6 +14202,11 @@ void amdgpu_dm_update_freesync_caps(struct drm_connector *connector,
if (dm_con_state)
dm_con_state->freesync_capable = freesync_capable;
+ drm_dbg_driver(adev_to_drm(adev),
+ "VRR: caps result: freesync_capable=%d min_vfreq=%d max_vfreq=%d\n",
+ freesync_capable, amdgpu_dm_connector->min_vfreq,
+ amdgpu_dm_connector->max_vfreq);
+
if (connector->state && amdgpu_dm_connector->dc_link && !freesync_capable &&
amdgpu_dm_connector->dc_link->replay_settings.config.replay_supported) {
amdgpu_dm_connector->dc_link->replay_settings.config.replay_supported = false;
diff --git a/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h b/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h
--- a/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h
+++ b/drivers/gpu/drm/amd/display/modules/inc/mod_info_packet.h
@@ -67,6 +67,10 @@ struct AS_Df_params {
struct frame_duration_op decrease;
};
+void mod_build_infopacket_vtem(const struct dc_stream_state *stream,
+ const struct mod_vrr_params *vrr, int fva_factor,
+ struct dc_info_packet *infopacket);
+
void mod_build_adaptive_sync_infopacket(const struct dc_stream_state *stream,
enum adaptive_sync_type asType, const struct AS_Df_params *param,
struct dc_info_packet *info_packet);
diff --git a/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c b/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c
--- a/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c
+++ b/drivers/gpu/drm/amd/display/modules/info_packet/info_packet.c
@@ -291,6 +291,21 @@ void set_vsc_packet_colorimetry_data(
info_packet->sb[18] = 0;
}
+static void set_field_with_mask(unsigned char *dest, unsigned int mask, unsigned int value)
+{
+ unsigned int shift = 0;
+
+ if (!mask || !dest)
+ return;
+
+ while (!((mask >> shift) & 1))
+ shift++;
+
+ *dest = *dest & ~mask;
+ value = value & (mask >> shift);
+ *dest = *dest | (value << shift);
+}
+
void mod_build_vsc_infopacket(const struct dc_stream_state *stream,
struct dc_info_packet *info_packet,
enum dc_color_space cs,
@@ -644,6 +659,100 @@ void mod_build_hf_vsif_infopacket(const struct dc_stream_state *stream,
info_packet->valid = true;
}
+static void build_vtem_infopacket_data(const struct dc_stream_state *stream,
+ const struct mod_vrr_params *vrr, int fva_factor,
+ struct dc_info_packet *infopacket)
+{
+ unsigned int field_rate_in_hz;
+
+ /* FVA Factor setting */
+ set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__FVA_FACTOR_M1,
+ (fva_factor > 0) ? (fva_factor - 1) : 0);
+ /* VRR Parameters */
+ if (vrr->state == VRR_STATE_ACTIVE_VARIABLE ||
+ vrr->state == VRR_STATE_ACTIVE_FIXED) {
+ set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__VRR_EN, 1);
+ } else {
+ set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__VRR_EN, 0);
+ }
+
+ if (vrr->state == VRR_STATE_ACTIVE_FIXED)
+ set_field_with_mask(&infopacket->sb[VTEM_MD0], MASK_VTEM_MD0__M_CONST, vrr->m_const);
+
+ if (!stream->timing.vic) {
+ set_field_with_mask(&infopacket->sb[VTEM_MD1], MASK_VTEM_MD1__BASE_VFRONT,
+ stream->timing.v_front_porch);
+
+
+ /* TODO: In dal2, we check mode flags for a reduced blanking timing.
+ * Need a way to relay that information to this function.
+ * if("ReducedBlanking")
+ * {
+ * set_field_with_mask(&infopacket->sb[VRR_VTEM_MD2], MASK__VRR_VTEM_MD2__RB, 1;
+ * }
+ */
+
+ field_rate_in_hz = stream->timing.pix_clk_100hz * 100;
+ field_rate_in_hz /= stream->timing.h_total;
+ field_rate_in_hz = (field_rate_in_hz + stream->timing.v_total / 2)
+ / stream->timing.v_total;
+
+ set_field_with_mask(&infopacket->sb[VTEM_MD2], MASK_VTEM_MD2__BASE_REFRESH_RATE_98,
+ field_rate_in_hz >> 8);
+ set_field_with_mask(&infopacket->sb[VTEM_MD3], MASK_VTEM_MD3__BASE_REFRESH_RATE_07,
+ field_rate_in_hz);
+
+ }
+
+ /*
+ * When no VTEM feature is enabled (neither VRR nor FVA), signal a
+ * zero-length data set (MLDS) by clearing Data_Set_Length. HDMI 2.1
+ * 10.10.2.4 requires the Source to either stop transmitting the VTEM
+ * or set Data_Set_Length = 0 when no feature is enabled; keeping the
+ * VTEM with Data_Set_Length = 0 preserves the every-MTW cadence while
+ * staying compliant (e.g. HDMI GCTS HF1-58 step 6.2).
+ */
+ if (vrr->state != VRR_STATE_ACTIVE_VARIABLE &&
+ vrr->state != VRR_STATE_ACTIVE_FIXED && fva_factor == 0)
+ set_field_with_mask(&infopacket->sb[VTEM_PB6],
+ MASK_VTEM_PB6__DATA_SET_LENGTH_LSB, 0);
+
+ infopacket->valid = true;
+}
+
+static void build_infopacket_header_vtem(enum signal_type signal,
+ struct dc_info_packet *infopacket)
+{
+ /* HEADER */
+
+ /* HB0, HB1, HB2 indicates PacketType VTEMPacket */
+ infopacket->hb0 = 0x7F;
+ infopacket->hb1 = 0xC0;
+ infopacket->hb2 = 0x00; /* sequence_index */
+
+ set_field_with_mask(&infopacket->sb[VTEM_PB0], MASK_VTEM_PB0__VFR, 1);
+ set_field_with_mask(&infopacket->sb[VTEM_PB2], MASK_VTEM_PB2__ORGANIZATION_ID, 1);
+ set_field_with_mask(&infopacket->sb[VTEM_PB3], MASK_VTEM_PB3__DATA_SET_TAG_MSB, 0);
+ set_field_with_mask(&infopacket->sb[VTEM_PB4], MASK_VTEM_PB4__DATA_SET_TAG_LSB, 1);
+ set_field_with_mask(&infopacket->sb[VTEM_PB5], MASK_VTEM_PB5__DATA_SET_LENGTH_MSB, 0);
+ set_field_with_mask(&infopacket->sb[VTEM_PB6], MASK_VTEM_PB6__DATA_SET_LENGTH_LSB, 4);
+}
+
+void mod_build_infopacket_vtem(const struct dc_stream_state *stream,
+ const struct mod_vrr_params *vrr, int fva_factor,
+ struct dc_info_packet *infopacket)
+{
+ /* VTEM info packet for HdmiVrr */
+
+ memset(infopacket, 0, sizeof(struct dc_info_packet));
+
+ /* VTEM Packet is structured differently */
+ build_infopacket_header_vtem(stream->signal, infopacket);
+ build_vtem_infopacket_data(stream, vrr, fva_factor, infopacket);
+
+ infopacket->valid = true;
+}
+
void mod_build_adaptive_sync_infopacket(const struct dc_stream_state *stream,
enum adaptive_sync_type asType,
const struct AS_Df_params *param,
diff --git a/drivers/gpu/drm/drm_edid.c b/drivers/gpu/drm/drm_edid.c
--- a/drivers/gpu/drm/drm_edid.c
+++ b/drivers/gpu/drm/drm_edid.c
@@ -6182,6 +6182,33 @@ static void drm_parse_ycbcr420_deep_color_info(struct drm_connector *connector,
hdmi->y420_dc_modes = dc_mask;
}
+static void drm_parse_hdmi_gaming_info(struct drm_hdmi_info *hdmi, const u8 *db)
+{
+ struct drm_hdmi_vrr_cap *vrr = &hdmi->vrr_cap;
+
+ if (cea_db_payload_len(db) < 8)
+ return;
+
+ hdmi->fapa_start_location = db[8] & DRM_EDID_FAPA_START_LOCATION;
+ hdmi->allm = db[8] & DRM_EDID_ALLM;
+ vrr->fva = db[8] & DRM_EDID_FVA;
+ vrr->cnmvrr = db[8] & DRM_EDID_CNMVRR;
+ vrr->cinema_vrr = db[8] & DRM_EDID_CINEMA_VRR;
+ vrr->mdelta = db[8] & DRM_EDID_MDELTA;
+
+ if (cea_db_payload_len(db) < 9)
+ return;
+
+ vrr->vrr_min = db[9] & DRM_EDID_VRR_MIN_MASK;
+ vrr->supported = (vrr->vrr_min > 0 && vrr->vrr_min <= 48);
+
+ if (cea_db_payload_len(db) < 10)
+ return;
+
+ vrr->vrr_max = (db[9] & DRM_EDID_VRR_MAX_UPPER_MASK) << 2 | db[10];
+ vrr->supported &= (vrr->vrr_max == 0 || vrr->vrr_max >= 100);
+}
+
static void drm_parse_dsc_info(struct drm_hdmi_dsc_cap *hdmi_dsc,
const u8 *hf_scds)
{
@@ -6308,6 +6335,8 @@ static void drm_parse_hdmi_forum_scds(struct drm_connector *connector,
drm_parse_ycbcr420_deep_color_info(connector, hf_scds);
+ drm_parse_hdmi_gaming_info(&connector->display_info.hdmi, hf_scds);
+
if (cea_db_payload_len(hf_scds) >= 11 && hf_scds[11]) {
drm_parse_dsc_info(hdmi_dsc, hf_scds);
dsc_support = true;
@@ -6317,6 +6346,19 @@ static void drm_parse_hdmi_forum_scds(struct drm_connector *connector,
"[CONNECTOR:%d:%s] HF-VSDB: max TMDS clock: %d KHz, HDMI 2.1 support: %s, DSC 1.2 support: %s\n",
connector->base.id, connector->name,
max_tmds_clock, str_yes_no(max_frl_rate), str_yes_no(dsc_support));
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] FAPA in blanking: %s, ALLM support: %s, Fast Vactive support: %s\n",
+ connector->base.id, connector->name, str_yes_no(hdmi->fapa_start_location),
+ str_yes_no(hdmi->allm), str_yes_no(hdmi->vrr_cap.fva));
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] Negative M VRR support: %s, CinemaVRR support: %s, Mdelta: %d\n",
+ connector->base.id, connector->name, str_yes_no(hdmi->vrr_cap.cnmvrr),
+ str_yes_no(hdmi->vrr_cap.cinema_vrr), hdmi->vrr_cap.mdelta);
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] VRRmin: %u, VRRmax: %u, VRR supported: %s\n",
+ connector->base.id, connector->name, hdmi->vrr_cap.vrr_min,
+ hdmi->vrr_cap.vrr_max, str_yes_no(hdmi->vrr_cap.supported));
+
}
static void drm_parse_hdmi_deep_color_info(struct drm_connector *connector,
diff --git a/include/drm/drm_connector.h b/include/drm/drm_connector.h
--- a/include/drm/drm_connector.h
+++ b/include/drm/drm_connector.h
@@ -254,6 +254,44 @@ struct drm_scdc {
struct drm_scrambling scrambling;
};
+/**
+ * struct drm_hdmi_vrr_cap - Information about VRR capabilities of a HDMI sink
+ *
+ * Describes the VRR support provided by HDMI 2.1 sink. The information is
+ * fetched fom additional HFVSDB blocks defined for HDMI 2.1.
+ */
+struct drm_hdmi_vrr_cap {
+ /** @fva: flag for Fast VActive (Quick Frame Transport) support */
+ bool fva;
+
+ /** @mcnmvrr: flag for Negative M VRR support */
+ bool cnmvrr;
+
+ /** @mcinema_vrr: flag for Cinema VRR support */
+ bool cinema_vrr;
+
+ /** @mdelta: flag for limited frame-to-frame compensation support */
+ bool mdelta;
+
+ /**
+ * @vrr_min : minimum supported variable refresh rate in Hz.
+ * Valid values only inide 1 - 48 range
+ */
+ u16 vrr_min;
+
+ /**
+ * @vrr_max : maximum supported variable refresh rate in Hz (optional).
+ * Valid values are either 0 (max based on video mode) or >= 100
+ */
+ u16 vrr_max;
+
+ /**
+ * @supported: flag for vrr support based on checking for VRRmin and
+ * VRRmax values having correct values.
+ */
+ bool supported;
+};
+
/**
* struct drm_hdmi_dsc_cap - DSC capabilities of HDMI sink
*
@@ -330,6 +368,15 @@ struct drm_hdmi_info {
/** @max_lanes: supported by sink */
u8 max_lanes;
+ /** @fapa_start_location: flag for the FAPA in blanking support */
+ bool fapa_start_location;
+
+ /** @allm: flag for Auto Low Latency Mode support by sink */
+ bool allm;
+
+ /** @vrr_cap: VRR capabilities of the sink */
+ struct drm_hdmi_vrr_cap vrr_cap;
+
/** @dsc_cap: DSC capabilities of the sink */
struct drm_hdmi_dsc_cap dsc_cap;
};
@@ -0,0 +1,15 @@
diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
--- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
+++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
@@ -10069,7 +10069,10 @@ static void update_freesync_state_on_stream(
&vrr_infopacket,
pack_sdp_v1_3);
- if (new_stream->sink->sink_signal == SIGNAL_TYPE_HDMI_FRL)
+ /* Per HDMI 2.1, VTEM is valid on TMDS as well as FRL */
+ if (new_stream->sink->sink_signal == SIGNAL_TYPE_HDMI_FRL ||
+ (new_stream->sink->sink_signal == SIGNAL_TYPE_HDMI_TYPE_A &&
+ aconn && aconn->base.display_info.hdmi.vrr_cap.supported))
mod_build_infopacket_vtem(new_stream, &vrr_params, 0, &vrr_infopacket);
new_crtc_state->freesync_vrr_info_changed |=
@@ -0,0 +1,28 @@
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c
index 1aed121f4ddb..081de37c2083 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c
@@ -216,8 +216,9 @@ int amdgpu_smu_pptable_id = -1;
* DISABLE_FRACTIONAL_PWM (bit 2) disabled by default
* PSR (bit 3) disabled by default
* EDP NO POWER SEQUENCING (bit 4) disabled by default
+ * FRL (bit 10) enabled by default
*/
-uint amdgpu_dc_feature_mask = 2;
+uint amdgpu_dc_feature_mask = DC_MULTI_MON_PP_MCLK_SWITCH_MASK | DC_FRL_MASK;
uint amdgpu_dc_debug_mask;
uint amdgpu_dc_visual_confirm;
int amdgpu_async_gfx_ring = 1;
diff --git a/drivers/gpu/drm/amd/include/amd_shared.h b/drivers/gpu/drm/amd/include/amd_shared.h
index 3fd38323a88b..bdd80c5e378e 100644
--- a/drivers/gpu/drm/amd/include/amd_shared.h
+++ b/drivers/gpu/drm/amd/include/amd_shared.h
@@ -287,7 +287,7 @@ enum DC_FEATURE_MASK {
*/
DC_REPLAY_MASK = (1 << 9),
/**
- * @DC_FRL_MASK: (0x400) disabled by default
+ * @DC_FRL_MASK: (0x400) enabled by default
*/
DC_FRL_MASK = (1 << 10),
};
@@ -0,0 +1,335 @@
diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
--- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
+++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
@@ -7096,6 +7096,11 @@ static void fill_stream_properties_from_drm_display_mode(
stream->output_color_space = get_output_color_space(timing_out, connector_state);
stream->content_type = get_output_content_type(connector_state);
+
+ /* DisplayID Type VII pass-through timings. */
+ if (mode_in->dsc_passthrough_timings_support && info->dp_dsc_bpp_x16 != 0) {
+ stream->timing.dsc_fixed_bits_per_pixel_x16 = info->dp_dsc_bpp_x16;
+ }
}
static void fill_audio_info(struct audio_info *audio_info,
@@ -7616,6 +7621,7 @@ create_stream_for_sink(struct drm_connector *connector,
struct drm_display_mode mode;
struct drm_display_mode saved_mode;
struct drm_display_mode *freesync_mode = NULL;
+ struct drm_display_mode *dsc_passthru_mode = NULL;
bool native_mode_found = false;
bool recalculate_timing = false;
bool scale = dm_state->scaling != RMX_OFF;
@@ -7707,6 +7713,16 @@ create_stream_for_sink(struct drm_connector *connector,
}
}
+ list_for_each_entry(dsc_passthru_mode, &connector->modes, head) {
+ if (dsc_passthru_mode->hdisplay == mode.hdisplay &&
+ dsc_passthru_mode->vdisplay == mode.vdisplay &&
+ drm_mode_vrefresh(dsc_passthru_mode) == mode_refresh) {
+ mode.dsc_passthrough_timings_support =
+ dsc_passthru_mode->dsc_passthrough_timings_support;
+ break;
+ }
+ }
+
if (recalculate_timing)
drm_mode_set_crtcinfo(&saved_mode, 0);
diff --git a/drivers/gpu/drm/drm_displayid_internal.h b/drivers/gpu/drm/drm_displayid_internal.h
--- a/drivers/gpu/drm/drm_displayid_internal.h
+++ b/drivers/gpu/drm/drm_displayid_internal.h
@@ -98,6 +98,7 @@ struct displayid_header {
u8 ext_count;
} __packed;
+#define DISPLAYID_BLOCK_REV GENMASK(2, 0)
struct displayid_block {
u8 tag;
u8 rev;
@@ -126,6 +127,7 @@ struct displayid_detailed_timings_1 {
__le16 vsw;
} __packed;
+#define DISPLAYID_BLOCK_PASSTHROUGH_TIMINGS_SUPPORT BIT(3)
struct displayid_detailed_timing_block {
struct displayid_block base;
struct displayid_detailed_timings_1 timings[];
@@ -138,19 +140,28 @@ struct displayid_formula_timings_9 {
u8 vrefresh;
} __packed;
+#define DISPLAYID_BLOCK_DESCRIPTOR_PAYLOAD_BYTES GENMASK(6, 4)
struct displayid_formula_timing_block {
struct displayid_block base;
struct displayid_formula_timings_9 timings[];
} __packed;
+#define DISPLAYID_VESA_DP_TYPE GENMASK(2, 0)
#define DISPLAYID_VESA_MSO_OVERLAP GENMASK(3, 0)
#define DISPLAYID_VESA_MSO_MODE GENMASK(6, 5)
+#define DISPLAYID_VESA_DSC_BPP_INT GENMASK(5, 0)
+#define DISPLAYID_VESA_DSC_BPP_FRACT GENMASK(3, 0)
+
+#define DISPLAYID_VESA_DP_TYPE_EDP 0
+#define DISPLAYID_VESA_DP_TYPE_DP 1
struct displayid_vesa_vendor_specific_block {
struct displayid_block base;
u8 oui[3];
u8 data_structure_type;
u8 mso;
+ u8 dsc_bpp_int;
+ u8 dsc_bpp_fract;
} __packed;
/*
diff --git a/drivers/gpu/drm/drm_edid.c b/drivers/gpu/drm/drm_edid.c
--- a/drivers/gpu/drm/drm_edid.c
+++ b/drivers/gpu/drm/drm_edid.c
@@ -45,6 +45,7 @@
#include <drm/drm_edid.h>
#include <drm/drm_eld.h>
#include <drm/drm_encoder.h>
+#include <drm/drm_fixed.h>
#include <drm/drm_print.h>
#include "drm_crtc_internal.h"
@@ -6683,12 +6684,13 @@ static bool drm_update_displayid_adaptive_sync_range(struct drm_connector *conne
return false;
}
-static void drm_parse_vesa_mso_data(struct drm_connector *connector,
- const struct displayid_block *block)
+static void drm_parse_vesa_specific_block(struct drm_connector *connector,
+ const struct displayid_block *block)
{
struct displayid_vesa_vendor_specific_block *vesa =
(struct displayid_vesa_vendor_specific_block *)block;
struct drm_display_info *info = &connector->display_info;
+ int dp_type;
if (block->num_bytes < 3) {
drm_dbg_kms(connector->dev,
@@ -6700,51 +6702,73 @@ static void drm_parse_vesa_mso_data(struct drm_connector *connector,
if (oui(vesa->oui[0], vesa->oui[1], vesa->oui[2]) != VESA_IEEE_OUI)
return;
- if (sizeof(*vesa) != sizeof(*block) + block->num_bytes) {
+ if (block->num_bytes < 5) {
drm_dbg_kms(connector->dev,
"[CONNECTOR:%d:%s] Unexpected VESA vendor block size\n",
connector->base.id, connector->name);
return;
}
- switch (FIELD_GET(DISPLAYID_VESA_MSO_MODE, vesa->mso)) {
- default:
- drm_dbg_kms(connector->dev, "[CONNECTOR:%d:%s] Reserved MSO mode value\n",
+ dp_type = FIELD_GET(DISPLAYID_VESA_DP_TYPE, vesa->data_structure_type);
+ if (dp_type > 1) {
+ drm_dbg_kms(connector->dev, "[CONNECTOR:%d:%s] Reserved dp type value\n",
connector->base.id, connector->name);
- fallthrough;
- case 0:
- info->mso_stream_count = 0;
- break;
- case 1:
- info->mso_stream_count = 2; /* 2 or 4 links */
- break;
- case 2:
- info->mso_stream_count = 4; /* 4 links */
- break;
}
- if (!info->mso_stream_count) {
+ /* MSO is only supported for eDP */
+ if (dp_type == DISPLAYID_VESA_DP_TYPE_EDP) {
+ switch (FIELD_GET(DISPLAYID_VESA_MSO_MODE, vesa->mso)) {
+ default:
+ drm_dbg_kms(connector->dev, "[CONNECTOR:%d:%s] Reserved MSO mode value\n",
+ connector->base.id, connector->name);
+ fallthrough;
+ case 0:
+ info->mso_stream_count = 0;
+ break;
+ case 1:
+ info->mso_stream_count = 2; /* 2 or 4 links */
+ break;
+ case 2:
+ info->mso_stream_count = 4; /* 4 links */
+ break;
+ }
+ }
+
+ if (info->mso_stream_count) {
+ info->mso_pixel_overlap = FIELD_GET(DISPLAYID_VESA_MSO_OVERLAP, vesa->mso);
+ if (info->mso_pixel_overlap > 8) {
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] Reserved MSO pixel overlap value %u\n",
+ connector->base.id, connector->name,
+ info->mso_pixel_overlap);
+ info->mso_pixel_overlap = 8;
+ }
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] MSO stream count %u, pixel overlap %u\n",
+ connector->base.id, connector->name,
+ info->mso_stream_count, info->mso_pixel_overlap);
+ } else {
info->mso_pixel_overlap = 0;
+ }
+
+ if (block->num_bytes < 7) {
+ /* DSC bpp is optional */
return;
}
- info->mso_pixel_overlap = FIELD_GET(DISPLAYID_VESA_MSO_OVERLAP, vesa->mso);
- if (info->mso_pixel_overlap > 8) {
- drm_dbg_kms(connector->dev,
- "[CONNECTOR:%d:%s] Reserved MSO pixel overlap value %u\n",
- connector->base.id, connector->name,
- info->mso_pixel_overlap);
- info->mso_pixel_overlap = 8;
- }
+ info->dp_dsc_bpp_x16 = FIELD_GET(DISPLAYID_VESA_DSC_BPP_INT, vesa->dsc_bpp_int) << 4 |
+ FIELD_GET(DISPLAYID_VESA_DSC_BPP_FRACT, vesa->dsc_bpp_fract);
- drm_dbg_kms(connector->dev,
- "[CONNECTOR:%d:%s] MSO stream count %u, pixel overlap %u\n",
- connector->base.id, connector->name,
- info->mso_stream_count, info->mso_pixel_overlap);
+ if (info->dp_dsc_bpp_x16 > 0) {
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] DSC bits per pixel " FXP_Q4_FMT "\n",
+ connector->base.id, connector->name,
+ FXP_Q4_ARGS(info->dp_dsc_bpp_x16));
+ }
}
-static void drm_update_mso(struct drm_connector *connector,
- const struct drm_edid *drm_edid)
+static void drm_update_vesa_specific_block(struct drm_connector *connector,
+ const struct drm_edid *drm_edid)
{
const struct displayid_block *block;
struct displayid_iter iter;
@@ -6752,7 +6776,7 @@ static void drm_update_mso(struct drm_connector *connector,
displayid_iter_edid_begin(drm_edid, &iter);
displayid_iter_for_each(block, &iter) {
if (block->tag == DATA_BLOCK_2_VENDOR_SPECIFIC)
- drm_parse_vesa_mso_data(connector, block);
+ drm_parse_vesa_specific_block(connector, block);
}
displayid_iter_end(&iter);
}
@@ -6789,6 +6813,7 @@ static void drm_reset_display_info(struct drm_connector *connector)
info->mso_stream_count = 0;
info->mso_pixel_overlap = 0;
info->max_dsc_bpp = 0;
+ info->dp_dsc_bpp_x16 = 0;
kfree(info->vics);
info->vics = NULL;
@@ -6930,7 +6955,7 @@ static void update_display_info(struct drm_connector *connector,
if (edid->features & DRM_EDID_FEATURE_RGB_YCRCB422)
info->color_formats |= BIT(DRM_OUTPUT_COLOR_FORMAT_YCBCR422);
- drm_update_mso(connector, drm_edid);
+ drm_update_vesa_specific_block(connector, drm_edid);
out:
if (drm_edid_has_internal_quirk(connector, EDID_QUIRK_NON_DESKTOP)) {
@@ -6960,8 +6985,8 @@ static void update_display_info(struct drm_connector *connector,
}
static struct drm_display_mode *drm_mode_displayid_detailed(struct drm_device *dev,
- const struct displayid_detailed_timings_1 *timings,
- bool type_7)
+ const struct displayid_block *block,
+ const struct displayid_detailed_timings_1 *timings)
{
struct drm_display_mode *mode;
unsigned int pixel_clock = (timings->pixel_clock[0] |
@@ -6977,11 +7002,16 @@ static struct drm_display_mode *drm_mode_displayid_detailed(struct drm_device *d
unsigned int vsync_width = le16_to_cpu(timings->vsw) + 1;
bool hsync_positive = le16_to_cpu(timings->hsync) & (1 << 15);
bool vsync_positive = le16_to_cpu(timings->vsync) & (1 << 15);
+ bool type_7 = block->tag == DATA_BLOCK_2_TYPE_7_DETAILED_TIMING;
mode = drm_mode_create(dev);
if (!mode)
return NULL;
+ if (type_7 && FIELD_GET(DISPLAYID_BLOCK_REV, block->rev) >= 1)
+ mode->dsc_passthrough_timings_support =
+ block->rev & DISPLAYID_BLOCK_PASSTHROUGH_TIMINGS_SUPPORT;
+
/* resolution is kHz for type VII, and 10 kHz for type I */
mode->clock = type_7 ? pixel_clock : pixel_clock * 10;
mode->hdisplay = hactive;
@@ -7014,7 +7044,6 @@ static int add_displayid_detailed_1_modes(struct drm_connector *connector,
int num_timings;
struct drm_display_mode *newmode;
int num_modes = 0;
- bool type_7 = block->tag == DATA_BLOCK_2_TYPE_7_DETAILED_TIMING;
/* blocks must be multiple of 20 bytes length */
if (block->num_bytes % 20)
return 0;
@@ -7023,7 +7052,7 @@ static int add_displayid_detailed_1_modes(struct drm_connector *connector,
for (i = 0; i < num_timings; i++) {
struct displayid_detailed_timings_1 *timings = &det->timings[i];
- newmode = drm_mode_displayid_detailed(connector->dev, timings, type_7);
+ newmode = drm_mode_displayid_detailed(connector->dev, block, timings);
if (!newmode)
continue;
@@ -7070,7 +7099,8 @@ static int add_displayid_formula_modes(struct drm_connector *connector,
struct drm_display_mode *newmode;
int num_modes = 0;
bool type_10 = block->tag == DATA_BLOCK_2_TYPE_10_FORMULA_TIMING;
- int timing_size = 6 + ((formula_block->base.rev & 0x70) >> 4);
+ int timing_size = 6 +
+ FIELD_GET(DISPLAYID_BLOCK_DESCRIPTOR_PAYLOAD_BYTES, formula_block->base.rev);
/* extended blocks are not supported yet */
if (timing_size != 6)
diff --git a/include/drm/drm_connector.h b/include/drm/drm_connector.h
--- a/include/drm/drm_connector.h
+++ b/include/drm/drm_connector.h
@@ -939,6 +939,12 @@ struct drm_display_info {
*/
u32 max_dsc_bpp;
+ /**
+ * @dp_dsc_bpp: DP Display-Stream-Compression (DSC) timing's target
+ * DSC bits per pixel in 6.4 fixed point format. 0 means undefined.
+ */
+ u16 dp_dsc_bpp_x16;
+
/**
* @vics: Array of vics_len VICs. Internal to EDID parsing.
*/
diff --git a/include/drm/drm_modes.h b/include/drm/drm_modes.h
--- a/include/drm/drm_modes.h
+++ b/include/drm/drm_modes.h
@@ -417,6 +417,16 @@ struct drm_display_mode {
*/
enum hdmi_picture_aspect picture_aspect_ratio;
+ /**
+ * @dsc_passthrough_timing_support:
+ *
+ * Indicates whether this mode timing descriptor is supported
+ * with specific target DSC bits per pixel only.
+ *
+ * VESA vendor-specific data block shall exist with the relevant
+ * DSC bits per pixel declaration when this flag is set to true.
+ */
+ bool dsc_passthrough_timings_support;
};
/**
@@ -0,0 +1,51 @@
diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
--- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
+++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
@@ -7622,6 +7622,7 @@ create_stream_for_sink(struct drm_connector *connector,
struct drm_display_mode saved_mode;
struct drm_display_mode *freesync_mode = NULL;
struct drm_display_mode *dsc_passthru_mode = NULL;
+ unsigned int dsc_passthru_match_flags;
bool native_mode_found = false;
bool recalculate_timing = false;
bool scale = dm_state->scaling != RMX_OFF;
@@ -7688,6 +7689,22 @@ create_stream_for_sink(struct drm_connector *connector,
mode_refresh = drm_mode_vrefresh(&mode);
+ dsc_passthru_match_flags = DRM_MODE_MATCH_TIMINGS |
+ DRM_MODE_MATCH_CLOCK |
+ DRM_MODE_MATCH_FLAGS |
+ DRM_MODE_MATCH_3D_FLAGS;
+ if (mode.picture_aspect_ratio)
+ dsc_passthru_match_flags |= DRM_MODE_MATCH_ASPECT_RATIO;
+
+ list_for_each_entry(dsc_passthru_mode, &connector->modes, head) {
+ if (drm_mode_match(dsc_passthru_mode, &mode,
+ dsc_passthru_match_flags)) {
+ mode.dsc_passthrough_timings_support =
+ dsc_passthru_mode->dsc_passthrough_timings_support;
+ break;
+ }
+ }
+
if (preferred_mode == NULL) {
/*
* This may not be an error, the use case is when we have no
@@ -7713,16 +7730,6 @@ create_stream_for_sink(struct drm_connector *connector,
}
}
- list_for_each_entry(dsc_passthru_mode, &connector->modes, head) {
- if (dsc_passthru_mode->hdisplay == mode.hdisplay &&
- dsc_passthru_mode->vdisplay == mode.vdisplay &&
- drm_mode_vrefresh(dsc_passthru_mode) == mode_refresh) {
- mode.dsc_passthrough_timings_support =
- dsc_passthru_mode->dsc_passthrough_timings_support;
- break;
- }
- }
-
if (recalculate_timing)
drm_mode_set_crtcinfo(&saved_mode, 0);
@@ -0,0 +1,26 @@
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c
@@ -1426,7 +1426,7 @@ int amdgpu_userq_post_reset(struct amdgpu_device *adev, bool vram_lost)
struct amdgpu_usermode_queue *queue;
const struct amdgpu_userq_funcs *userq_funcs;
unsigned long queue_id;
- int r = 0;
+ int ret = 0, r;
xa_for_each(&adev->userq_doorbell_xa, queue_id, queue) {
if (queue->state == AMDGPU_USERQ_STATE_HUNG && !vram_lost) {
@@ -1435,11 +1435,12 @@ int amdgpu_userq_post_reset(struct amdgpu_device *adev, bool vram_lost)
r = userq_funcs->map(queue);
if (r) {
dev_err(adev->dev, "Failed to remap queue %ld\n", queue_id);
+ ret = r;
continue;
}
queue->state = AMDGPU_USERQ_STATE_MAPPED;
}
}
- return r;
+ return ret;
}
@@ -0,0 +1,16 @@
diff --git a/drivers/gpu/drm/i915/display/intel_cdclk.c b/drivers/gpu/drm/i915/display/intel_cdclk.c
--- a/drivers/gpu/drm/i915/display/intel_cdclk.c
+++ b/drivers/gpu/drm/i915/display/intel_cdclk.c
@@ -2366,8 +2366,10 @@ static void bxt_sanitize_cdclk(struct intel_display *display)
* dividers both syncing to an active pipe, or asynchronously
* (PIPE_NONE).
*/
- cdctl &= ~bxt_cdclk_cd2x_pipe(display, INVALID_PIPE);
- cdctl |= bxt_cdclk_cd2x_pipe(display, INVALID_PIPE);
+ if (DISPLAY_VER(display) < 30) {
+ cdctl &= ~bxt_cdclk_cd2x_pipe(display, INVALID_PIPE);
+ cdctl |= bxt_cdclk_cd2x_pipe(display, INVALID_PIPE);
+ }
if (cdctl != expected) {
if (DISPLAY_VER(display) < 20) {
@@ -0,0 +1,18 @@
diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
--- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
+++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c
@@ -4160,8 +4160,12 @@ static void update_connector_ext_caps(struct amdgpu_dm_connector *aconnector)
else if (!IS_ERR_OR_NULL(panel_backlight_quirk) &&
panel_backlight_quirk->force_pwm)
caps->aux_support = false;
- if (caps->aux_support)
- aconnector->dc_link->backlight_control_type = BACKLIGHT_CONTROL_AMD_AUX;
+ if (caps->aux_support) {
+ if (aconnector->dc_link->dpcd_caps.panel_luminance_control)
+ aconnector->dc_link->backlight_control_type = BACKLIGHT_CONTROL_VESA_AUX;
+ else
+ aconnector->dc_link->backlight_control_type = BACKLIGHT_CONTROL_AMD_AUX;
+ }
luminance_range = &conn_base->display_info.luminance_range;
@@ -0,0 +1,16 @@
diff --git a/drivers/gpu/drm/i915/display/intel_dp.c b/drivers/gpu/drm/i915/display/intel_dp.c
--- a/drivers/gpu/drm/i915/display/intel_dp.c
+++ b/drivers/gpu/drm/i915/display/intel_dp.c
@@ -3219,11 +3219,7 @@ static bool intel_dp_needs_as_sdp(struct intel_dp *intel_dp,
if (drm_dp_is_branch(intel_dp->dpcd))
return false;
- if (intel_psr_needs_alpm_aux_less(intel_dp, crtc_state) &&
- !intel_psr_pr_async_video_timing_supported(intel_dp))
- return true;
-
- return intel_vrr_possible(crtc_state);
+ return crtc_state->vrr.enable;
}
static void intel_dp_compute_as_sdp(struct intel_dp *intel_dp,
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,507 @@
diff --git a/sound/core/pcm_native.c b/sound/core/pcm_native.c
--- a/sound/core/pcm_native.c
+++ b/sound/core/pcm_native.c
@@ -4023,20 +4023,33 @@ int snd_pcm_mmap_data(struct snd_pcm_substream *substream, struct file *file,
return -EINVAL;
}
runtime = substream->runtime;
- if (runtime->state == SNDRV_PCM_STATE_OPEN)
- return -EBADFD;
- if (!(runtime->info & SNDRV_PCM_INFO_MMAP))
- return -ENXIO;
+ /* don't race with buffer reallocation in hw_params/hw_free */
+ if (!atomic_inc_unless_negative(&runtime->buffer_accessing))
+ return -EBUSY;
+ if (runtime->state == SNDRV_PCM_STATE_OPEN) {
+ err = -EBADFD;
+ goto out;
+ }
+ if (!(runtime->info & SNDRV_PCM_INFO_MMAP)) {
+ err = -ENXIO;
+ goto out;
+ }
if (runtime->access == SNDRV_PCM_ACCESS_RW_INTERLEAVED ||
- runtime->access == SNDRV_PCM_ACCESS_RW_NONINTERLEAVED)
- return -EINVAL;
+ runtime->access == SNDRV_PCM_ACCESS_RW_NONINTERLEAVED) {
+ err = -EINVAL;
+ goto out;
+ }
size = area->vm_end - area->vm_start;
offset = area->vm_pgoff << PAGE_SHIFT;
dma_bytes = PAGE_ALIGN(runtime->dma_bytes);
- if ((size_t)size > dma_bytes)
- return -EINVAL;
- if (offset > dma_bytes - size)
- return -EINVAL;
+ if ((size_t)size > dma_bytes) {
+ err = -EINVAL;
+ goto out;
+ }
+ if (offset > dma_bytes - size) {
+ err = -EINVAL;
+ goto out;
+ }
area->vm_ops = &snd_pcm_vm_ops_data;
area->vm_private_data = substream;
@@ -4046,6 +4059,8 @@ int snd_pcm_mmap_data(struct snd_pcm_substream *substream, struct file *file,
err = snd_pcm_lib_default_mmap(substream, area);
if (!err)
atomic_inc(&substream->mmap_count);
+out:
+ atomic_dec(&runtime->buffer_accessing);
return err;
}
EXPORT_SYMBOL(snd_pcm_mmap_data);
diff --git a/sound/core/ump.c b/sound/core/ump.c
--- a/sound/core/ump.c
+++ b/sound/core/ump.c
@@ -1335,6 +1335,8 @@ static void update_legacy_names(struct snd_ump_endpoint *ump)
{
struct snd_rawmidi *rmidi = ump->legacy_rmidi;
+ if (!rmidi)
+ return;
update_legacy_substreams(ump, rmidi, SNDRV_RAWMIDI_STREAM_INPUT);
update_legacy_substreams(ump, rmidi, SNDRV_RAWMIDI_STREAM_OUTPUT);
}
@@ -1343,6 +1345,8 @@ static void ump_legacy_set_rawmidi_name(struct snd_ump_endpoint *ump)
{
struct snd_rawmidi *rmidi = ump->legacy_rmidi;
+ if (!rmidi)
+ return;
snprintf(rmidi->name, sizeof(rmidi->name), "%.68s (MIDI 1.0)",
ump->core.name);
}
diff --git a/sound/drivers/dummy.c b/sound/drivers/dummy.c
--- a/sound/drivers/dummy.c
+++ b/sound/drivers/dummy.c
@@ -807,7 +807,7 @@ static int snd_dummy_capsrc_put(struct snd_kcontrol *kcontrol, struct snd_ctl_el
left = ucontrol->value.integer.value[0] & 1;
right = ucontrol->value.integer.value[1] & 1;
guard(spinlock_irq)(&dummy->mixer_lock);
- change = dummy->capture_source[addr][0] != left &&
+ change = dummy->capture_source[addr][0] != left ||
dummy->capture_source[addr][1] != right;
dummy->capture_source[addr][0] = left;
dummy->capture_source[addr][1] = right;
diff --git a/sound/hda/codecs/cirrus/cs420x.c b/sound/hda/codecs/cirrus/cs420x.c
--- a/sound/hda/codecs/cirrus/cs420x.c
+++ b/sound/hda/codecs/cirrus/cs420x.c
@@ -571,6 +571,7 @@ static const struct hda_model_fixup cs4208_models[] = {
static const struct hda_quirk cs4208_fixup_tbl[] = {
SND_PCI_QUIRK_VENDOR(0x106b, "Apple", CS4208_MAC_AUTO),
+ SND_PCI_QUIRK(0x8086, 0x7270, "MacBookAir 7,2", CS4208_MAC_AUTO),
{} /* terminator */
};
@@ -583,6 +584,7 @@ static const struct hda_quirk cs4208_mac_fixup_tbl[] = {
SND_PCI_QUIRK(0x106b, 0x7800, "MacPro 6,1", CS4208_MACMINI),
SND_PCI_QUIRK(0x106b, 0x7b00, "MacBookPro 12,1", CS4208_MBP11),
SND_PCI_QUIRK(0x106b, 0x7f00, "iMac 16,1", CS4208_MBP11),
+ SND_PCI_QUIRK(0x8086, 0x7270, "MacBookAir 7,2", CS4208_MBA6),
{} /* terminator */
};
diff --git a/sound/hda/codecs/conexant.c b/sound/hda/codecs/conexant.c
--- a/sound/hda/codecs/conexant.c
+++ b/sound/hda/codecs/conexant.c
@@ -249,6 +249,25 @@ static void cx_update_headset_mic_vref(struct hda_codec *codec, struct hda_jack_
}
}
+#define SN6140_S3_AFG_D0_DELAY_MS 1000
+
+static void cx_set_power_state(struct hda_codec *codec, hda_nid_t fg,
+ unsigned int power_state)
+{
+ snd_hda_codec_write_sync(codec, fg, 0, AC_VERB_SET_POWER_STATE, power_state);
+
+ /*
+ * SN6140 may not respond to AFG D0 immediately after S3.
+ * Wait before the D0 verb so the power-state command itself succeeds.
+ */
+ if (codec->core.vendor_id == 0x14f11f87 &&
+ power_state == AC_PWRST_D0 &&
+ codec->core.dev.power.power_state.event == PM_EVENT_RESUME)
+ msleep(SN6140_S3_AFG_D0_DELAY_MS);
+
+ snd_hda_codec_set_power_to_all(codec, fg, power_state);
+}
+
static int cx_suspend(struct hda_codec *codec)
{
cx_auto_shutdown(codec);
@@ -1308,6 +1327,7 @@ static const struct hda_codec_ops cx_codec_ops = {
.init = cx_init,
.unsol_event = snd_hda_jack_unsol_event,
.suspend = cx_suspend,
+ .set_power_state = cx_set_power_state,
.check_power_status = snd_hda_gen_check_power_status,
.stream_pm = snd_hda_gen_stream_pm,
};
diff --git a/sound/hda/codecs/realtek/alc269.c b/sound/hda/codecs/realtek/alc269.c
--- a/sound/hda/codecs/realtek/alc269.c
+++ b/sound/hda/codecs/realtek/alc269.c
@@ -2379,6 +2379,33 @@ static void alc_fixup_headset_mode_alc255_no_hp_mic(struct hda_codec *codec,
}
}
+/*
+ * On the Acer Aspire A515-57G (and possibly other models sharing this
+ * board), if headphones are already inserted into the combo jack before
+ * the codec powers up (cold boot), the impedance-based headset-type
+ * sensing races and misclassifies the jack, driving the wrong output
+ * configuration (audible as missing center-panned/vocal content). A
+ * genuine physical unplug/replug after boot fixes it by forcing a fresh
+ * sense transient. Mirror that here on cold boot only: give the sense
+ * hardware time to settle, then force a fresh classification.
+ */
+static void alc_fixup_headset_mode_acer_coldboot(struct hda_codec *codec,
+ const struct hda_fixup *fix, int action)
+{
+ struct alc_spec *spec = codec->spec;
+
+ alc_fixup_headset_mode(codec, fix, action);
+
+ if (action == HDA_FIXUP_ACT_INIT &&
+ !is_s3_resume(codec) && !is_s4_resume(codec) &&
+ spec->current_headset_mode != ALC_HEADSET_MODE_UNPLUGGED) {
+ msleep(500);
+ spec->current_headset_mode = ALC_HEADSET_MODE_UNKNOWN;
+ spec->current_headset_type = ALC_HEADSET_TYPE_UNKNOWN;
+ alc_fixup_headset_mode(codec, fix, action);
+ }
+}
+
static void alc288_update_headset_jack_cb(struct hda_codec *codec,
struct hda_jack_callback *jack)
{
@@ -4248,6 +4275,7 @@ enum {
ALC282_FIXUP_ACER_DISABLE_LINEOUT,
ALC255_FIXUP_ACER_LIMIT_INT_MIC_BOOST,
ALC256_FIXUP_ACER_HEADSET_MIC,
+ ALC256_FIXUP_ACER_COLDBOOT,
ALC285_FIXUP_IDEAPAD_S740_COEF,
ALC285_FIXUP_HP_LIMIT_INT_MIC_BOOST,
ALC295_FIXUP_ASUS_DACS,
@@ -6311,6 +6339,12 @@ static const struct hda_fixup alc269_fixups[] = {
.chained = true,
.chain_id = ALC269_FIXUP_HEADSET_MODE_NO_HP_MIC
},
+ [ALC256_FIXUP_ACER_COLDBOOT] = {
+ .type = HDA_FIXUP_FUNC,
+ .v.func = alc_fixup_headset_mode_acer_coldboot,
+ .chained = true,
+ .chain_id = ALC256_FIXUP_ACER_SFG16_MICMUTE_LED,
+ },
[ALC285_FIXUP_IDEAPAD_S740_COEF] = {
.type = HDA_FIXUP_FUNC,
.v.func = alc285_fixup_ideapad_s740_coef,
@@ -7148,13 +7182,14 @@ static const struct hda_quirk alc269_fixup_tbl[] = {
SND_PCI_QUIRK(0x1025, 0x1597, "Acer Nitro 5 AN517-55", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1025, 0x159e, "Acer Nitro 5 AN515-46", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1025, 0x160e, "Acer PT316-51S", ALC2XX_FIXUP_HEADSET_MIC),
- SND_PCI_QUIRK(0x1025, 0x1616, "Acer Aspire A515-57", ALC256_FIXUP_ACER_SFG16_MICMUTE_LED),
+ SND_PCI_QUIRK(0x1025, 0x1616, "Acer Aspire A515-57", ALC256_FIXUP_ACER_COLDBOOT),
SND_PCI_QUIRK(0x1025, 0x161f, "Acer S40-54", ALC256_FIXUP_ACER_MIC_NO_PRESENCE),
SND_PCI_QUIRK(0x1025, 0x1640, "Acer Aspire A315-44P", ALC256_FIXUP_ACER_SFG16_MICMUTE_LED),
SND_PCI_QUIRK(0x1025, 0x166c, "Acer Predator PH16-71", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1025, 0x1679, "Acer Nitro 16 AN16-41", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1025, 0x169a, "Acer Swift SFG16", ALC256_FIXUP_ACER_SFG16_MICMUTE_LED),
SND_PCI_QUIRK(0x1025, 0x171e, "Acer Nitro ANV15-51", ALC245_FIXUP_ACER_MICMUTE_LED),
+ SND_PCI_QUIRK(0x1025, 0x1731, "Acer Predator PHN16-72", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1025, 0x173a, "Acer Swift SFG14-73", ALC245_FIXUP_ACER_MICMUTE_LED),
SND_PCI_QUIRK(0x1025, 0x1758, "Acer Nitro ANV15-41", ALC245_FIXUP_ACER_MICMUTE_LED),
SND_PCI_QUIRK(0x1025, 0x1826, "Acer Helios ZPC", ALC287_FIXUP_PREDATOR_SPK_CS35L41_I2C_2),
@@ -8100,6 +8135,7 @@ static const struct hda_quirk alc269_fixup_tbl[] = {
SND_PCI_QUIRK(0x17aa, 0x3801, "Lenovo Yoga9 14IAP7", ALC287_FIXUP_YOGA9_14IAP7_BASS_SPK_PIN),
HDA_CODEC_QUIRK(0x17aa, 0x3802, "DuetITL 2021", ALC287_FIXUP_YOGA7_14ITL_SPEAKERS),
SND_PCI_QUIRK(0x17aa, 0x3802, "Lenovo Yoga Pro 9 14IRP8", ALC287_FIXUP_TAS2781_I2C),
+ SND_PCI_QUIRK(0x17aa, 0x380b, "Lenovo Yoga Slim 9 14ILL10", ALC287_FIXUP_YOGA9_14IAP7_BASS_SPK_PIN),
/* Yoga Pro 9 16IMH9 and Legion 7 16ITHG6 share PCI SSID 17aa:3811
* with Legion S7 15IMH05; use codec SSID to distinguish them
*/
@@ -8287,6 +8323,7 @@ static const struct hda_quirk alc269_fixup_tbl[] = {
SND_PCI_QUIRK(0x1d05, 0x3034, "TongFang X6KK45xU", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1d05, 0x30ba, "TongFang XxAF5xxx", ALC2XX_FIXUP_HEADSET_MIC),
SND_PCI_QUIRK(0x1d17, 0x3288, "Haier Boyue G42", ALC269VC_FIXUP_ACER_VCOPPERBOX_PINS),
+ SND_PCI_QUIRK(0x1d19, 0x0006, "VAIO VJS131", ALC233_FIXUP_ASUS_MIC_NO_PRESENCE),
SND_PCI_QUIRK(0x1d72, 0x1602, "RedmiBook", ALC255_FIXUP_XIAOMI_HEADSET_MIC),
SND_PCI_QUIRK(0x1d72, 0x1701, "XiaomiNotebook Pro", ALC298_FIXUP_DELL1_MIC_NO_PRESENCE),
SND_PCI_QUIRK(0x1d72, 0x1901, "RedmiBook 14", ALC256_FIXUP_ASUS_HEADSET_MIC),
diff --git a/sound/hda/core/device.c b/sound/hda/core/device.c
--- a/sound/hda/core/device.c
+++ b/sound/hda/core/device.c
@@ -404,6 +404,7 @@ static void setup_fg_nodes(struct hdac_device *codec)
*/
int snd_hdac_refresh_widgets(struct hdac_device *codec)
{
+ hda_nid_t fg = codec->afg ? codec->afg : codec->mfg;
hda_nid_t start_nid;
int nums, err = 0;
@@ -412,10 +413,10 @@ int snd_hdac_refresh_widgets(struct hdac_device *codec)
* widgets array.
*/
guard(mutex)(&codec->widget_lock);
- nums = snd_hdac_get_sub_nodes(codec, codec->afg, &start_nid);
+ nums = snd_hdac_get_sub_nodes(codec, fg, &start_nid);
if (!start_nid || nums <= 0 || nums >= 0xff) {
dev_err(&codec->dev, "cannot read sub nodes for FG 0x%02x\n",
- codec->afg);
+ fg);
return -EINVAL;
}
diff --git a/sound/usb/caiaq/audio.c b/sound/usb/caiaq/audio.c
--- a/sound/usb/caiaq/audio.c
+++ b/sound/usb/caiaq/audio.c
@@ -828,16 +828,13 @@ int snd_usb_caiaq_audio_init(struct snd_usb_caiaqdev *cdev)
cdev->data_urbs_in = alloc_urbs(cdev, SNDRV_PCM_STREAM_CAPTURE, &ret);
if (ret < 0) {
- kfree(cdev->data_cb_info);
- free_urbs(cdev->data_urbs_in);
+ snd_usb_caiaq_audio_free(cdev);
return ret;
}
cdev->data_urbs_out = alloc_urbs(cdev, SNDRV_PCM_STREAM_PLAYBACK, &ret);
if (ret < 0) {
- kfree(cdev->data_cb_info);
- free_urbs(cdev->data_urbs_in);
- free_urbs(cdev->data_urbs_out);
+ snd_usb_caiaq_audio_free(cdev);
return ret;
}
@@ -858,6 +855,9 @@ void snd_usb_caiaq_audio_free(struct snd_usb_caiaqdev *cdev)
dev_dbg(dev, "%s(%p)\n", __func__, cdev);
free_urbs(cdev->data_urbs_in);
+ cdev->data_urbs_in = NULL;
free_urbs(cdev->data_urbs_out);
+ cdev->data_urbs_out = NULL;
kfree(cdev->data_cb_info);
+ cdev->data_cb_info = NULL;
}
diff --git a/sound/usb/fcp.c b/sound/usb/fcp.c
--- a/sound/usb/fcp.c
+++ b/sound/usb/fcp.c
@@ -191,6 +191,10 @@ static int fcp_usb(struct usb_mixer_interface *mixer, u32 opcode,
const int max_retries = 5;
int err;
+ CLASS(snd_usb_lock, pm)(mixer->chip);
+ if (pm.err < 0)
+ return -EIO;
+
if (!private->urb)
return -ENODEV;
@@ -1026,6 +1030,10 @@ static int fcp_init(struct usb_mixer_interface *mixer,
struct usb_device *dev = mixer->chip->dev;
int err;
+ CLASS(snd_usb_lock, pm)(mixer->chip);
+ if (pm.err < 0)
+ return -EIO;
+
err = snd_usb_ctl_msg(dev, usb_rcvctrlpipe(dev, 0),
FCP_USB_REQ_STEP0,
USB_RECIP_INTERFACE | USB_TYPE_CLASS | USB_DIR_IN,
diff --git a/sound/usb/mixer_maps.c b/sound/usb/mixer_maps.c
--- a/sound/usb/mixer_maps.c
+++ b/sound/usb/mixer_maps.c
@@ -518,6 +518,19 @@ static const struct usbmix_name_map audient_id14_map[] = {
{}
};
+/*
+ * Audient iD24: feature unit 12 ("Speaker Playback Volume") sits in the
+ * monitor-mixer branch and does not apply volume to all of its channels;
+ * when userspace adopts it as the master playback volume, the left main
+ * output stays at 0 dB while the right one is attenuated, producing a
+ * stereo imbalance. Rename it so that it is not picked up as the
+ * stream's master volume control.
+ */
+static const struct usbmix_name_map audient_id24_map[] = {
+ { 12, "Monitor Mix Playback" }, /* FU, partial channel coverage */
+ {}
+};
+
/*
* Control map entries
*/
@@ -611,6 +624,11 @@ static const struct usbmix_ctl_map usbmix_ctl_maps[] = {
.id = USB_ID(0x2708, 0x0008),
.map = audient_id14_map,
},
+ {
+ /* Audient iD24 */
+ .id = USB_ID(0x2708, 0x000d),
+ .map = audient_id24_map,
+ },
{
/* KEF X300A */
.id = USB_ID(0x27ac, 0x1000),
diff --git a/sound/usb/mixer_quirks.c b/sound/usb/mixer_quirks.c
--- a/sound/usb/mixer_quirks.c
+++ b/sound/usb/mixer_quirks.c
@@ -3480,6 +3480,10 @@ static int snd_rme_digiface_write_reg(struct snd_kcontrol *kcontrol, int item, u
struct usb_device *dev = chip->dev;
int err;
+ CLASS(snd_usb_lock, pm)(chip);
+ if (pm.err < 0)
+ return -EIO;
+
err = snd_usb_ctl_msg(dev, usb_sndctrlpipe(dev, 0),
item,
USB_DIR_OUT | USB_TYPE_VENDOR | USB_RECIP_DEVICE,
@@ -3499,6 +3503,10 @@ static int snd_rme_digiface_read_status(struct snd_kcontrol *kcontrol, u32 statu
__le32 buf[4] = {};
int err;
+ CLASS(snd_usb_lock, pm)(chip);
+ if (pm.err < 0)
+ return -EIO;
+
err = snd_usb_ctl_msg(dev, usb_rcvctrlpipe(dev, 0),
RME_DIGIFACE_READ_STATUS,
USB_DIR_IN | USB_TYPE_VENDOR | USB_RECIP_DEVICE,
diff --git a/sound/usb/mixer_s1810c.c b/sound/usb/mixer_s1810c.c
--- a/sound/usb/mixer_s1810c.c
+++ b/sound/usb/mixer_s1810c.c
@@ -474,6 +474,10 @@ snd_s1810c_switch_get(struct snd_kcontrol *kctl,
u32 state = 0;
int ret;
+ CLASS(snd_usb_lock, pm)(mixer->chip);
+ if (pm.err < 0)
+ return -EIO;
+
guard(mutex)(&private->data_mutex);
ret = snd_s1810c_get_switch_state(mixer, kctl, &state);
if (ret < 0)
@@ -504,6 +508,10 @@ snd_s1810c_switch_set(struct snd_kcontrol *kctl,
u32 newval = 0;
int ret = 0;
+ CLASS(snd_usb_lock, pm)(mixer->chip);
+ if (pm.err < 0)
+ return -EIO;
+
guard(mutex)(&private->data_mutex);
ret = snd_s1810c_get_switch_state(mixer, kctl, &curval);
if (ret < 0)
diff --git a/sound/usb/mixer_scarlett.c b/sound/usb/mixer_scarlett.c
--- a/sound/usb/mixer_scarlett.c
+++ b/sound/usb/mixer_scarlett.c
@@ -707,6 +707,10 @@ static int scarlett_ctl_meter_get(struct snd_kcontrol *kctl,
int idx = snd_usb_ctrl_intf(elem->head.mixer->hostif) | (elem->head.id << 8);
int err;
+ CLASS(snd_usb_lock, pm)(chip);
+ if (pm.err < 0)
+ return -EIO;
+
err = snd_usb_ctl_msg(chip->dev,
usb_rcvctrlpipe(chip->dev, 0),
UAC2_CS_MEM,
diff --git a/sound/usb/mixer_scarlett2.c b/sound/usb/mixer_scarlett2.c
--- a/sound/usb/mixer_scarlett2.c
+++ b/sound/usb/mixer_scarlett2.c
@@ -2603,9 +2603,9 @@ static int scarlett2_usb_rx(struct usb_device *dev, int interface,
}
/* Send a proprietary format request to the Scarlett interface */
-static int scarlett2_usb(
- struct usb_mixer_interface *mixer, u32 cmd,
- void *req_data, u16 req_size, void *resp_data, u16 resp_size)
+static int scarlett2_usb_nopm(struct usb_mixer_interface *mixer, u32 cmd,
+ void *req_data, u16 req_size,
+ void *resp_data, u16 resp_size)
{
struct scarlett2_data *private = mixer->private_data;
struct usb_device *dev = mixer->chip->dev;
@@ -2713,6 +2713,18 @@ static int scarlett2_usb(
return err;
}
+static int scarlett2_usb(struct usb_mixer_interface *mixer, u32 cmd,
+ void *req_data, u16 req_size,
+ void *resp_data, u16 resp_size)
+{
+ CLASS(snd_usb_lock, pm)(mixer->chip);
+ if (pm.err < 0)
+ return -EIO;
+
+ return scarlett2_usb_nopm(mixer, cmd, req_data, req_size,
+ resp_data, resp_size);
+}
+
/* Send a USB message to get data; result placed in *buf */
static int scarlett2_usb_get(
struct usb_mixer_interface *mixer,
@@ -3020,9 +3032,21 @@ static int scarlett2_usb_set_config_buf(
/* Send SCARLETT2_USB_DATA_CMD SCARLETT2_USB_CONFIG_SAVE */
static void scarlett2_config_save(struct usb_mixer_interface *mixer)
{
- int err;
+ __le32 req = cpu_to_le32(SCARLETT2_USB_CONFIG_SAVE);
+ int err = scarlett2_usb(mixer, SCARLETT2_USB_DATA_CMD,
+ &req, sizeof(req), NULL, 0);
+
+ if (err < 0)
+ usb_audio_err(mixer->chip, "config save failed: %d\n", err);
+}
+
+/* The USB suspend callback must not acquire another PM reference. */
+static void scarlett2_config_save_nopm(struct usb_mixer_interface *mixer)
+{
+ __le32 req = cpu_to_le32(SCARLETT2_USB_CONFIG_SAVE);
+ int err = scarlett2_usb_nopm(mixer, SCARLETT2_USB_DATA_CMD,
+ &req, sizeof(req), NULL, 0);
- err = scarlett2_usb_activate_config(mixer, SCARLETT2_USB_CONFIG_SAVE);
if (err < 0)
usb_audio_err(mixer->chip, "config save failed: %d\n", err);
}
@@ -8639,7 +8663,7 @@ static void scarlett2_private_suspend(struct usb_mixer_interface *mixer)
struct scarlett2_data *private = mixer->private_data;
if (cancel_delayed_work_sync(&private->work))
- scarlett2_config_save(private->mixer);
+ scarlett2_config_save_nopm(private->mixer);
scarlett2_cleanup_urb(mixer);
}
diff --git a/sound/usb/mixer_us16x08.c b/sound/usb/mixer_us16x08.c
--- a/sound/usb/mixer_us16x08.c
+++ b/sound/usb/mixer_us16x08.c
@@ -151,6 +151,9 @@ static const char *const route_names[] = {
static int snd_us16x08_recv_urb(struct snd_usb_audio *chip,
unsigned char *buf, int size)
{
+ CLASS(snd_usb_lock, pm)(chip);
+ if (pm.err < 0)
+ return -EIO;
guard(mutex)(&chip->mutex);
snd_usb_ctl_msg(chip->dev,
@@ -165,6 +168,10 @@ static int snd_us16x08_recv_urb(struct snd_usb_audio *chip,
*/
static int snd_us16x08_send_urb(struct snd_usb_audio *chip, char *buf, int size)
{
+ CLASS(snd_usb_lock, pm)(chip);
+ if (pm.err < 0)
+ return -EIO;
+
return snd_usb_ctl_msg(chip->dev, usb_sndctrlpipe(chip->dev, 0),
SND_US16X08_URB_REQUEST, SND_US16X08_URB_REQUESTTYPE,
0, 0, buf, size);
@@ -0,0 +1,19 @@
diff --git a/sound/soc/intel/boards/sof_sdw.c b/sound/soc/intel/boards/sof_sdw.c
--- a/sound/soc/intel/boards/sof_sdw.c
+++ b/sound/soc/intel/boards/sof_sdw.c
@@ -845,6 +845,15 @@ static const struct dmi_system_id sof_sdw_quirk_table[] = {
},
.driver_data = (void *)(SOC_SDW_PCH_DMIC),
},
+ {
+ /* Dell XPS 13 DX13260 (Wildcat Lake), CS42L43 + 2x CS35L63 sidecar amps */
+ .callback = sof_sdw_quirk_cb,
+ .matches = {
+ DMI_MATCH(DMI_SYS_VENDOR, "Dell Inc"),
+ DMI_EXACT_MATCH(DMI_PRODUCT_SKU, "0E53")
+ },
+ .driver_data = (void *)(SOC_SDW_SIDECAR_AMPS),
+ },
{}
};
@@ -0,0 +1,30 @@
diff --git a/sound/soc/codecs/rt766-sdca.c b/sound/soc/codecs/rt766-sdca.c
--- a/sound/soc/codecs/rt766-sdca.c
+++ b/sound/soc/codecs/rt766-sdca.c
@@ -936,9 +936,8 @@ static int rt766_sdca_pcm_hw_params(struct snd_pcm_substream *substream,
{
struct snd_soc_component *component = dai->component;
struct rt766_sdca_priv *rt766 = snd_soc_component_get_drvdata(component);
- struct sdw_stream_config stream_config;
+ struct sdw_stream_config stream_config = {0};
struct sdw_port_config port_config;
- enum sdw_data_direction direction;
struct sdw_stream_runtime *sdw_stream;
unsigned int sampling_rate;
int retval, port;
@@ -957,7 +956,6 @@ static int rt766_sdca_pcm_hw_params(struct snd_pcm_substream *substream,
/* SoundWire specific configuration */
if (substream->stream == SNDRV_PCM_STREAM_PLAYBACK) {
- direction = SDW_DATA_DIR_RX;
if (dai->id == RT766_AIF1)
port = 3;
else if (dai->id == RT766_AIF2)
@@ -965,7 +963,6 @@ static int rt766_sdca_pcm_hw_params(struct snd_pcm_substream *substream,
else
return -EINVAL;
} else {
- direction = SDW_DATA_DIR_TX;
if (dai->id == RT766_AIF1)
port = 12;
else if (dai->id == RT766_AIF3)
@@ -0,0 +1,39 @@
diff --git a/sound/hda/codecs/realtek/alc269.c b/sound/hda/codecs/realtek/alc269.c
--- a/sound/hda/codecs/realtek/alc269.c
+++ b/sound/hda/codecs/realtek/alc269.c
@@ -4226,6 +4226,7 @@ enum {
ALC294_FIXUP_ASUS_GU502_VERBS,
ALC294_FIXUP_ASUS_G513_PINS,
ALC285_FIXUP_ASUS_G533Z_PINS,
+ ALC285_FIXUP_ASUS_G733Z_VERBS,
ALC285_FIXUP_HP_GPIO_LED,
ALC285_FIXUP_HP_MUTE_LED,
ALC285_FIXUP_HP_SPECTRE_X360_MUTE_LED,
@@ -7129,6 +7130,19 @@ static const struct hda_fixup alc269_fixups[] = {
.chained = true,
.chain_id = ALC287_FIXUP_TXNW2781_I2C,
},
+ [ALC285_FIXUP_ASUS_G733Z_VERBS] = {
+ .type = HDA_FIXUP_VERBS,
+ .v.verbs = (const struct hda_verb[]) {
+ /* Enables internal speaker */
+ {0x17, AC_VERB_SET_CONNECT_SEL, 0x00},
+ {0x14, AC_VERB_SET_PIN_WIDGET_CONTROL, 0x40},
+ {0x14, AC_VERB_SET_AMP_GAIN_MUTE, 0xb000},
+ {0x1e, AC_VERB_SET_PIN_WIDGET_CONTROL, 0x40},
+ {0x16, AC_VERB_SET_PIN_WIDGET_CONTROL, 0x00},
+ {0x1b, AC_VERB_SET_PIN_WIDGET_CONTROL, 0x00},
+ { }
+ }
+ },
};
static const struct hda_quirk alc269_fixup_tbl[] = {
@@ -7775,6 +7789,7 @@ static const struct hda_quirk alc269_fixup_tbl[] = {
SND_PCI_QUIRK(0x1043, 0x12a3, "Asus N7691ZM", ALC269_FIXUP_ASUS_N7601ZM),
SND_PCI_QUIRK(0x1043, 0x12af, "ASUS UX582ZS", ALC245_FIXUP_CS35L41_SPI_2),
SND_PCI_QUIRK(0x1043, 0x12b4, "ASUS B3405CCA / P3405CCA", ALC294_FIXUP_ASUS_CS35L41_SPI_2),
+ SND_PCI_QUIRK(0x1043, 0x12bf, "ASUS ROG Strix G733ZW", ALC285_FIXUP_ASUS_G733Z_VERBS),
SND_PCI_QUIRK(0x1043, 0x12e0, "ASUS X541SA", ALC256_FIXUP_ASUS_MIC_NO_PRESENCE),
SND_PCI_QUIRK(0x1043, 0x12f0, "ASUS X541UV", ALC256_FIXUP_ASUS_MIC_NO_PRESENCE),
SND_PCI_QUIRK(0x1043, 0x1313, "Asus K42JZ", ALC269VB_FIXUP_ASUS_MIC_NO_PRESENCE),
@@ -0,0 +1,17 @@
diff --git a/sound/soc/amd/yc/acp6x-mach.c b/sound/soc/amd/yc/acp6x-mach.c
--- a/sound/soc/amd/yc/acp6x-mach.c
+++ b/sound/soc/amd/yc/acp6x-mach.c
@@ -619,6 +619,13 @@ static const struct dmi_system_id yc_acp_quirk_table[] = {
DMI_MATCH(DMI_PRODUCT_NAME, "Swift SFA16-41"),
}
},
+ {
+ .driver_data = &acp6x_card,
+ .matches = {
+ DMI_MATCH(DMI_BOARD_VENDOR, "MDU"),
+ DMI_MATCH(DMI_PRODUCT_NAME, "Aspire A314-23P"),
+ }
+ },
{
.driver_data = &acp6x_card,
.matches = {
@@ -0,0 +1,41 @@
diff --git a/drivers/media/pci/intel/ipu-bridge.c b/drivers/media/pci/intel/ipu-bridge.c
--- a/drivers/media/pci/intel/ipu-bridge.c
+++ b/drivers/media/pci/intel/ipu-bridge.c
@@ -173,6 +173,19 @@ static const struct acpi_device_id ivsc_acpi_ids[] = {
{ "INTC10E1" }, /* PTL */
};
+/*
+ * The subset of ivsc_acpi_ids[] which are IVSC, rather than CVS, devices. The
+ * CVS IDs are deliberately not listed here: new ones keep being added, whereas
+ * this list is complete.
+ */
+static const struct acpi_device_id ivsc_only_acpi_ids[] = {
+ { "INTC1059" },
+ { "INTC1095" },
+ { "INTC100A" },
+ { "INTC10CF" },
+ { }
+};
+
static struct acpi_device *ipu_bridge_get_ivsc_acpi_dev(struct acpi_device *adev)
{
unsigned int i;
@@ -224,6 +237,17 @@ static struct device *ipu_bridge_get_ivsc_csi_dev(struct acpi_device *adev)
return csi_dev;
}
+ /*
+ * The lookups below match on the ACPI companion alone. That is fine for
+ * CVS, which binds a driver to that very device, but not for IVSC: there
+ * the ACPI device also has a driverless platform device, which would be
+ * returned instead of the mei-csi client. Return NULL for IVSC so that
+ * the caller fails and the probe is retried once the IVSC device shows
+ * up.
+ */
+ if (!acpi_match_device_ids(adev, ivsc_only_acpi_ids))
+ return NULL;
+
/* Try to locate CVS device on the I2C bus */
csi_dev = bus_find_device_by_acpi_dev(&i2c_bus_type, adev);
if (csi_dev)
@@ -0,0 +1,33 @@
diff --git a/drivers/acpi/scan.c b/drivers/acpi/scan.c
--- a/drivers/acpi/scan.c
+++ b/drivers/acpi/scan.c
@@ -862,6 +862,7 @@ static const char * const acpi_honor_dep_ids[] = {
"INTC10DE", /* CVS (LNL) driver must be loaded to allow camera streaming */
"INTC10E0", /* CVS (ARL) driver must be loaded to allow camera streaming */
"INTC10E1", /* CVS (PTL) driver must be loaded to allow camera streaming */
+ "INTC10FA", /* CVS (NVL) driver must be loaded to allow camera streaming */
"RSCV0001", /* RISC-V PLIC */
"RSCV0002", /* RISC-V APLIC */
"RSCV0005", /* RISC-V SBI MPXY MBOX */
diff --git a/drivers/media/i2c/cvs/core.c b/drivers/media/i2c/cvs/core.c
--- a/drivers/media/i2c/cvs/core.c
+++ b/drivers/media/i2c/cvs/core.c
@@ -962,6 +962,7 @@ static const struct acpi_device_id intel_cvs_acpi_match[] = {
{ "INTC10DE" }, /* LNL */
{ "INTC10E0" }, /* ARL */
{ "INTC10E1" }, /* PTL */
+ { "INTC10FA" }, /* NVL */
{ }
};
MODULE_DEVICE_TABLE(acpi, intel_cvs_acpi_match);
diff --git a/drivers/media/pci/intel/ipu-bridge.c b/drivers/media/pci/intel/ipu-bridge.c
--- a/drivers/media/pci/intel/ipu-bridge.c
+++ b/drivers/media/pci/intel/ipu-bridge.c
@@ -171,6 +171,7 @@ static const struct acpi_device_id ivsc_acpi_ids[] = {
{ "INTC10DE" }, /* LNL */
{ "INTC10E0" }, /* ARL */
{ "INTC10E1" }, /* PTL */
+ { "INTC10FA" }, /* NVL */
};
/*
@@ -0,0 +1,33 @@
diff --git a/drivers/media/i2c/cvs/core.c b/drivers/media/i2c/cvs/core.c
--- a/drivers/media/i2c/cvs/core.c
+++ b/drivers/media/i2c/cvs/core.c
@@ -723,8 +723,6 @@ static int cvs_core_probe(struct device *dev, struct i2c_client *i2c)
}
if (ctx->res == ICVS_FULLCAP) {
- struct gpio_desc *wake;
-
ctx->rst = devm_gpiod_get(dev, "rst", GPIOD_OUT_HIGH);
if (IS_ERR(ctx->rst)) {
ret = dev_err_probe(dev, PTR_ERR(ctx->rst),
@@ -732,14 +730,12 @@ static int cvs_core_probe(struct device *dev, struct i2c_client *i2c)
goto err_put_ipu;
}
- wake = devm_gpiod_get(dev, "wake", GPIOD_IN);
- if (IS_ERR(wake)) {
- ret = dev_err_probe(dev, PTR_ERR(wake),
- "failed to get wake GPIO\n");
- goto err_put_ipu;
- }
-
- ctx->irq = gpiod_to_irq(wake);
+ /*
+ * Do not request the line: another device's _CRS may list
+ * the same pin, and its driver would then fail with -EBUSY.
+ */
+ ctx->irq = acpi_dev_gpio_irq_get_by(ACPI_COMPANION(dev),
+ "wake", 0);
if (ctx->irq < 0) {
ret = dev_err_probe(dev, ctx->irq,
"failed to get wake IRQ\n");
@@ -0,0 +1,76 @@
diff --git a/drivers/input/mouse/focaltech.c b/drivers/input/mouse/focaltech.c
--- a/drivers/input/mouse/focaltech.c
+++ b/drivers/input/mouse/focaltech.c
@@ -78,8 +78,8 @@ struct focaltech_finger_state {
* Absolute position (from the bottom left corner) of the
* finger.
*/
- unsigned int x;
- unsigned int y;
+ int x;
+ int y;
};
/*
@@ -108,7 +108,7 @@ struct focaltech_hw_state {
};
struct focaltech_data {
- unsigned int x_max, y_max;
+ int x_max, y_max;
struct focaltech_hw_state state;
};
@@ -126,17 +126,16 @@ static void focaltech_report_state(struct psmouse *psmouse)
input_mt_slot(dev, i);
input_mt_report_slot_state(dev, MT_TOOL_FINGER, active);
if (active) {
- unsigned int clamped_x, clamped_y;
/*
* The touchpad might report invalid data, so we clamp
* the resulting values so that we do not confuse
- * userspace.
+ * userspace or accumulate coordinate wind-up.
*/
- clamped_x = clamp(finger->x, 0U, priv->x_max);
- clamped_y = clamp(finger->y, 0U, priv->y_max);
- input_report_abs(dev, ABS_MT_POSITION_X, clamped_x);
+ finger->x = clamp(finger->x, 0, priv->x_max);
+ finger->y = clamp(finger->y, 0, priv->y_max);
+ input_report_abs(dev, ABS_MT_POSITION_X, finger->x);
input_report_abs(dev, ABS_MT_POSITION_Y,
- priv->y_max - clamped_y);
+ priv->y_max - finger->y);
input_report_abs(dev, ABS_TOOL_WIDTH, state->width);
}
}
diff --git a/drivers/input/mouse/psmouse-base.c b/drivers/input/mouse/psmouse-base.c
--- a/drivers/input/mouse/psmouse-base.c
+++ b/drivers/input/mouse/psmouse-base.c
@@ -267,7 +267,15 @@ void psmouse_set_state(struct psmouse *psmouse, enum psmouse_state new_state)
*/
static int psmouse_handle_byte(struct psmouse *psmouse)
{
- psmouse_ret_t rc = psmouse->protocol_handler(psmouse);
+ psmouse_ret_t rc;
+
+ /* protocol_handler is NULL when device is being disconnected */
+ if (unlikely(!psmouse->protocol_handler)) {
+ psmouse->pktcnt = 0;
+ return 0;
+ }
+
+ rc = psmouse->protocol_handler(psmouse);
switch (rc) {
case PSMOUSE_BAD_DATA:
@@ -1466,6 +1474,9 @@ static void psmouse_disconnect(struct serio *serio)
psmouse_deactivate(parent);
}
+ scoped_guard(serio_pause_rx, serio)
+ psmouse->protocol_handler = NULL;
+
if (psmouse->disconnect)
psmouse->disconnect(psmouse);
@@ -0,0 +1,11 @@
diff --git a/drivers/i2c/i2c-core-acpi.c b/drivers/i2c/i2c-core-acpi.c
--- a/drivers/i2c/i2c-core-acpi.c
+++ b/drivers/i2c/i2c-core-acpi.c
@@ -373,6 +373,7 @@ static const struct acpi_device_id i2c_acpi_force_100khz_device_ids[] = {
* the device works without issues on Windows at what is expected to be
* a 400KHz frequency. The root cause of the issue is not known.
*/
+ { "ASUE140D", 0 },
{ "DLL0945", 0 },
{ "ELAN0678", 0 },
{ "ELAN06FA", 0 },
@@ -0,0 +1,24 @@
diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c
--- a/drivers/hid/hid-asus.c
+++ b/drivers/hid/hid-asus.c
@@ -1294,12 +1294,14 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id)
return ret;
}
- for (int r = 0; r < ARRAY_SIZE(asus_report_id_init); r++) {
- if (asus_has_report_id(hdev, asus_report_id_init[r])) {
- ret = asus_kbd_init(hdev, asus_report_id_init[r]);
- if (ret < 0)
- hid_warn(hdev, "Failed to initialize 0x%x: %d.\n",
- asus_report_id_init[r], ret);
+ if (!drvdata->tp) {
+ for (int r = 0; r < ARRAY_SIZE(asus_report_id_init); r++) {
+ if (asus_has_report_id(hdev, asus_report_id_init[r])) {
+ ret = asus_kbd_init(hdev, asus_report_id_init[r]);
+ if (ret < 0)
+ hid_warn(hdev, "Failed to initialize 0x%x: %d.\n",
+ asus_report_id_init[r], ret);
+ }
}
}
@@ -0,0 +1,47 @@
diff --git a/drivers/thunderbolt/stream.c b/drivers/thunderbolt/stream.c
index 4cc86d8d6491..43d29d69dcdb 100644
--- a/drivers/thunderbolt/stream.c
+++ b/drivers/thunderbolt/stream.c
@@ -512,8 +512,10 @@ tbstream_dev_alloc_tx(struct tbstream_dev *sdev, enum tbstream_frame_pdf pdf,
dma_sync_single_for_cpu(dma_dev, sf->frame.buffer_phy, size,
DMA_TO_DEVICE);
if (pdf == TBSTREAM_DATA) {
- if (copy_page_from_iter(sf->page, 0, size, from) != size)
+ if (copy_page_from_iter(sf->page, 0, size, from) != size) {
+ sdev->tx_ring.cons--;
return ERR_PTR(-EFAULT);
+ }
} else {
memset(page_address(sf->page), 0, size);
}
@@ -671,7 +673,7 @@ tbstream_dev_fops_read_iter(struct kiocb *kiocb, struct iov_iter *to)
}
nbytes = 0;
- while (nbytes < iov_iter_count(to)) {
+ while (iov_iter_count(to)) {
struct tbstream_frame *sf;
size_t size, sf_size;
@@ -693,7 +695,7 @@ tbstream_dev_fops_read_iter(struct kiocb *kiocb, struct iov_iter *to)
}
sf_size = tb_ring_frame_size(&sf->frame);
- size = min(iov_iter_count(to) - nbytes, sf_size);
+ size = min(iov_iter_count(to), sf_size);
if (copy_page_to_iter(sf->page, sf->offset, size, to) != size) {
ret = -EFAULT;
@@ -763,10 +765,10 @@ tbstream_dev_fops_write_iter(struct kiocb *kiocb, struct iov_iter *from)
}
nbytes = 0;
- while (nbytes < iov_iter_count(from)) {
+ while (iov_iter_count(from)) {
size_t size;
- size = min(iov_iter_count(from) - nbytes, TB_MAX_FRAME_SIZE);
+ size = min(iov_iter_count(from), TB_MAX_FRAME_SIZE);
ret = tbstream_dev_send_data(sdev, from, size);
if (ret) {
/*
@@ -0,0 +1,477 @@
diff --git a/drivers/thunderbolt/nhi.c b/drivers/thunderbolt/nhi.c
index 35e3c119d5ee..a4816db5cacd 100644
--- a/drivers/thunderbolt/nhi.c
+++ b/drivers/thunderbolt/nhi.c
@@ -235,6 +235,12 @@ static void ring_write_descriptors(struct tb_ring *ring)
{
struct ring_frame *frame, *n;
struct ring_desc *descriptor;
+ u32 flags;
+
+ flags = RING_DESC_POSTED;
+ if (!(ring->flags & RING_FLAG_NO_INTERRUPT))
+ flags |= RING_DESC_INTERRUPT;
+
list_for_each_entry_safe(frame, n, &ring->queue, list) {
if (ring_full(ring))
break;
@@ -242,7 +248,7 @@ static void ring_write_descriptors(struct tb_ring *ring)
descriptor = &ring->descriptors[ring->head];
descriptor->phys = frame->buffer_phy;
descriptor->time = 0;
- descriptor->flags = RING_DESC_POSTED | RING_DESC_INTERRUPT;
+ descriptor->flags = flags;
if (ring->is_tx) {
descriptor->length = frame->size;
descriptor->eof = frame->eof;
@@ -339,8 +345,9 @@ EXPORT_SYMBOL_GPL(__tb_ring_enqueue);
* @ring: Ring to poll
*
* This function can be called when @start_poll callback of the @ring
- * has been called. It will read one completed frame from the ring and
- * return it to the caller.
+ * has been called or the ring is created with %RING_FLAG_NO_INTERRUPT.
+ * It will read one completed frame from the ring and return it to the
+ * caller.
*
* Return: Pointer to &struct ring_frame, %NULL if there is no more
* completed frames.
@@ -538,6 +545,12 @@ static struct tb_ring *tb_ring_alloc(struct tb_nhi *nhi, u32 hop, int size,
dev_dbg(nhi->dev, "allocating %s ring %d of size %d\n",
transmit ? "TX" : "RX", hop, size);
+ if ((flags & RING_FLAG_NO_INTERRUPT) && start_poll) {
+ dev_WARN(nhi->dev,
+ "start_poll() and NO_INTERRUPT cannot be used at the same time\n");
+ return NULL;
+ }
+
ring = kzalloc_obj(*ring);
if (!ring)
return NULL;
@@ -568,7 +581,7 @@ static struct tb_ring *tb_ring_alloc(struct tb_nhi *nhi, u32 hop, int size,
if (!ring->descriptors)
goto err_free_ring;
- if (nhi->ops->request_ring_irq) {
+ if (!(flags & RING_FLAG_NO_INTERRUPT) && nhi->ops->request_ring_irq) {
if (nhi->ops->request_ring_irq(ring, flags & RING_FLAG_NO_SUSPEND))
goto err_free_descs;
}
@@ -701,7 +714,8 @@ void tb_ring_start(struct tb_ring *ring)
ring_iowrite32options(ring, flags, 0);
}
- ring_interrupt_active(ring, true);
+ if (!(ring->flags & RING_FLAG_NO_INTERRUPT))
+ ring_interrupt_active(ring, true);
ring->running = true;
err:
spin_unlock(&ring->lock);
@@ -761,7 +775,8 @@ void tb_ring_stop(struct tb_ring *ring)
RING_TYPE(ring), ring->hop);
goto err;
}
- ring_interrupt_active(ring, false);
+ if (!(ring->flags & RING_FLAG_NO_INTERRUPT))
+ ring_interrupt_active(ring, false);
ring_iowrite32options(ring, 0, 0);
ring_iowrite64desc(ring, 0, 0);
diff --git a/drivers/thunderbolt/stream.c b/drivers/thunderbolt/stream.c
index 43d29d69dcdb..a7d9fda66875 100644
--- a/drivers/thunderbolt/stream.c
+++ b/drivers/thunderbolt/stream.c
@@ -9,10 +9,12 @@
#define pr_fmt(fmt) "tbstream: " fmt
+#include <linux/delay.h>
#include <linux/configfs.h>
#include <linux/file.h>
#include <linux/fs.h>
#include <linux/idr.h>
+#include <linux/ktime.h>
#include <linux/miscdevice.h>
#include <linux/module.h>
#include <linux/mutex.h>
@@ -128,6 +130,7 @@ struct tbstream_ring {
* @out_hopid: Out HopID
* @ring_size: Size of the rings
* @throttling: Interrupt throttling rate in ns
+ * @busy_poll: Instead of interrupts, busy poll the rings
* @users: Number of times @cdev has been opened
* @closed: CLOSE packet was received
* @removed: Userspace removed the ConfigFS group underneath.
@@ -147,6 +150,7 @@ struct tbstream_dev {
int out_hopid;
unsigned int ring_size;
unsigned int throttling;
+ bool busy_poll;
int users;
bool closed;
bool removed;
@@ -536,10 +540,39 @@ tbstream_dev_send_data(struct tbstream_dev *sdev, struct iov_iter *from,
return tb_ring_tx(sdev->tx_ring.ring, &sf->frame);
}
+static void
+tbstream_dev_poll_ring(struct tbstream_dev *sdev, struct tbstream_ring *ring)
+{
+ struct ring_frame *frame;
+
+ if (!sdev->busy_poll)
+ return;
+
+ while ((frame = tb_ring_poll(ring->ring)))
+ frame->callback(ring->ring, frame, false);
+}
+
static int tbstream_dev_send_close(struct tbstream_dev *sdev)
{
struct tbstream_frame *sf;
+ if (sdev->busy_poll) {
+ /*
+ * When busy polling it's the write(2) path that
+ * advances the completions so it is possible that the
+ * ring is full at this point. Advance the ring here so
+ * that there is room for the CLOSE packet to be sent.
+ */
+ ktime_t timeout = ktime_add_ms(ktime_get(), 500);
+
+ do {
+ if (tbstream_ring_available(&sdev->tx_ring))
+ break;
+ tbstream_dev_poll_ring(sdev, &sdev->tx_ring);
+ fsleep(15);
+ } while (ktime_before(ktime_get(), timeout));
+ }
+
sf = tbstream_dev_alloc_tx(sdev, TBSTREAM_CLOSE, NULL, SZ_256);
if (IS_ERR(sf))
return PTR_ERR(sf);
@@ -549,12 +582,15 @@ static int tbstream_dev_send_close(struct tbstream_dev *sdev)
static int tbstream_dev_start(struct tbstream_dev *sdev)
{
struct tb_xdomain *xd = tbstream_dev_xdomain(sdev);
+ unsigned int flags = RING_FLAG_FRAME | RING_FLAG_E2E;
u16 sof_mask, eof_mask;
struct tb_ring *ring;
int ret, e2e_tx_hop;
- ring = tb_ring_alloc_tx(xd->tb->nhi, -1, sdev->ring_size,
- RING_FLAG_FRAME | RING_FLAG_E2E);
+ if (sdev->busy_poll)
+ flags |= RING_FLAG_NO_INTERRUPT;
+
+ ring = tb_ring_alloc_tx(xd->tb->nhi, -1, sdev->ring_size, flags);
if (!ring)
return -ENOMEM;
sdev->tx_ring.ring = ring;
@@ -567,9 +603,8 @@ static int tbstream_dev_start(struct tbstream_dev *sdev)
sof_mask = BIT(TBSTREAM_FRAME_START);
eof_mask = BIT(TBSTREAM_DATA) | BIT(TBSTREAM_CLOSE);
- ring = tb_ring_alloc_rx(xd->tb->nhi, -1, sdev->ring_size,
- RING_FLAG_FRAME | RING_FLAG_E2E, e2e_tx_hop,
- sof_mask, eof_mask, NULL, NULL);
+ ring = tb_ring_alloc_rx(xd->tb->nhi, -1, sdev->ring_size, flags,
+ e2e_tx_hop, sof_mask, eof_mask, NULL, NULL);
if (!ring) {
ret = -ENOMEM;
goto err_free_tx_buffers;
@@ -607,15 +642,43 @@ static int tbstream_dev_start(struct tbstream_dev *sdev)
return ret;
}
+static bool tbstream_dev_tx_drained(const struct tbstream_dev *sdev)
+{
+ const struct tbstream_ring *ring = &sdev->tx_ring;
+
+ /*
+ * Everything is completed when number of free TX slots is back
+ * to the maximum.
+ */
+ return ring->prod - ring->cons == tb_ring_size(ring->ring) - 1;
+}
+
static void tbstream_dev_stop(struct tbstream_dev *sdev)
{
struct tb_xdomain *xd;
- /* Wait for the ring to complete any outstanding frames */
- tb_ring_flush(sdev->tx_ring.ring, 500);
- tb_ring_stop(sdev->tx_ring.ring);
- tb_ring_flush(sdev->rx_ring.ring, 500);
- tb_ring_stop(sdev->rx_ring.ring);
+ if (sdev->busy_poll) {
+ /*
+ * When busy polling we must advance the ring ourselves
+ * to push all outstanding frames on the wire.
+ */
+ ktime_t timeout = ktime_add_ms(ktime_get(), 500);
+
+ do {
+ if (tbstream_dev_tx_drained(sdev))
+ break;
+ tbstream_dev_poll_ring(sdev, &sdev->tx_ring);
+ fsleep(15);
+ } while (ktime_before(ktime_get(), timeout));
+
+ tb_ring_stop(sdev->tx_ring.ring);
+ tb_ring_stop(sdev->rx_ring.ring);
+ } else {
+ tb_ring_flush(sdev->tx_ring.ring, 500);
+ tb_ring_stop(sdev->tx_ring.ring);
+ tb_ring_flush(sdev->rx_ring.ring, 500);
+ tb_ring_stop(sdev->rx_ring.ring);
+ }
xd = tbstream_dev_xdomain(sdev);
if (xd) {
@@ -633,10 +696,24 @@ static void tbstream_dev_stop(struct tbstream_dev *sdev)
sdev->tx_ring.ring = NULL;
}
+/* Use only with read_iter/write_iter() to handle nowait */
+static int tbstream_dev_lock(struct tbstream_dev *sdev, bool nowait)
+{
+ if (nowait) {
+ if (!mutex_trylock(&sdev->lock))
+ return -EAGAIN;
+ } else {
+ if (mutex_lock_interruptible(&sdev->lock))
+ return -ERESTARTSYS;
+ }
+ return 0;
+}
+
static ssize_t
tbstream_dev_fops_read_iter(struct kiocb *kiocb, struct iov_iter *to)
{
struct file *file = kiocb->ki_filp;
+ bool nowait = file->f_flags & O_NONBLOCK || kiocb->ki_flags & IOCB_NOWAIT;
struct tbstream_dev *sdev = to_tbstream_dev(file->private_data);
size_t nbytes;
int ret;
@@ -645,31 +722,50 @@ tbstream_dev_fops_read_iter(struct kiocb *kiocb, struct iov_iter *to)
if (ret)
return ret;
- if (mutex_lock_interruptible(&sdev->lock))
- return -ERESTARTSYS;
+ ret = tbstream_dev_lock(sdev, nowait);
+ if (ret)
+ return ret;
- while (!tbstream_ring_available(&sdev->rx_ring)) {
- mutex_unlock(&sdev->lock);
-
- if (file->f_flags & O_NONBLOCK)
- return -EAGAIN;
- ret = wait_event_interruptible(sdev->wait,
- tbstream_ring_available(&sdev->rx_ring) ||
- tbstream_dev_valid(sdev) != 0 ||
- tbstream_dev_closed(sdev) ||
- tbstream_dev_removed(sdev));
- if (ret)
- return ret;
+ for (;;) {
+ /* When busy polling, advance any completions manually */
+ tbstream_dev_poll_ring(sdev, &sdev->rx_ring);
ret = tbstream_dev_valid(sdev);
+ if (ret) {
+ mutex_unlock(&sdev->lock);
+ return ret;
+ }
+
+ if (tbstream_dev_closed(sdev) || tbstream_dev_removed(sdev)) {
+ mutex_unlock(&sdev->lock);
+ return 0;
+ }
+
+ if (tbstream_ring_available(&sdev->rx_ring))
+ break;
+
+ mutex_unlock(&sdev->lock);
+
+ if (nowait)
+ return -EAGAIN;
+
+ if (sdev->busy_poll) {
+ if (signal_pending(current))
+ return -ERESTARTSYS;
+ cond_resched();
+ } else {
+ ret = wait_event_interruptible(sdev->wait,
+ tbstream_ring_available(&sdev->rx_ring) ||
+ tbstream_dev_valid(sdev) != 0 ||
+ tbstream_dev_closed(sdev) ||
+ tbstream_dev_removed(sdev));
+ if (ret)
+ return ret;
+ }
+
+ ret = tbstream_dev_lock(sdev, nowait);
if (ret)
return ret;
-
- if (tbstream_dev_closed(sdev) || tbstream_dev_removed(sdev))
- return 0;
-
- if (mutex_lock_interruptible(&sdev->lock))
- return -ERESTARTSYS;
}
nbytes = 0;
@@ -729,6 +825,7 @@ static ssize_t
tbstream_dev_fops_write_iter(struct kiocb *kiocb, struct iov_iter *from)
{
struct file *file = kiocb->ki_filp;
+ bool nowait = file->f_flags & O_NONBLOCK || kiocb->ki_flags & IOCB_NOWAIT;
struct tbstream_dev *sdev = to_tbstream_dev(file->private_data);
size_t nbytes;
int ret;
@@ -737,31 +834,49 @@ tbstream_dev_fops_write_iter(struct kiocb *kiocb, struct iov_iter *from)
if (ret)
return ret;
- if (mutex_lock_interruptible(&sdev->lock))
- return -ERESTARTSYS;
+ ret = tbstream_dev_lock(sdev, nowait);
+ if (ret)
+ return ret;
- while (!tbstream_ring_available(&sdev->tx_ring)) {
- mutex_unlock(&sdev->lock);
-
- if (file->f_flags & O_NONBLOCK)
- return -EAGAIN;
- ret = wait_event_interruptible(sdev->wait,
- tbstream_ring_available(&sdev->tx_ring) ||
- tbstream_dev_valid(sdev) != 0 ||
- tbstream_dev_closed(sdev) ||
- tbstream_dev_removed(sdev));
- if (ret)
- return ret;
+ for (;;) {
+ tbstream_dev_poll_ring(sdev, &sdev->tx_ring);
ret = tbstream_dev_valid(sdev);
+ if (ret) {
+ mutex_unlock(&sdev->lock);
+ return ret;
+ }
+
+ if (tbstream_dev_closed(sdev) || tbstream_dev_removed(sdev)) {
+ mutex_unlock(&sdev->lock);
+ return -ENXIO;
+ }
+
+ if (tbstream_ring_available(&sdev->tx_ring))
+ break;
+
+ mutex_unlock(&sdev->lock);
+
+ if (nowait)
+ return -EAGAIN;
+
+ if (sdev->busy_poll) {
+ if (signal_pending(current))
+ return -ERESTARTSYS;
+ cond_resched();
+ } else {
+ ret = wait_event_interruptible(sdev->wait,
+ tbstream_ring_available(&sdev->tx_ring) ||
+ tbstream_dev_valid(sdev) != 0 ||
+ tbstream_dev_closed(sdev) ||
+ tbstream_dev_removed(sdev));
+ if (ret)
+ return ret;
+ }
+
+ ret = tbstream_dev_lock(sdev, nowait);
if (ret)
return ret;
-
- if (tbstream_dev_closed(sdev) || tbstream_dev_removed(sdev))
- return -ENXIO;
-
- if (mutex_lock_interruptible(&sdev->lock))
- return -ERESTARTSYS;
}
nbytes = 0;
@@ -795,6 +910,13 @@ tbstream_dev_fops_poll(struct file *file, struct poll_table_struct *wait)
struct tbstream_dev *sdev = to_tbstream_dev(file->private_data);
__poll_t mask = 0;
+ /*
+ * Without interrupts there is nothing that can wake us up so
+ * return failure instead.
+ */
+ if (sdev->busy_poll)
+ return EPOLLERR;
+
poll_wait(file, &sdev->wait, wait);
guard(mutex)(&sdev->lock);
if (tbstream_dev_valid(sdev) != 0) {
@@ -904,6 +1026,35 @@ tbstream_dev_from_group(struct config_group *group)
return container_of(group, struct tbstream_dev, group);
}
+static ssize_t tbstream_dev_busy_poll_show(struct config_item *item, char *buf)
+{
+ struct config_group *group = to_config_group(item);
+ struct tbstream_dev *sdev = tbstream_dev_from_group(group);
+
+ return sysfs_emit(buf, "%u\n", sdev->busy_poll);
+}
+
+static ssize_t
+tbstream_dev_busy_poll_store(struct config_item *item, const char *buf,
+ size_t count)
+{
+ struct config_group *group = to_config_group(item);
+ struct tbstream_dev *sdev = tbstream_dev_from_group(group);
+ bool busy_poll;
+ int ret;
+
+ ret = kstrtobool(buf, &busy_poll);
+ if (ret)
+ return ret;
+
+ guard(mutex)(&sdev->lock);
+ if (sdev->users)
+ return -EBUSY;
+ sdev->busy_poll = busy_poll;
+ return count;
+}
+CONFIGFS_ATTR(tbstream_dev_, busy_poll);
+
static ssize_t tbstream_dev_index_show(struct config_item *item, char *buf)
{
struct config_group *group = to_config_group(item);
@@ -1210,6 +1361,7 @@ tbstream_dev_throttling_store(struct config_item *item, const char *buf,
CONFIGFS_ATTR(tbstream_dev_, throttling);
static struct configfs_attribute *tbstream_dev_attrs[] = {
+ &tbstream_dev_attr_busy_poll,
&tbstream_dev_attr_index,
&tbstream_dev_attr_in_hopid,
&tbstream_dev_attr_out_hopid,
diff --git a/include/linux/thunderbolt.h b/include/linux/thunderbolt.h
index 557288c0274b..8367d106d81b 100644
--- a/include/linux/thunderbolt.h
+++ b/include/linux/thunderbolt.h
@@ -592,6 +592,8 @@ struct tb_ring {
#define RING_FLAG_FRAME BIT(1)
/* Enable end-to-end flow control */
#define RING_FLAG_E2E BIT(2)
+/* Do not enable interrupt for the ring */
+#define RING_FLAG_NO_INTERRUPT BIT(3)
struct ring_frame;
typedef void (*ring_cb)(struct tb_ring *, struct ring_frame *, bool canceled);
@@ -0,0 +1,132 @@
diff --git a/drivers/usb/typec/altmodes/displayport.c b/drivers/usb/typec/altmodes/displayport.c
index 263a89c5f324..5ee33a69b9cf 100644
--- a/drivers/usb/typec/altmodes/displayport.c
+++ b/drivers/usb/typec/altmodes/displayport.c
@@ -790,7 +790,6 @@ int dp_altmode_probe(struct typec_altmode *alt)
dp->alt = alt;
alt->desc = "DisplayPort";
- typec_altmode_set_ops(alt, &dp_altmode_ops);
if (plug) {
plug->desc = "Displayport";
@@ -811,6 +810,10 @@ int dp_altmode_probe(struct typec_altmode *alt)
if (plug)
typec_altmode_set_drvdata(plug, dp);
+ if ((alt->vdo & DP_CAP_RECEPTACLE) && typec_cable_altmode_unsupported(alt))
+ return 0;
+
+ typec_altmode_set_ops(alt, &dp_altmode_ops);
if (!alt->mode_selection) {
dp->state = plug ? DP_STATE_ENTER_PRIME : DP_STATE_ENTER;
schedule_work(&dp->work);
diff --git a/drivers/usb/typec/altmodes/thunderbolt.c b/drivers/usb/typec/altmodes/thunderbolt.c
index 32250b94262a..2eccdddf1b1f 100644
--- a/drivers/usb/typec/altmodes/thunderbolt.c
+++ b/drivers/usb/typec/altmodes/thunderbolt.c
@@ -284,6 +284,10 @@ static int tbt_altmode_probe(struct typec_altmode *alt)
alt->desc = "Thunderbolt3";
typec_altmode_set_drvdata(alt, tbt);
+
+ if (typec_cable_altmode_unsupported(alt))
+ return 0;
+
typec_altmode_set_ops(alt, &tbt_altmode_ops);
if (!alt->mode_selection && tbt_ready(alt)) {
diff --git a/drivers/usb/typec/class.c b/drivers/usb/typec/class.c
index 0595e8cb83aa..54c7810b563b 100644
--- a/drivers/usb/typec/class.c
+++ b/drivers/usb/typec/class.c
@@ -1429,6 +1429,77 @@ int typec_cable_is_active(struct typec_cable *cable)
}
EXPORT_SYMBOL_GPL(typec_cable_is_active);
+enum typec_cable_altmode_support {
+ CABLE_SUPPORT_UNKNOWN,
+ CABLE_SUPPORTED,
+ CABLE_NOT_SUPPORTED,
+};
+
+static enum typec_cable_altmode_support
+typec_cable_check_altmode_support(struct typec_cable *cable,
+ struct typec_altmode *alt)
+{
+ struct typec_altmode *plug;
+ u32 speed;
+
+ /*
+ * Check if the cable has an e-marker, supports modal operation, and the
+ * SOP' altmode nodes are created.
+ */
+ plug = typec_altmode_get_plug(alt, TYPEC_PLUG_SOP_P);
+ if (plug) {
+ typec_altmode_put_plug(plug);
+ return CABLE_SUPPORTED;
+ }
+
+ /* The identity is not specified */
+ if (!cable->identity)
+ return CABLE_SUPPORT_UNKNOWN;
+
+ /* Non-e-marked cable */
+ if (!cable->identity->id_header)
+ return CABLE_NOT_SUPPORTED;
+
+ switch (PD_IDH_PTYPE(cable->identity->id_header)) {
+ case IDH_PTYPE_PCABLE:
+ speed = VDO_TYPEC_CABLE_SPEED(cable->identity->vdo[0]);
+ if (speed == CABLE_USB2_ONLY)
+ return CABLE_NOT_SUPPORTED;
+ return CABLE_SUPPORTED;
+ case IDH_PTYPE_ACABLE:
+ /*
+ * Active cables must establish an SOP' communication
+ * node. Since that check failed at the beginning of
+ * this function, this active cable does not support
+ * this specific altmode.
+ */
+ return CABLE_NOT_SUPPORTED;
+ }
+
+ return CABLE_SUPPORT_UNKNOWN;
+}
+
+/**
+ * typec_cable_altmode_unsupported - Check if a cable restricts altmode
+ * @alt: The Alternate Mode to evaluate
+ *
+ * Returns true if the connected cable is incapable of handling the altmode.
+ */
+bool typec_cable_altmode_unsupported(struct typec_altmode *alt)
+{
+ enum typec_cable_altmode_support support = CABLE_SUPPORT_UNKNOWN;
+ struct typec_cable *cable;
+
+ cable = typec_cable_get(typec_altmode2port(alt));
+ if (cable) {
+ support = typec_cable_check_altmode_support(cable, alt);
+ typec_cable_put(cable);
+ }
+
+ return support == CABLE_NOT_SUPPORTED;
+}
+EXPORT_SYMBOL_GPL(typec_cable_altmode_unsupported);
+
/**
* typec_cable_set_identity - Report result from Discover Identity command
* @cable: The cable updated identity values
diff --git a/include/linux/usb/typec.h b/include/linux/usb/typec.h
index d61ec38216fa..10a783b738ef 100644
--- a/include/linux/usb/typec.h
+++ b/include/linux/usb/typec.h
@@ -337,6 +337,7 @@ void typec_unregister_cable(struct typec_cable *cable);
struct typec_cable *typec_cable_get(struct typec_port *port);
void typec_cable_put(struct typec_cable *cable);
int typec_cable_is_active(struct typec_cable *cable);
+bool typec_cable_altmode_unsupported(struct typec_altmode *alt);
struct typec_plug *typec_register_plug(struct typec_cable *cable,
struct typec_plug_desc *desc);
@@ -0,0 +1,29 @@
diff --git a/drivers/usb/core/message.c b/drivers/usb/core/message.c
--- a/drivers/usb/core/message.c
+++ b/drivers/usb/core/message.c
@@ -1052,6 +1052,25 @@ int usb_string(struct usb_device *dev, int index, char *buf, size_t size)
UTF16_LITTLE_ENDIAN, buf, size);
buf[err] = 0;
+ /*
+ * Some devices report string descriptors with a declared length
+ * greater than the actual serial, leaving uninitialized firmware
+ * memory (often including C0 control characters) appended to the
+ * returned string. Truncate at the first control character so
+ * callers get a clean, well-formed string.
+ */
+ {
+ int i;
+ for (i = 0; i < err; i++) {
+ unsigned char c = buf[i];
+ if (c < 0x20 || c == 0x7f) {
+ buf[i] = 0;
+ err = i;
+ break;
+ }
+ }
+ }
+
if (tbuf[1] != USB_DT_STRING)
dev_dbg(&dev->dev,
"wrong descriptor type %02x for string %d (\"%s\")\n",
@@ -0,0 +1,18 @@
diff --git a/drivers/net/wireguard/queueing.h b/drivers/net/wireguard/queueing.h
--- a/drivers/net/wireguard/queueing.h
+++ b/drivers/net/wireguard/queueing.h
@@ -78,12 +78,14 @@ static inline void wg_reset_packet(struct sk_buff *skb, bool encapsulating)
u8 l4_hash = skb->l4_hash;
u8 sw_hash = skb->sw_hash;
u32 hash = skb->hash;
+ u8 tstamp_type = skb->tstamp_type;
skb_scrub_packet(skb, true);
memset(&skb->headers, 0, sizeof(skb->headers));
if (encapsulating) {
skb->l4_hash = l4_hash;
skb->sw_hash = sw_hash;
skb->hash = hash;
+ skb->tstamp_type = tstamp_type;
}
skb->queue_mapping = 0;
skb->nohdr = 0;
@@ -0,0 +1,12 @@
diff --git a/drivers/bluetooth/btusb.c b/drivers/bluetooth/btusb.c
--- a/drivers/bluetooth/btusb.c
+++ b/drivers/bluetooth/btusb.c
@@ -801,6 +801,8 @@ static const struct usb_device_id quirks_table[] = {
BTUSB_WIDEBAND_SPEECH },
{ USB_DEVICE(0x13d3, 0x3613), .driver_info = BTUSB_MEDIATEK |
BTUSB_WIDEBAND_SPEECH },
+ { USB_DEVICE(0x13d3, 0x3625), .driver_info = BTUSB_MEDIATEK |
+ BTUSB_WIDEBAND_SPEECH },
{ USB_DEVICE(0x13d3, 0x3627), .driver_info = BTUSB_MEDIATEK |
BTUSB_WIDEBAND_SPEECH },
{ USB_DEVICE(0x13d3, 0x3628), .driver_info = BTUSB_MEDIATEK |
@@ -0,0 +1,32 @@
diff --git a/drivers/net/wireless/realtek/rtw89/fw.c b/drivers/net/wireless/realtek/rtw89/fw.c
--- a/drivers/net/wireless/realtek/rtw89/fw.c
+++ b/drivers/net/wireless/realtek/rtw89/fw.c
@@ -11780,7 +11780,7 @@ static void rtw89_fw_cmd_ofld_write_rf(struct rtw89_dev *rtwdev,
static void rtw89_fw_cmd_ofld_udelay(struct rtw89_dev *rtwdev, u32 us)
{
struct rtw89_fw_cmd_ofld_arg cmd = {
- .src = RTW89_FW_CMD_OFLD_SRC_OTHER,
+ .src = RTW89_FW_CMD_OFLD_SRC_BB,
.type = RTW89_FW_CMD_OFLD_DELAY,
.value = us,
};
@@ -11794,7 +11794,7 @@ static void rtw89_fw_cmd_ofld_udelay(struct rtw89_dev *rtwdev, u32 us)
static void rtw89_fw_cmd_ofld_mdelay(struct rtw89_dev *rtwdev, u32 ms)
{
struct rtw89_fw_cmd_ofld_arg cmd = {
- .src = RTW89_FW_CMD_OFLD_SRC_OTHER,
+ .src = RTW89_FW_CMD_OFLD_SRC_BB,
.type = RTW89_FW_CMD_OFLD_DELAY,
.value = ms * 1000,
};
diff --git a/drivers/net/wireless/realtek/rtw89/fw.h b/drivers/net/wireless/realtek/rtw89/fw.h
--- a/drivers/net/wireless/realtek/rtw89/fw.h
+++ b/drivers/net/wireless/realtek/rtw89/fw.h
@@ -3142,7 +3142,6 @@ enum rtw89_fw_cmd_ofld_arg_src {
RTW89_FW_CMD_OFLD_SRC_RF,
RTW89_FW_CMD_OFLD_SRC_MAC,
RTW89_FW_CMD_OFLD_SRC_RF_DDIE,
- RTW89_FW_CMD_OFLD_SRC_OTHER,
};
enum rtw89_fw_cmd_ofld_arg_type {
@@ -0,0 +1,27 @@
diff --git a/drivers/net/wireless/intel/iwlwifi/mld/mac80211.c b/drivers/net/wireless/intel/iwlwifi/mld/mac80211.c
--- a/drivers/net/wireless/intel/iwlwifi/mld/mac80211.c
+++ b/drivers/net/wireless/intel/iwlwifi/mld/mac80211.c
@@ -516,6 +516,10 @@ iwl_mld_mac80211_tx(struct ieee80211_hw *hw,
u32 link_id = u32_get_bits(info->control.flags,
IEEE80211_TX_CTRL_MLO_LINK);
+ if (unlikely(test_bit(STATUS_FW_ERROR, &mld->trans->status))) {
+ ieee80211_free_txskb(hw, skb);
+ return;
+ }
/* In AP mode, mgmt frames are sent on the bcast station,
* so the FW can't translate the MLD addr to the link addr. Do it here
*/
diff --git a/drivers/net/wireless/intel/iwlwifi/mld/tx.c b/drivers/net/wireless/intel/iwlwifi/mld/tx.c
--- a/drivers/net/wireless/intel/iwlwifi/mld/tx.c
+++ b/drivers/net/wireless/intel/iwlwifi/mld/tx.c
@@ -987,6 +987,9 @@ void iwl_mld_tx_from_txq(struct iwl_mld *mld, struct ieee80211_txq *txq)
struct sk_buff *skb = NULL;
u8 zero_addr[ETH_ALEN] = {};
+ if (unlikely(test_bit(STATUS_FW_ERROR, &mld->trans->status)))
+ return;
+
/*
* Don't transmit during firmware restart. The firmware is dead,
* so iwl_trans_tx() would return -EIO for each frame. Avoid the
@@ -0,0 +1,16 @@
diff --git a/drivers/pci/quirks.c b/drivers/pci/quirks.c
--- a/drivers/pci/quirks.c
+++ b/drivers/pci/quirks.c
@@ -108,7 +108,11 @@ int pcie_failed_link_retrain(struct pci_dev *dev)
pcie_capability_read_word(dev, PCI_EXP_LNKSTA, &lnksta);
pcie_capability_read_word(dev, PCI_EXP_LNKCTL2, &oldlnkctl2);
- if (!(lnksta & PCI_EXP_LNKSTA_DLLLA) && pcie_lbms_seen(dev, lnksta)) {
+ if (lnksta & PCI_EXP_LNKSTA_DLLLA) {
+ ;
+ } else if (PCIE_LNKCTL2_TLS2SPEED(oldlnkctl2) == PCIE_SPEED_2_5GT) {
+ return ret;
+ } else if (pcie_lbms_seen(dev, lnksta)) {
pci_info(dev, "broken device, retraining non-functional downstream link at 2.5GT/s\n");
ret = pcie_set_target_speed(dev, PCIE_SPEED_2_5GT, false);
if (ret)
@@ -0,0 +1,31 @@
diff --git a/drivers/hwmon/applesmc.c b/drivers/hwmon/applesmc.c
--- a/drivers/hwmon/applesmc.c
+++ b/drivers/hwmon/applesmc.c
@@ -33,6 +33,7 @@
#include <linux/workqueue.h>
#include <linux/err.h>
#include <linux/bits.h>
+#include <asm/barrier.h>
/* data port used by Apple SMC */
#define APPLESMC_DATA_PORT 0x300
@@ -372,7 +373,8 @@ static const struct applesmc_entry *applesmc_get_entry_by_index(int index)
__be32 be;
int ret = 0;
- if (cache->valid)
+ /* Pairs with smp_store_release() to ensure cache contents are visible */
+ if (smp_load_acquire(&cache->valid))
return cache;
mutex_lock(&smcreg.mutex);
@@ -391,7 +393,8 @@ static const struct applesmc_entry *applesmc_get_entry_by_index(int index)
cache->len = info[0];
memcpy(cache->type, &info[1], 4);
cache->flags = info[5];
- cache->valid = true;
+ /* Pairs with smp_load_acquire() to commit cache contents before setting valid */
+ smp_store_release(&cache->valid, true);
out:
mutex_unlock(&smcreg.mutex);
@@ -0,0 +1,22 @@
diff --git a/drivers/hwmon/applesmc.c b/drivers/hwmon/applesmc.c
--- a/drivers/hwmon/applesmc.c
+++ b/drivers/hwmon/applesmc.c
@@ -1252,12 +1252,17 @@ static void applesmc_release_light_sensor(void)
static int applesmc_create_key_backlight(void)
{
+ int ret;
+
if (!smcreg.has_key_backlight)
return 0;
applesmc_led_wq = create_singlethread_workqueue("applesmc-led");
if (!applesmc_led_wq)
return -ENOMEM;
- return led_classdev_register(&pdev->dev, &applesmc_backlight);
+ ret = led_classdev_register(&pdev->dev, &applesmc_backlight);
+ if (ret)
+ destroy_workqueue(applesmc_led_wq);
+ return ret;
}
static void applesmc_release_key_backlight(void)
@@ -0,0 +1,23 @@
diff --git a/drivers/acpi/power.c b/drivers/acpi/power.c
--- a/drivers/acpi/power.c
+++ b/drivers/acpi/power.c
@@ -1121,6 +1121,19 @@ static const struct dmi_system_id dmi_leave_unused_power_resources_on[] = {
DMI_MATCH(DMI_PRODUCT_NAME, "ZERO"),
}
+ },
+ {
+ /*
+ * Acer Swift 3 SF314-56G with NVIDIA MX250 suffers from the
+ * same ACPI power resource regression as the Thunderobot ZERO.
+ * The power resource controlling the dGPU is incorrectly turned off
+ * during initialization, causing the GPU to fall off the bus (D3cold).
+ */
+ .matches = {
+ DMI_MATCH(DMI_SYS_VENDOR, "Acer"),
+ DMI_MATCH(DMI_PRODUCT_NAME, "Swift SF314-56G"),
+ }
+
},
{}
};
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,34 @@
diff --git a/drivers/platform/x86/amd/pmf/util.c b/drivers/platform/x86/amd/pmf/util.c
index 1111111..2222222 100644
--- a/drivers/platform/x86/amd/pmf/util.c
+++ b/drivers/platform/x86/amd/pmf/util.c
@@ -79,9 +79,9 @@ static int amd_pmf_populate_data(struct amd_pmf_dev *pdev, struct amd_pmf_info *
static long amd_pmf_set_ioctl(struct file *filp, unsigned int cmd, unsigned long arg)
{
- struct amd_pmf_dev *pdev = filp->private_data;
void __user *argp = (void __user *)arg;
struct amd_pmf_info info = {};
+ struct amd_pmf_dev *pdev;
size_t copy_size;
__u64 user_size;
int ret;
@@ -94,6 +94,10 @@ static long amd_pmf_set_ioctl(struct file *filp, unsigned int cmd, unsigned long
return -EFAULT;
guard(mutex)(&pmf_util_lock);
+ pdev = pmf_dev_handle;
+ if (!pdev)
+ return -ENODEV;
+
ret = amd_pmf_populate_data(pdev, &info);
if (ret)
return ret;
@@ -115,7 +119,6 @@ static int amd_pmf_open(struct inode *inode, struct file *filp)
if (!pmf_dev_handle)
return -ENODEV;
- filp->private_data = pmf_dev_handle;
return 0;
}
@@ -0,0 +1,142 @@
diff --git a/include/uapi/linux/futex.h b/include/uapi/linux/futex.h
--- a/include/uapi/linux/futex.h
+++ b/include/uapi/linux/futex.h
@@ -22,6 +22,7 @@
#define FUTEX_WAIT_REQUEUE_PI 11
#define FUTEX_CMP_REQUEUE_PI 12
#define FUTEX_LOCK_PI2 13
+#define FUTEX_WAIT_MULTIPLE 31
#define FUTEX_PRIVATE_FLAG 128
#define FUTEX_CLOCK_REALTIME 256
@@ -126,6 +127,18 @@ struct futex_waitv {
__u32 __reserved;
};
+/**
+ * struct futex_wait_block - Block of futexes to be waited for
+ * @uaddr: User address of the futex
+ * @val: Futex value expected by userspace
+ * @bitset: Bitset for the optional bitmasked wakeup
+ */
+struct futex_wait_block {
+ __u32 __user *uaddr;
+ __u32 val;
+ __u32 bitset;
+};
+
/*
* Support for robust futexes: the kernel cleans up held futexes at
* thread exit time.
diff --git a/kernel/futex/syscalls.c b/kernel/futex/syscalls.c
--- a/kernel/futex/syscalls.c
+++ b/kernel/futex/syscalls.c
@@ -169,6 +169,7 @@ static __always_inline bool futex_cmd_has_timeout(u32 cmd)
case FUTEX_LOCK_PI2:
case FUTEX_WAIT_BITSET:
case FUTEX_WAIT_REQUEUE_PI:
+ case FUTEX_WAIT_MULTIPLE:
return true;
}
return false;
@@ -181,13 +182,79 @@ futex_init_timeout(u32 cmd, u32 op, struct timespec64 *ts, ktime_t *t)
return -EINVAL;
*t = timespec64_to_ktime(*ts);
- if (cmd == FUTEX_WAIT)
+ if (cmd == FUTEX_WAIT || cmd == FUTEX_WAIT_MULTIPLE)
*t = ktime_add_safe(ktime_get(), *t);
else if (cmd != FUTEX_LOCK_PI && !(op & FUTEX_CLOCK_REALTIME))
*t = timens_ktime_to_host(CLOCK_MONOTONIC, *t);
return 0;
}
+/**
+ * futex_read_wait_block - Read an array of futex_wait_block from userspace
+ * @uaddr: Userspace address of the block
+ * @count: Number of blocks to be read
+ *
+ * This function creates and allocate an array of futex_q (we zero it to
+ * initialize the fields) and then, for each futex_wait_block element from
+ * userspace, fill a futex_q element with proper values.
+ */
+inline struct futex_vector *futex_read_wait_block(u32 __user *uaddr, u32 count)
+{
+ unsigned int i;
+ struct futex_vector *futexv;
+ struct futex_wait_block fwb;
+ struct futex_wait_block __user *entry =
+ (struct futex_wait_block __user *)uaddr;
+
+ if (!count || count > FUTEX_WAITV_MAX)
+ return ERR_PTR(-EINVAL);
+
+ futexv = kcalloc(count, sizeof(*futexv), GFP_KERNEL);
+ if (!futexv)
+ return ERR_PTR(-ENOMEM);
+
+ for (i = 0; i < count; i++) {
+ if (copy_from_user(&fwb, &entry[i], sizeof(fwb))) {
+ kfree(futexv);
+ return ERR_PTR(-EFAULT);
+ }
+
+ futexv[i].w.flags = FUTEX_32;
+ futexv[i].w.val = fwb.val;
+ futexv[i].w.uaddr = (uintptr_t) (fwb.uaddr);
+ futexv[i].q = futex_q_init;
+ }
+
+ return futexv;
+}
+
+int futex_wait_multiple(struct futex_vector *vs, unsigned int count,
+ struct hrtimer_sleeper *to);
+
+int futex_opcode_31(ktime_t *abs_time, u32 __user *uaddr, int count)
+{
+ int ret;
+ struct futex_vector *vs;
+ struct hrtimer_sleeper *to = NULL, timeout;
+
+ to = futex_setup_timer(abs_time, &timeout, 0, 0);
+
+ vs = futex_read_wait_block(uaddr, count);
+
+ if (IS_ERR(vs))
+ return PTR_ERR(vs);
+
+ ret = futex_wait_multiple(vs, count, abs_time ? to : NULL);
+ kfree(vs);
+
+ if (to) {
+ hrtimer_cancel(&to->timer);
+ destroy_hrtimer_on_stack(&to->timer);
+ }
+
+ return ret;
+}
+
SYSCALL_DEFINE6(futex, u32 __user *, uaddr, int, op, u32, val,
const struct __kernel_timespec __user *, utime,
u32 __user *, uaddr2, u32, val3)
@@ -207,6 +274,9 @@ SYSCALL_DEFINE6(futex, u32 __user *, uaddr, int, op, u32, val,
tp = &t;
}
+ if (cmd == FUTEX_WAIT_MULTIPLE)
+ return futex_opcode_31(tp, uaddr, val);
+
return do_futex(uaddr, op, val, tp, uaddr2, (unsigned long)utime, val3);
}
@@ -521,6 +591,9 @@ SYSCALL_DEFINE6(futex_time32, u32 __user *, uaddr, int, op, u32, val,
tp = &t;
}
+ if (cmd == FUTEX_WAIT_MULTIPLE)
+ return futex_opcode_31(tp, uaddr, val);
+
return do_futex(uaddr, op, val, tp, uaddr2, (unsigned long)utime, val3);
}
#endif /* CONFIG_COMPAT_32BIT_TIME */
@@ -0,0 +1,35 @@
diff --git a/kernel/futex/syscalls.c b/kernel/futex/syscalls.c
--- a/kernel/futex/syscalls.c
+++ b/kernel/futex/syscalls.c
@@ -198,7 +198,7 @@
* initialize the fields) and then, for each futex_wait_block element from
* userspace, fill a futex_q element with proper values.
*/
-inline struct futex_vector *futex_read_wait_block(u32 __user *uaddr, u32 count)
+static inline struct futex_vector *futex_read_wait_block(u32 __user *uaddr, u32 count)
{
unsigned int i;
struct futex_vector *futexv;
@@ -231,19 +231,18 @@
int futex_wait_multiple(struct futex_vector *vs, unsigned int count,
struct hrtimer_sleeper *to);
-int futex_opcode_31(ktime_t *abs_time, u32 __user *uaddr, int count)
+static int futex_opcode_31(ktime_t *abs_time, u32 __user *uaddr, int count)
{
int ret;
struct futex_vector *vs;
struct hrtimer_sleeper *to = NULL, timeout;
- to = futex_setup_timer(abs_time, &timeout, 0, 0);
-
vs = futex_read_wait_block(uaddr, count);
-
if (IS_ERR(vs))
return PTR_ERR(vs);
+ to = futex_setup_timer(abs_time, &timeout, 0, 0);
+
ret = futex_wait_multiple(vs, count, abs_time ? to : NULL);
kfree(vs);
@@ -0,0 +1,88 @@
diff --git a/include/uapi/linux/futex.h b/include/uapi/linux/futex.h
--- a/include/uapi/linux/futex.h
+++ b/include/uapi/linux/futex.h
@@ -134,7 +134,7 @@
* @bitset: Bitset for the optional bitmasked wakeup
*/
struct futex_wait_block {
- __u32 __user *uaddr;
+ __aligned_u64 uaddr;
__u32 val;
__u32 bitset;
};
diff --git a/kernel/futex/syscalls.c b/kernel/futex/syscalls.c
--- a/kernel/futex/syscalls.c
+++ b/kernel/futex/syscalls.c
@@ -198,7 +198,8 @@
* initialize the fields) and then, for each futex_wait_block element from
* userspace, fill a futex_q element with proper values.
*/
-static inline struct futex_vector *futex_read_wait_block(u32 __user *uaddr, u32 count)
+static inline struct futex_vector *futex_read_wait_block(u32 __user *uaddr, u32 count,
+ unsigned int flags)
{
unsigned int i;
struct futex_vector *futexv;
@@ -219,10 +220,17 @@
return ERR_PTR(-EFAULT);
}
- futexv[i].w.flags = FUTEX_32;
+ if (!fwb.bitset) {
+ kfree(futexv);
+ return ERR_PTR(-EINVAL);
+ }
+
+ futexv[i].w.flags = flags;
futexv[i].w.val = fwb.val;
- futexv[i].w.uaddr = (uintptr_t) (fwb.uaddr);
+ futexv[i].w.uaddr = fwb.uaddr;
futexv[i].q = futex_q_init;
+ futexv[i].q.bitset = fwb.bitset;
+ futexv[i].q.wake = futex_wake_mark;
}
return futexv;
@@ -231,17 +239,21 @@
int futex_wait_multiple(struct futex_vector *vs, unsigned int count,
struct hrtimer_sleeper *to);
-static int futex_opcode_31(ktime_t *abs_time, u32 __user *uaddr, int count)
+static int futex_opcode_31(ktime_t *abs_time, u32 __user *uaddr, int count, int op)
{
int ret;
struct futex_vector *vs;
struct hrtimer_sleeper *to = NULL, timeout;
+ unsigned int flags = futex_to_flags(op);
+
+ if (flags & FLAGS_CLOCKRT)
+ return -ENOSYS;
- vs = futex_read_wait_block(uaddr, count);
+ vs = futex_read_wait_block(uaddr, count, flags);
if (IS_ERR(vs))
return PTR_ERR(vs);
- to = futex_setup_timer(abs_time, &timeout, 0, 0);
+ to = futex_setup_timer(abs_time, &timeout, flags, current->timer_slack_ns);
ret = futex_wait_multiple(vs, count, abs_time ? to : NULL);
kfree(vs);
@@ -274,7 +286,7 @@
}
if (cmd == FUTEX_WAIT_MULTIPLE)
- return futex_opcode_31(tp, uaddr, val);
+ return futex_opcode_31(tp, uaddr, val, op);
return do_futex(uaddr, op, val, tp, uaddr2, (unsigned long)utime, val3);
}
@@ -591,7 +603,7 @@
}
if (cmd == FUTEX_WAIT_MULTIPLE)
- return futex_opcode_31(tp, uaddr, val);
+ return futex_opcode_31(tp, uaddr, val, op);
return do_futex(uaddr, op, val, tp, uaddr2, (unsigned long)utime, val3);
}
@@ -0,0 +1,132 @@
diff --git c/drivers/gpu/drm/xe/xe_shrinker.c i/drivers/gpu/drm/xe/xe_shrinker.c
--- c/drivers/gpu/drm/xe/xe_shrinker.c
+++ i/drivers/gpu/drm/xe/xe_shrinker.c
@@ -54,13 +54,14 @@ xe_shrinker_mod_pages(struct xe_shrinker *shrinker, long shrinkable, long purgea
write_unlock(&shrinker->lock);
}
-static s64 __xe_shrinker_walk(struct xe_device *xe,
+static int __xe_shrinker_walk(struct xe_device *xe,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
- unsigned long to_scan, unsigned long *scanned)
+ unsigned long to_scan, unsigned long *scanned,
+ unsigned long *freed)
{
unsigned int mem_type;
- s64 freed = 0, lret;
+ s64 lret;
for (mem_type = XE_PL_SYSTEM; mem_type <= XE_PL_TT; ++mem_type) {
struct ttm_resource_manager *man = ttm_manager_type(&xe->ttm, mem_type);
@@ -82,7 +83,7 @@ static s64 __xe_shrinker_walk(struct xe_device *xe,
if (lret < 0)
return lret;
- freed += lret;
+ *freed += lret;
if (*scanned >= to_scan)
break;
}
@@ -90,7 +91,7 @@ static s64 __xe_shrinker_walk(struct xe_device *xe,
xe_assert(xe, !IS_ERR(ttm_bo));
}
- return freed;
+ return 0;
}
/*
@@ -99,40 +100,35 @@ static s64 __xe_shrinker_walk(struct xe_device *xe,
* add writeback. This avoids stalls and explicit writebacks with light or
* moderate memory pressure.
*/
-static s64 xe_shrinker_walk(struct xe_device *xe,
+static int xe_shrinker_walk(struct xe_device *xe,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
- unsigned long to_scan, unsigned long *scanned)
+ unsigned long to_scan, unsigned long *scanned,
+ unsigned long *freed)
{
bool no_wait_gpu = true;
struct xe_bo_shrink_flags save_flags = flags;
- s64 lret, freed;
+ int ret;
swap(no_wait_gpu, ctx->no_wait_gpu);
save_flags.writeback = false;
- lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned);
+ ret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned, freed);
swap(no_wait_gpu, ctx->no_wait_gpu);
- if (lret < 0 || *scanned >= to_scan)
- return lret;
+ if (ret || *scanned >= to_scan)
+ return ret;
- freed = lret;
if (!ctx->no_wait_gpu) {
- lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned);
- if (lret < 0)
- return lret;
- freed += lret;
- if (*scanned >= to_scan)
- return freed;
+ ret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned,
+ freed);
+ if (ret || *scanned >= to_scan)
+ return ret;
}
- if (flags.writeback) {
- lret = __xe_shrinker_walk(xe, ctx, flags, to_scan, scanned);
- if (lret < 0)
- return lret;
- freed += lret;
- }
+ if (flags.writeback)
+ ret = __xe_shrinker_walk(xe, ctx, flags, to_scan, scanned,
+ freed);
- return freed;
+ return ret;
}
static unsigned long
@@ -214,7 +210,6 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
bool runtime_pm;
bool purgeable;
bool can_backup = !!(sc->gfp_mask & __GFP_FS);
- s64 lret;
nr_to_scan = sc->nr_to_scan;
@@ -225,12 +220,9 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
/* Might need runtime PM. Try to wake early if it looks like it. */
runtime_pm = xe_shrinker_runtime_pm_get(shrinker, false, nr_to_scan, can_backup);
- if (purgeable && nr_scanned < nr_to_scan) {
- lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
- nr_to_scan, &nr_scanned);
- if (lret >= 0)
- freed += lret;
- }
+ if (purgeable && nr_scanned < nr_to_scan)
+ xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
+ nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
if (nr_scanned >= nr_to_scan || !can_backup)
@@ -242,10 +234,8 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
shrink_flags.purge = false;
- lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
- nr_to_scan, &nr_scanned);
- if (lret >= 0)
- freed += lret;
+ xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
+ nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
out:
@@ -0,0 +1,149 @@
diff --git c/drivers/gpu/drm/xe/xe_shrinker.c i/drivers/gpu/drm/xe/xe_shrinker.c
--- c/drivers/gpu/drm/xe/xe_shrinker.c
+++ i/drivers/gpu/drm/xe/xe_shrinker.c
@@ -54,13 +54,33 @@ xe_shrinker_mod_pages(struct xe_shrinker *shrinker, long shrinkable, long purgea
write_unlock(&shrinker->lock);
}
-static int __xe_shrinker_walk(struct xe_device *xe,
+static bool __xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker)
+{
+ struct xe_device *xe = shrinker->xe;
+
+ if (xe_pm_runtime_get_if_active(xe))
+ return true;
+
+ if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) {
+ xe_pm_runtime_get(xe);
+ return true;
+ }
+
+ queue_work(xe->unordered_wq, &shrinker->pm_worker);
+
+ return false;
+}
+
+static int __xe_shrinker_walk(struct xe_shrinker *shrinker,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
unsigned long to_scan, unsigned long *scanned,
unsigned long *freed)
{
+ struct xe_device *xe = shrinker->xe;
unsigned int mem_type;
+ bool rpm = false;
+ int ret = 0;
s64 lret;
for (mem_type = XE_PL_SYSTEM; mem_type <= XE_PL_TT; ++mem_type) {
@@ -75,23 +95,36 @@ static int __xe_shrinker_walk(struct xe_device *xe,
if (!man || !man->use_tt)
continue;
+ if (mem_type != XE_PL_SYSTEM && !rpm &&
+ xe_device_is_l2_flush_optimized(xe)) {
+ if (!__xe_shrinker_runtime_pm_get(shrinker))
+ break;
+ rpm = true;
+ }
+
ttm_bo_lru_for_each_reserved_guarded(&curs, man, &arg, ttm_bo) {
if (!ttm_bo_shrink_suitable(ttm_bo, ctx))
continue;
lret = xe_bo_shrink(ctx, ttm_bo, flags, scanned);
- if (lret < 0)
- return lret;
+ if (lret < 0) {
+ ret = lret;
+ goto out;
+ }
*freed += lret;
if (*scanned >= to_scan)
- break;
+ goto out;
}
/* Trylocks should never error, just fail. */
xe_assert(xe, !IS_ERR(ttm_bo));
}
- return 0;
+out:
+ if (rpm)
+ xe_pm_runtime_put(xe);
+
+ return ret;
}
/*
@@ -100,7 +133,7 @@ static int __xe_shrinker_walk(struct xe_device *xe,
* add writeback. This avoids stalls and explicit writebacks with light or
* moderate memory pressure.
*/
-static int xe_shrinker_walk(struct xe_device *xe,
+static int xe_shrinker_walk(struct xe_shrinker *shrinker,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
unsigned long to_scan, unsigned long *scanned,
@@ -112,20 +145,21 @@ static int xe_shrinker_walk(struct xe_device *xe,
swap(no_wait_gpu, ctx->no_wait_gpu);
save_flags.writeback = false;
- ret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned, freed);
+ ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned,
+ freed);
swap(no_wait_gpu, ctx->no_wait_gpu);
if (ret || *scanned >= to_scan)
return ret;
if (!ctx->no_wait_gpu) {
- ret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned,
+ ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned,
freed);
if (ret || *scanned >= to_scan)
return ret;
}
if (flags.writeback)
- ret = __xe_shrinker_walk(xe, ctx, flags, to_scan, scanned,
+ ret = __xe_shrinker_walk(shrinker, ctx, flags, to_scan, scanned,
freed);
return ret;
@@ -176,16 +210,7 @@ static bool xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker, bool force,
return false;
}
- if (!xe_pm_runtime_get_if_active(xe)) {
- if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) {
- xe_pm_runtime_get(xe);
- return true;
- }
- queue_work(xe->unordered_wq, &shrinker->pm_worker);
- return false;
- }
-
- return true;
+ return __xe_shrinker_runtime_pm_get(shrinker);
}
static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm)
@@ -221,7 +246,7 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
runtime_pm = xe_shrinker_runtime_pm_get(shrinker, false, nr_to_scan, can_backup);
if (purgeable && nr_scanned < nr_to_scan)
- xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
+ xe_shrinker_walk(shrinker, &ctx, shrink_flags,
nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
@@ -234,7 +259,7 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
shrink_flags.purge = false;
- xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
+ xe_shrinker_walk(shrinker, &ctx, shrink_flags,
nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
@@ -0,0 +1,39 @@
diff --git c/drivers/gpu/drm/xe/xe_shrinker.c i/drivers/gpu/drm/xe/xe_shrinker.c
--- c/drivers/gpu/drm/xe/xe_shrinker.c
+++ i/drivers/gpu/drm/xe/xe_shrinker.c
@@ -71,6 +71,12 @@ static bool __xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker)
return false;
}
+static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm)
+{
+ if (runtime_pm)
+ xe_pm_runtime_put(shrinker->xe);
+}
+
static int __xe_shrinker_walk(struct xe_shrinker *shrinker,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
@@ -121,8 +127,7 @@ static int __xe_shrinker_walk(struct xe_shrinker *shrinker,
}
out:
- if (rpm)
- xe_pm_runtime_put(xe);
+ xe_shrinker_runtime_pm_put(shrinker, rpm);
return ret;
}
@@ -213,12 +218,6 @@ static bool xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker, bool force,
return __xe_shrinker_runtime_pm_get(shrinker);
}
-static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm)
-{
- if (runtime_pm)
- xe_pm_runtime_put(shrinker->xe);
-}
-
static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc)
{
struct xe_shrinker *shrinker = to_xe_shrinker(shrink);
@@ -0,0 +1,381 @@
diff --git c/include/linux/mmzone.h i/include/linux/mmzone.h
--- c/include/linux/mmzone.h
+++ i/include/linux/mmzone.h
@@ -1465,6 +1465,39 @@ struct memory_failure_stats {
};
#endif
+/*
+ * Per-pgdat state machine for the kswapd "opportunistic compaction" hint.
+ *
+ * wakeup_kswapd() collapses the gfp flags of all wakers that arrive between
+ * two kswapd runs into a single tri-state, which kswapd then forwards to the
+ * shrinkers via shrink_control::opportunistic_compaction:
+ *
+ * KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION
+ * Initial state after kswapd consumes the previous value. No waker has
+ * been observed yet for the upcoming run.
+ *
+ * KSWAPD_NO_OPPORTUNISTIC_COMPACTION
+ * At least one waker is an order-0 allocation, or a high-order
+ * allocation that cannot tolerate failure (i.e., not eligible for
+ * opportunistic behaviour). Shrinkers must do their normal best-effort
+ * work; the hint is cleared.
+ *
+ * KSWAPD_OPPORTUNISTIC_COMPACTION
+ * All wakers seen so far are high-order allocations that may fail
+ * (__GFP_NORETRY or __GFP_RETRY_MAYFAIL, without __GFP_NOFAIL). Shrinkers
+ * may skip work that is unlikely to produce a contiguous high-order
+ * block (e.g., evicting working-set pages).
+ *
+ * The state is sticky in the "NO" direction within a single kswapd run: once
+ * any non-eligible waker is observed, subsequent eligible wakers cannot
+ * upgrade it back to KSWAPD_OPPORTUNISTIC_COMPACTION.
+ */
+enum kswapd_opportunistic_compaction_type {
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION = 0,
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION,
+ KSWAPD_OPPORTUNISTIC_COMPACTION,
+};
+
/*
* On NUMA machines, each NUMA node would have a pg_data_t to describe
* it's memory layout. On UMA machines there is a single pglist_data which
@@ -1529,6 +1562,13 @@ typedef struct pglist_data {
#endif
struct task_struct *kswapd; /* Protected by kswapd_lock */
int kswapd_order;
+ /*
+ * Aggregated opportunistic-compaction hint for the next kswapd run.
+ * Updated by wakeup_kswapd() based on the gfp flags / order of each
+ * waker, and consumed (and reset) by kswapd before balance_pgdat().
+ * See enum kswapd_opportunistic_compaction_type for the state machine.
+ */
+ atomic_t kswapd_opportunistic_compaction;
enum zone_type kswapd_highest_zoneidx;
atomic_t kswapd_failures; /* Number of 'reclaimed == 0' runs */
diff --git c/include/linux/shrinker.h i/include/linux/shrinker.h
--- c/include/linux/shrinker.h
+++ i/include/linux/shrinker.h
@@ -37,6 +37,26 @@ struct shrink_control {
/* current node being shrunk (for NUMA aware shrinkers) */
int nid;
+ /*
+ * Opportunistic compaction hint.
+ *
+ * Set by the reclaim path to tell shrinkers that this pass is
+ * driven by an order > 0 allocation that the caller is willing to
+ * have fail (e.g., __GFP_NORETRY / __GFP_RETRY_MAYFAIL without
+ * __GFP_NOFAIL). Such allocations only really benefit from
+ * shrinking when doing so frees up a contiguous, high-order block;
+ * thrashing working sets in the hope of producing one is typically
+ * counter-productive.
+ *
+ * Shrinkers that can produce naturally-aligned high-order folios
+ * (see shrink_control::order) should treat this as a hint to skip
+ * costly work that is unlikely to help compaction (for example,
+ * evicting hot/working-set pages just to free single pages).
+ *
+ * Only meaningful when @order > 0; ignored otherwise.
+ */
+ bool opportunistic_compaction;
+
/*
* How many objects scan_objects should scan and try to reclaim.
* This is reset before every call, so it is safe for callees
diff --git c/mm/internal.h i/mm/internal.h
--- c/mm/internal.h
+++ i/mm/internal.h
@@ -1767,7 +1767,7 @@ void __meminit __init_page_from_nid(unsigned long pfn, int nid);
/* shrinker related functions */
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
- int priority);
+ int priority, bool opportunistic_compaction);
int shmem_add_to_page_cache(struct folio *folio,
struct address_space *mapping,
diff --git c/mm/shrinker.c i/mm/shrinker.c
--- c/mm/shrinker.c
+++ i/mm/shrinker.c
@@ -474,7 +474,7 @@ static unsigned long do_shrink_slab(struct shrink_control *shrinkctl,
#ifdef CONFIG_MEMCG
static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
- struct mem_cgroup *memcg, int priority)
+ struct mem_cgroup *memcg, int priority, bool opportunistic_compaction)
{
struct shrinker_info *info;
unsigned long ret, freed = 0;
@@ -536,6 +536,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
.gfp_mask = gfp_mask,
.nid = nid,
.memcg = memcg,
+ .opportunistic_compaction = opportunistic_compaction,
};
struct shrinker *shrinker;
int shrinker_id = calc_shrinker_id(index, offset);
@@ -595,7 +596,8 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
}
#else /* !CONFIG_MEMCG */
static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
- struct mem_cgroup *memcg, int priority)
+ struct mem_cgroup *memcg, int priority,
+ bool opportunistic_compaction)
{
return 0;
}
@@ -607,6 +609,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
* @nid: node whose slab caches to target
* @memcg: memory cgroup whose slab caches to target
* @priority: the reclaim priority
+ * @opportunistic_compaction: do compaction opportunistically (e.g., do not swap working sets)
*
* Call the shrink functions to age shrinkable caches.
*
@@ -622,7 +625,7 @@ static unsigned long shrink_slab_memcg(gfp_t gfp_mask, int nid,
* Returns the number of reclaimed slab objects.
*/
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
- int priority)
+ int priority, bool opportunistic_compaction)
{
unsigned long ret, freed = 0;
struct shrinker *shrinker;
@@ -635,7 +638,8 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
* oom.
*/
if (!mem_cgroup_disabled() && !mem_cgroup_is_root(memcg))
- return shrink_slab_memcg(gfp_mask, nid, memcg, priority);
+ return shrink_slab_memcg(gfp_mask, nid, memcg, priority,
+ opportunistic_compaction);
/*
* lockless algorithm of global shrink.
@@ -664,6 +668,7 @@ unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
.gfp_mask = gfp_mask,
.nid = nid,
.memcg = memcg,
+ .opportunistic_compaction = opportunistic_compaction,
};
if (!shrinker_try_get(shrinker))
diff --git c/mm/vmscan.c i/mm/vmscan.c
--- c/mm/vmscan.c
+++ i/mm/vmscan.c
@@ -96,6 +96,14 @@ struct scan_control {
/* Swappiness value for proactive reclaim. Always use sc_swappiness()! */
int *proactive_swappiness;
+ /*
+ * Opportunistic compaction hint snapshotted from the pgdat at the
+ * start of this reclaim pass. Forwarded to shrinkers through
+ * shrink_control::opportunistic_compaction so they can skip
+ * non-productive work for failable high-order allocations.
+ */
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
+
/* Can active folios be deactivated as part of reclaim? */
#define DEACTIVATE_ANON 1
#define DEACTIVATE_FILE 2
@@ -198,6 +206,29 @@ struct scan_control {
*/
int vm_swappiness = 60;
+/*
+ * Is @gfp_flags a high-order allocation that is eligible for the
+ * "opportunistic compaction" treatment in kswapd / shrinkers?
+ *
+ * The caller must be willing to tolerate failure (__GFP_NORETRY or
+ * __GFP_RETRY_MAYFAIL) and must not have set __GFP_NOFAIL. For such
+ * allocations there is little value in burning working-set pages just to
+ * scrape together a single high-order block: if compaction can't easily
+ * succeed, the caller would rather see the allocation fail.
+ */
+static bool gfp_opportunistic_compaction(gfp_t gfp_flags)
+{
+ return (gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL)) &&
+ !(gfp_flags & __GFP_NOFAIL);
+}
+
+static bool sc_opportunistic_compaction(struct scan_control *sc)
+{
+ return sc->order && (sc->kswapd_opportunistic_compaction ==
+ KSWAPD_OPPORTUNISTIC_COMPACTION || (!current_is_kswapd() &&
+ gfp_opportunistic_compaction(sc->gfp_mask)));
+}
+
static int sc_swappiness(struct scan_control *sc, struct mem_cgroup *memcg)
{
if (sc->proactive && sc->proactive_swappiness)
@@ -411,7 +442,7 @@ static unsigned long drop_slab_node(int nid)
memcg = mem_cgroup_iter(NULL, NULL, NULL);
do {
- freed += shrink_slab(GFP_KERNEL, nid, memcg, 0);
+ freed += shrink_slab(GFP_KERNEL, nid, memcg, 0, false);
} while ((memcg = mem_cgroup_iter(NULL, memcg, NULL)) != NULL);
return freed;
@@ -5081,6 +5112,7 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc)
unsigned long reclaimed = sc->nr_reclaimed;
struct mem_cgroup *memcg = lruvec_memcg(lruvec);
struct pglist_data *pgdat = lruvec_pgdat(lruvec);
+ bool opportunistic_compaction = sc_opportunistic_compaction(sc);
/* lru_gen_age_node() called mem_cgroup_calculate_protection() */
if (mem_cgroup_below_min(NULL, memcg))
@@ -5096,7 +5128,8 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc)
need_rotate = try_to_shrink_lruvec(lruvec, sc);
- shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority);
+ shrink_slab(sc->gfp_mask, pgdat->node_id, memcg, sc->priority,
+ opportunistic_compaction);
if (!sc->proactive)
vmpressure(sc->gfp_mask, sc->order, memcg, false,
@@ -6163,6 +6196,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc)
struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat);
unsigned long reclaimed;
unsigned long scanned;
+ bool opportunistic_compaction = sc_opportunistic_compaction(sc);
/*
* This loop can become CPU-bound when target memcgs
@@ -6200,7 +6234,7 @@ static void shrink_node_memcgs(pg_data_t *pgdat, struct scan_control *sc)
shrink_lruvec(lruvec, sc);
shrink_slab(sc->gfp_mask, pgdat->node_id, memcg,
- sc->priority);
+ sc->priority, opportunistic_compaction);
/* Record the group's reclaim efficiency */
if (!sc->proactive)
@@ -7133,8 +7167,14 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx)
* found to have free_pages <= high_wmark_pages(zone), any page in that zone
* or lower is eligible for reclaim until at least one usable zone is
* balanced.
+ *
+ * @kswapd_opportunistic_compaction is the aggregated hint produced by
+ * wakeup_kswapd() for this run; it is propagated into scan_control so that
+ * shrinkers can skip costly work that is unlikely to help compaction when
+ * all wakers are failable high-order allocations.
*/
-static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx)
+static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx,
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction)
{
int i;
unsigned long nr_soft_reclaimed;
@@ -7148,6 +7188,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx)
.gfp_mask = GFP_KERNEL,
.order = order,
.may_unmap = 1,
+ .kswapd_opportunistic_compaction = kswapd_opportunistic_compaction,
};
trace_mm_vmscan_balance_pgdat_begin(pgdat->node_id, order,
@@ -7372,8 +7413,10 @@ static enum zone_type kswapd_highest_zoneidx(pg_data_t *pgdat,
return curr_idx == MAX_NR_ZONES ? prev_highest_zoneidx : curr_idx;
}
-static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
- unsigned int highest_zoneidx)
+static void
+kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
+ unsigned int highest_zoneidx,
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction)
{
long remaining = 0;
DEFINE_WAIT(wait);
@@ -7419,6 +7462,11 @@ static void kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_o
if (READ_ONCE(pgdat->kswapd_order) < reclaim_order)
WRITE_ONCE(pgdat->kswapd_order, reclaim_order);
+
+ if (kswapd_opportunistic_compaction ==
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION)
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
}
finish_wait(&pgdat->kswapd_wait, &wait);
@@ -7475,6 +7523,7 @@ static int kswapd(void *p)
unsigned int highest_zoneidx = MAX_NR_ZONES - 1;
pg_data_t *pgdat = (pg_data_t *)p;
struct task_struct *tsk = current;
+ enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
/*
* Tell the memory management that we're a "memory allocator",
@@ -7492,6 +7541,8 @@ static int kswapd(void *p)
set_freezable();
WRITE_ONCE(pgdat->kswapd_order, 0);
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
WRITE_ONCE(pgdat->kswapd_highest_zoneidx, MAX_NR_ZONES);
atomic_set(&pgdat->nr_writeback_throttled, 0);
for ( ; ; ) {
@@ -7500,13 +7551,18 @@ static int kswapd(void *p)
alloc_order = reclaim_order = READ_ONCE(pgdat->kswapd_order);
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
highest_zoneidx);
+ kswapd_opportunistic_compaction =
+ atomic_read(&pgdat->kswapd_opportunistic_compaction);
kswapd_try_sleep:
kswapd_try_to_sleep(pgdat, alloc_order, reclaim_order,
- highest_zoneidx);
+ highest_zoneidx, kswapd_opportunistic_compaction);
/* Read the new order and highest_zoneidx */
alloc_order = READ_ONCE(pgdat->kswapd_order);
+ kswapd_opportunistic_compaction =
+ atomic_xchg(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
highest_zoneidx);
WRITE_ONCE(pgdat->kswapd_order, 0);
@@ -7533,7 +7589,8 @@ static int kswapd(void *p)
trace_mm_vmscan_kswapd_wake(pgdat->node_id, highest_zoneidx,
alloc_order);
reclaim_order = balance_pgdat(pgdat, alloc_order,
- highest_zoneidx);
+ highest_zoneidx,
+ kswapd_opportunistic_compaction);
if (reclaim_order < alloc_order)
goto kswapd_try_sleep;
}
@@ -7571,6 +7628,28 @@ void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order,
if (READ_ONCE(pgdat->kswapd_order) < order)
WRITE_ONCE(pgdat->kswapd_order, order);
+ /*
+ * Fold this waker into the per-pgdat opportunistic-compaction hint
+ * that kswapd will pick up at the start of its next run.
+ *
+ * The state is sticky in the "NO" direction: once any waker in this
+ * batch is order-0 or a non-failable high-order allocation, the hint
+ * stays cleared until kswapd consumes it. Only when every waker so
+ * far is a failable high-order allocation do we set
+ * KSWAPD_OPPORTUNISTIC_COMPACTION, asking shrinkers to skip work
+ * that won't realistically help compaction.
+ */
+ if (atomic_read(&pgdat->kswapd_opportunistic_compaction) !=
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION) {
+ if (!order || !gfp_opportunistic_compaction(gfp_flags))
+ atomic_set(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
+ else if (order && gfp_opportunistic_compaction(gfp_flags))
+ atomic_cmpxchg(&pgdat->kswapd_opportunistic_compaction,
+ KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION,
+ KSWAPD_OPPORTUNISTIC_COMPACTION);
+ }
+
if (!waitqueue_active(&pgdat->kswapd_wait))
return;
@@ -0,0 +1,236 @@
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
--- a/include/linux/mmzone.h
+++ b/include/linux/mmzone.h
@@ -1680,7 +1680,7 @@ enum kswapd_clear_hopeless_reason {
};
void wakeup_kswapd(struct zone *zone, gfp_t gfp_mask, int order,
- enum zone_type highest_zoneidx);
+ bool opportunistic, enum zone_type highest_zoneidx);
void kswapd_try_clear_hopeless(struct pglist_data *pgdat,
unsigned int order, int highest_zoneidx);
void kswapd_clear_hopeless(pg_data_t *pgdat, enum kswapd_clear_hopeless_reason reason);
diff --git a/include/linux/swap.h b/include/linux/swap.h
--- a/include/linux/swap.h
+++ b/include/linux/swap.h
@@ -348,7 +348,8 @@ extern void swap_setup(void);
/* linux/mm/vmscan.c */
extern unsigned long zone_reclaimable_pages(struct zone *zone);
extern unsigned long try_to_free_pages(struct zonelist *zonelist, int order,
- gfp_t gfp_mask, nodemask_t *mask);
+ int alloc_order, gfp_t gfp_mask,
+ nodemask_t *mask);
unsigned long lruvec_lru_size(struct lruvec *lruvec, enum lru_list lru, int zone_idx);
#define MEMCG_RECLAIM_MAY_SWAP (1 << 1)
diff --git a/mm/internal.h b/mm/internal.h
--- a/mm/internal.h
+++ b/mm/internal.h
@@ -1765,6 +1765,28 @@ void __meminit __init_single_page(struct page *page, unsigned long pfn,
unsigned long zone, int nid);
void __meminit __init_page_from_nid(unsigned long pfn, int nid);
+/*
+ * Is this allocation eligible for the "opportunistic compaction" treatment
+ * in kswapd and the shrinkers?
+ *
+ * The caller must be willing to tolerate failure (__GFP_NORETRY or
+ * __GFP_RETRY_MAYFAIL) and must not have set __GFP_NOFAIL. For such
+ * allocations there is little value in burning working-set pages just to
+ * scrape together a single high-order block: if compaction can't easily
+ * succeed, the caller would rather see the allocation fail.
+ *
+ * @order is the order the caller asked for, not the order reclaim was
+ * promoted to. defrag_mode raises the reclaim order to pageblock_order,
+ * which says nothing about what the caller can tolerate.
+ */
+static inline bool gfp_opportunistic_compaction(gfp_t gfp_flags, int order)
+{
+ if (!order || (gfp_flags & __GFP_NOFAIL))
+ return false;
+
+ return !!(gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL));
+}
+
/* shrinker related functions */
unsigned long shrink_slab(gfp_t gfp_mask, int nid, struct mem_cgroup *memcg,
int priority, bool opportunistic_compaction);
diff --git a/mm/migrate.c b/mm/migrate.c
--- a/mm/migrate.c
+++ b/mm/migrate.c
@@ -2727,7 +2727,7 @@ int migrate_misplaced_folio_prepare(struct folio *folio,
return -EAGAIN;
wakeup_kswapd(pgdat->node_zones + z, 0,
- folio_order(folio), ZONE_MOVABLE);
+ folio_order(folio), false, ZONE_MOVABLE);
return -EAGAIN;
}
diff --git a/mm/page_alloc.c b/mm/page_alloc.c
--- a/mm/page_alloc.c
+++ b/mm/page_alloc.c
@@ -3422,7 +3422,7 @@ struct page *rmqueue(struct zone *preferred_zone,
if ((alloc_flags & ALLOC_KSWAPD) &&
unlikely(test_bit(ZONE_BOOSTED_WATERMARK, &zone->flags))) {
clear_bit(ZONE_BOOSTED_WATERMARK, &zone->flags);
- wakeup_kswapd(zone, 0, 0, zone_idx(zone));
+ wakeup_kswapd(zone, 0, 0, false, zone_idx(zone));
}
VM_BUG_ON_PAGE(page && bad_range(zone, page), page);
@@ -4441,6 +4441,7 @@ static unsigned int check_retry_zonelist(unsigned int seq)
/* Perform direct synchronous page reclaim */
static unsigned long
__perform_reclaim(gfp_t gfp_mask, unsigned int order,
+ unsigned int alloc_order,
const struct alloc_context *ac)
{
unsigned int noreclaim_flag;
@@ -4453,7 +4454,7 @@ __perform_reclaim(gfp_t gfp_mask, unsigned int order,
fs_reclaim_acquire(gfp_mask);
noreclaim_flag = memalloc_noreclaim_save();
- progress = try_to_free_pages(ac->zonelist, order, gfp_mask,
+ progress = try_to_free_pages(ac->zonelist, order, alloc_order, gfp_mask,
ac->nodemask);
memalloc_noreclaim_restore(noreclaim_flag);
@@ -4480,7 +4481,7 @@ __alloc_pages_direct_reclaim(gfp_t gfp_mask, unsigned int order,
reclaim_order = max(order, pageblock_order);
psi_memstall_enter(&pflags);
- *did_some_progress = __perform_reclaim(gfp_mask, reclaim_order, ac);
+ *did_some_progress = __perform_reclaim(gfp_mask, reclaim_order, order, ac);
if (unlikely(!(*did_some_progress)))
goto out;
@@ -4512,19 +4513,24 @@ static void wake_all_kswapds(unsigned int order, gfp_t gfp_mask,
pg_data_t *last_pgdat = NULL;
enum zone_type highest_zoneidx = ac->highest_zoneidx;
unsigned int reclaim_order;
+ bool opportunistic;
if (defrag_mode)
reclaim_order = max(order, pageblock_order);
else
reclaim_order = order;
+ /* Classify what the caller asked for, not what reclaim was raised to. */
+ opportunistic = gfp_opportunistic_compaction(gfp_mask, order);
+
for_each_zone_zonelist_nodemask(zone, z, ac->zonelist, highest_zoneidx,
ac->nodemask) {
if (!managed_zone(zone))
continue;
if (last_pgdat == zone->zone_pgdat)
continue;
- wakeup_kswapd(zone, gfp_mask, reclaim_order, highest_zoneidx);
+ wakeup_kswapd(zone, gfp_mask, reclaim_order, opportunistic,
+ highest_zoneidx);
last_pgdat = zone->zone_pgdat;
}
}
diff --git a/mm/vmscan.c b/mm/vmscan.c
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -160,6 +160,12 @@ struct scan_control {
/* Allocation order */
s8 order;
+ /*
+ * The order the allocation asked for. @order above may have been
+ * raised to pageblock_order by defrag_mode.
+ */
+ s8 alloc_order;
+
/* Scan (total_size >> priority) pages at once */
s8 priority;
@@ -206,27 +212,19 @@ struct scan_control {
*/
int vm_swappiness = 60;
-/*
- * Is @gfp_flags a high-order allocation that is eligible for the
- * "opportunistic compaction" treatment in kswapd / shrinkers?
- *
- * The caller must be willing to tolerate failure (__GFP_NORETRY or
- * __GFP_RETRY_MAYFAIL) and must not have set __GFP_NOFAIL. For such
- * allocations there is little value in burning working-set pages just to
- * scrape together a single high-order block: if compaction can't easily
- * succeed, the caller would rather see the allocation fail.
- */
-static bool gfp_opportunistic_compaction(gfp_t gfp_flags)
-{
- return (gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL)) &&
- !(gfp_flags & __GFP_NOFAIL);
-}
-
static bool sc_opportunistic_compaction(struct scan_control *sc)
{
- return sc->order && (sc->kswapd_opportunistic_compaction ==
- KSWAPD_OPPORTUNISTIC_COMPACTION || (!current_is_kswapd() &&
- gfp_opportunistic_compaction(sc->gfp_mask)));
+ if (!sc->order)
+ return false;
+
+ if (sc->kswapd_opportunistic_compaction ==
+ KSWAPD_OPPORTUNISTIC_COMPACTION)
+ return true;
+
+ if (current_is_kswapd())
+ return false;
+
+ return gfp_opportunistic_compaction(sc->gfp_mask, sc->alloc_order);
}
static int sc_swappiness(struct scan_control *sc, struct mem_cgroup *memcg)
@@ -6781,7 +6779,8 @@ static bool throttle_direct_reclaim(gfp_t gfp_mask, struct zonelist *zonelist,
}
unsigned long try_to_free_pages(struct zonelist *zonelist, int order,
- gfp_t gfp_mask, nodemask_t *nodemask)
+ int alloc_order, gfp_t gfp_mask,
+ nodemask_t *nodemask)
{
unsigned long nr_reclaimed;
struct scan_control sc = {
@@ -6789,6 +6788,7 @@ unsigned long try_to_free_pages(struct zonelist *zonelist, int order,
.gfp_mask = current_gfp_context(gfp_mask),
.reclaim_idx = gfp_zone(gfp_mask),
.order = order,
+ .alloc_order = alloc_order,
.nodemask = nodemask,
.priority = DEF_PRIORITY,
.may_writepage = 1,
@@ -7608,7 +7608,7 @@ static int kswapd(void *p)
* needed.
*/
void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order,
- enum zone_type highest_zoneidx)
+ bool opportunistic, enum zone_type highest_zoneidx)
{
pg_data_t *pgdat;
enum zone_type curr_idx;
@@ -7641,10 +7641,10 @@ void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order,
*/
if (atomic_read(&pgdat->kswapd_opportunistic_compaction) !=
KSWAPD_NO_OPPORTUNISTIC_COMPACTION) {
- if (!order || !gfp_opportunistic_compaction(gfp_flags))
+ if (!opportunistic)
atomic_set(&pgdat->kswapd_opportunistic_compaction,
KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
- else if (order && gfp_opportunistic_compaction(gfp_flags))
+ else
atomic_cmpxchg(&pgdat->kswapd_opportunistic_compaction,
KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION,
KSWAPD_OPPORTUNISTIC_COMPACTION);
@@ -7942,6 +7942,7 @@ int node_reclaim(struct pglist_data *pgdat, gfp_t gfp_mask, unsigned int order)
.nr_to_reclaim = max(nr_pages, SWAP_CLUSTER_MAX),
.gfp_mask = current_gfp_context(gfp_mask),
.order = order,
+ .alloc_order = order,
.priority = NODE_RECLAIM_PRIORITY,
.may_writepage = !!(node_reclaim_mode & RECLAIM_WRITE),
.may_unmap = !!(node_reclaim_mode & RECLAIM_UNMAP),
@@ -0,0 +1,232 @@
diff --git c/include/linux/mmzone.h i/include/linux/mmzone.h
--- c/include/linux/mmzone.h
+++ i/include/linux/mmzone.h
@@ -1498,6 +1498,32 @@ enum kswapd_opportunistic_compaction_type {
KSWAPD_OPPORTUNISTIC_COMPACTION,
};
+/*
+ * pgdat->kswapd_request holds the order and the hint above in one word. A
+ * waker folds both in with a single cmpxchg and kswapd takes both with a
+ * single exchange, so a run cannot serve one waker's order under another
+ * waker's hint.
+ */
+#define KSWAPD_REQUEST_HINT_MASK 3
+#define KSWAPD_REQUEST_ORDER_SHIFT 2
+
+static inline int kswapd_request(int order,
+ enum kswapd_opportunistic_compaction_type hint)
+{
+ return (order << KSWAPD_REQUEST_ORDER_SHIFT) | hint;
+}
+
+static inline int kswapd_request_order(int request)
+{
+ return request >> KSWAPD_REQUEST_ORDER_SHIFT;
+}
+
+static inline enum kswapd_opportunistic_compaction_type
+kswapd_request_hint(int request)
+{
+ return request & KSWAPD_REQUEST_HINT_MASK;
+}
+
/*
* On NUMA machines, each NUMA node would have a pg_data_t to describe
* it's memory layout. On UMA machines there is a single pglist_data which
@@ -1561,14 +1587,15 @@ typedef struct pglist_data {
struct mutex kswapd_lock;
#endif
struct task_struct *kswapd; /* Protected by kswapd_lock */
- int kswapd_order;
/*
- * Aggregated opportunistic-compaction hint for the next kswapd run.
- * Updated by wakeup_kswapd() based on the gfp flags / order of each
- * waker, and consumed (and reset) by kswapd before balance_pgdat().
- * See enum kswapd_opportunistic_compaction_type for the state machine.
+ * The next kswapd run's request: the highest order asked for since
+ * kswapd last consumed one, and the aggregated opportunistic-
+ * compaction hint, in one word. Updated by wakeup_kswapd() from the
+ * gfp flags and order of each waker, and consumed by kswapd before
+ * balance_pgdat(). See kswapd_request() and
+ * enum kswapd_opportunistic_compaction_type.
*/
- atomic_t kswapd_opportunistic_compaction;
+ atomic_t kswapd_request;
enum zone_type kswapd_highest_zoneidx;
atomic_t kswapd_failures; /* Number of 'reclaimed == 0' runs */
diff --git c/mm/mm_init.c i/mm/mm_init.c
--- c/mm/mm_init.c
+++ i/mm/mm_init.c
@@ -1553,7 +1553,8 @@ void __ref free_area_init_core_hotplug(struct pglist_data *pgdat)
* when it starts in the near future.
*/
pgdat->nr_zones = 0;
- pgdat->kswapd_order = 0;
+ atomic_set(&pgdat->kswapd_request,
+ kswapd_request(0, KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION));
pgdat->kswapd_highest_zoneidx = 0;
pgdat->node_start_pfn = 0;
pgdat->node_present_pages = 0;
diff --git c/mm/vmscan.c i/mm/vmscan.c
--- c/mm/vmscan.c
+++ i/mm/vmscan.c
@@ -7413,6 +7413,49 @@ static enum zone_type kswapd_highest_zoneidx(pg_data_t *pgdat,
return curr_idx == MAX_NR_ZONES ? prev_highest_zoneidx : curr_idx;
}
+/*
+ * How a caller votes on the opportunistic-compaction hint when it folds a
+ * request into pgdat->kswapd_request.
+ */
+enum kswapd_vote {
+ KSWAPD_VOTE_NONE,
+ KSWAPD_VOTE_NO_OPPORTUNISTIC,
+ KSWAPD_VOTE_OPPORTUNISTIC,
+};
+
+/*
+ * Raise the requested order to @order and fold in @vote, in one step.
+ *
+ * The hint is sticky in the "NO" direction: once any waker votes against
+ * opportunistic treatment, the hint stays cleared until kswapd consumes the
+ * request. KSWAPD_VOTE_NONE raises the order and leaves the hint alone.
+ *
+ * Order and hint live in the same word, so a waker whose order kswapd
+ * observes always contributed its vote to the hint kswapd observes with it.
+ */
+static void kswapd_request_fold(pg_data_t *pgdat, int order,
+ enum kswapd_vote vote)
+{
+ int old = atomic_read(&pgdat->kswapd_request);
+ int new;
+
+ do {
+ enum kswapd_opportunistic_compaction_type hint =
+ kswapd_request_hint(old);
+ int req_order = max(kswapd_request_order(old), order);
+
+ if (vote == KSWAPD_VOTE_NO_OPPORTUNISTIC)
+ hint = KSWAPD_NO_OPPORTUNISTIC_COMPACTION;
+ else if (vote == KSWAPD_VOTE_OPPORTUNISTIC &&
+ hint == KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION)
+ hint = KSWAPD_OPPORTUNISTIC_COMPACTION;
+
+ new = kswapd_request(req_order, hint);
+ if (new == old)
+ return;
+ } while (!atomic_try_cmpxchg(&pgdat->kswapd_request, &old, new));
+}
+
static void
kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
unsigned int highest_zoneidx,
@@ -7460,13 +7503,11 @@ kswapd_try_to_sleep(pg_data_t *pgdat, int alloc_order, int reclaim_order,
kswapd_highest_zoneidx(pgdat,
highest_zoneidx));
- if (READ_ONCE(pgdat->kswapd_order) < reclaim_order)
- WRITE_ONCE(pgdat->kswapd_order, reclaim_order);
-
- if (kswapd_opportunistic_compaction ==
- KSWAPD_NO_OPPORTUNISTIC_COMPACTION)
- atomic_set(&pgdat->kswapd_opportunistic_compaction,
- KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
+ kswapd_request_fold(pgdat, reclaim_order,
+ kswapd_opportunistic_compaction ==
+ KSWAPD_NO_OPPORTUNISTIC_COMPACTION ?
+ KSWAPD_VOTE_NO_OPPORTUNISTIC :
+ KSWAPD_VOTE_NONE);
}
finish_wait(&pgdat->kswapd_wait, &wait);
@@ -7524,6 +7565,7 @@ static int kswapd(void *p)
pg_data_t *pgdat = (pg_data_t *)p;
struct task_struct *tsk = current;
enum kswapd_opportunistic_compaction_type kswapd_opportunistic_compaction;
+ int request;
/*
* Tell the memory management that we're a "memory allocator",
@@ -7540,32 +7582,30 @@ static int kswapd(void *p)
tsk->flags |= PF_MEMALLOC | PF_KSWAPD;
set_freezable();
- WRITE_ONCE(pgdat->kswapd_order, 0);
- atomic_set(&pgdat->kswapd_opportunistic_compaction,
- KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
+ atomic_set(&pgdat->kswapd_request,
+ kswapd_request(0, KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION));
WRITE_ONCE(pgdat->kswapd_highest_zoneidx, MAX_NR_ZONES);
atomic_set(&pgdat->nr_writeback_throttled, 0);
for ( ; ; ) {
bool was_frozen;
- alloc_order = reclaim_order = READ_ONCE(pgdat->kswapd_order);
+ request = atomic_read(&pgdat->kswapd_request);
+ alloc_order = reclaim_order = kswapd_request_order(request);
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
highest_zoneidx);
- kswapd_opportunistic_compaction =
- atomic_read(&pgdat->kswapd_opportunistic_compaction);
+ kswapd_opportunistic_compaction = kswapd_request_hint(request);
kswapd_try_sleep:
kswapd_try_to_sleep(pgdat, alloc_order, reclaim_order,
highest_zoneidx, kswapd_opportunistic_compaction);
- /* Read the new order and highest_zoneidx */
- alloc_order = READ_ONCE(pgdat->kswapd_order);
- kswapd_opportunistic_compaction =
- atomic_xchg(&pgdat->kswapd_opportunistic_compaction,
- KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION);
+ /* Take the new request and highest_zoneidx */
+ request = atomic_xchg(&pgdat->kswapd_request,
+ kswapd_request(0, KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION));
+ alloc_order = kswapd_request_order(request);
+ kswapd_opportunistic_compaction = kswapd_request_hint(request);
highest_zoneidx = kswapd_highest_zoneidx(pgdat,
highest_zoneidx);
- WRITE_ONCE(pgdat->kswapd_order, 0);
WRITE_ONCE(pgdat->kswapd_highest_zoneidx, MAX_NR_ZONES);
if (kthread_freezable_should_stop(&was_frozen))
@@ -7625,30 +7665,15 @@ void wakeup_kswapd(struct zone *zone, gfp_t gfp_flags, int order,
if (curr_idx == MAX_NR_ZONES || curr_idx < highest_zoneidx)
WRITE_ONCE(pgdat->kswapd_highest_zoneidx, highest_zoneidx);
- if (READ_ONCE(pgdat->kswapd_order) < order)
- WRITE_ONCE(pgdat->kswapd_order, order);
-
/*
- * Fold this waker into the per-pgdat opportunistic-compaction hint
- * that kswapd will pick up at the start of its next run.
- *
- * The state is sticky in the "NO" direction: once any waker in this
- * batch is order-0 or a non-failable high-order allocation, the hint
- * stays cleared until kswapd consumes it. Only when every waker so
- * far is a failable high-order allocation do we set
- * KSWAPD_OPPORTUNISTIC_COMPACTION, asking shrinkers to skip work
- * that won't realistically help compaction.
+ * Fold this waker's order and its vote on the opportunistic-compaction
+ * hint into the request kswapd will pick up at the start of its next
+ * run. Both change together, so the run that serves this order also
+ * carries this waker's vote.
*/
- if (atomic_read(&pgdat->kswapd_opportunistic_compaction) !=
- KSWAPD_NO_OPPORTUNISTIC_COMPACTION) {
- if (!opportunistic)
- atomic_set(&pgdat->kswapd_opportunistic_compaction,
- KSWAPD_NO_OPPORTUNISTIC_COMPACTION);
- else
- atomic_cmpxchg(&pgdat->kswapd_opportunistic_compaction,
- KSWAPD_UNSET_OPPORTUNISTIC_COMPACTION,
- KSWAPD_OPPORTUNISTIC_COMPACTION);
- }
+ kswapd_request_fold(pgdat, order,
+ opportunistic ? KSWAPD_VOTE_OPPORTUNISTIC :
+ KSWAPD_VOTE_NO_OPPORTUNISTIC);
if (!waitqueue_active(&pgdat->kswapd_wait))
return;
@@ -0,0 +1,28 @@
diff --git c/mm/internal.h i/mm/internal.h
--- c/mm/internal.h
+++ i/mm/internal.h
@@ -1778,12 +1778,24 @@ void __meminit __init_page_from_nid(unsigned long pfn, int nid);
* @order is the order the caller asked for, not the order reclaim was
* promoted to. defrag_mode raises the reclaim order to pageblock_order,
* which says nothing about what the caller can tolerate.
+ *
+ * A transparent huge page allocation sets neither flag, yet a failed
+ * collapse or fault falls back to small pages. Recognise it by its mask.
+ * In this tree the complete GFP_TRANSHUGE_LIGHT set is carried by the
+ * huge page fault path, khugepaged and the large folio migration
+ * targets. The huge zero folio clears __GFP_MOVABLE and so keeps normal
+ * shrinker behaviour. The three other sites that combine __GFP_COMP
+ * with __GFP_NOMEMALLOC do not carry the movable, highmem and
+ * filesystem bits, so they do not match.
*/
static inline bool gfp_opportunistic_compaction(gfp_t gfp_flags, int order)
{
if (!order || (gfp_flags & __GFP_NOFAIL))
return false;
+ if ((gfp_flags & GFP_TRANSHUGE_LIGHT) == GFP_TRANSHUGE_LIGHT)
+ return true;
+
return !!(gfp_flags & (__GFP_NORETRY | __GFP_RETRY_MAYFAIL));
}
@@ -0,0 +1,21 @@
diff --git a/mm/huge_memory.c b/mm/huge_memory.c
--- a/mm/huge_memory.c
+++ b/mm/huge_memory.c
@@ -4476,6 +4476,17 @@ static unsigned long deferred_split_count(struct shrinker *shrink,
{
unsigned long count;
+ /*
+ * Splitting a partially mapped folio during an opportunistic
+ * high-order pass frees its unmapped tail pages but shatters the
+ * folio, so the pass works against the contiguous block it is
+ * reclaiming for. Report nothing rather than SHRINK_EMPTY: for a
+ * memcg aware shrinker SHRINK_EMPTY clears the memcg shrinker bit,
+ * which would also skip the next normal pass.
+ */
+ if (sc->opportunistic_compaction)
+ return 0;
+
count = list_lru_shrink_count(&deferred_split_lru, sc);
return count ?: SHRINK_EMPTY;
}
@@ -0,0 +1,39 @@
diff --git c/drivers/gpu/drm/xe/xe_shrinker.c i/drivers/gpu/drm/xe/xe_shrinker.c
--- c/drivers/gpu/drm/xe/xe_shrinker.c
+++ i/drivers/gpu/drm/xe/xe_shrinker.c
@@ -174,10 +174,17 @@ static unsigned long
xe_shrinker_count(struct shrinker *shrink, struct shrink_control *sc)
{
struct xe_shrinker *shrinker = to_xe_shrinker(shrink);
- unsigned long num_pages;
+ unsigned long num_pages = 0;
bool can_backup = !!(sc->gfp_mask & __GFP_FS);
- num_pages = ttm_backup_bytes_avail() >> PAGE_SHIFT;
+ /*
+ * Skip accounting backup-able pages when this is an opportunistic
+ * high-order pass: TTM backup work shrinks at native page granularity
+ * and is unlikely to produce the contiguous block the caller wants,
+ * so don't advertise it as reclaimable for this hint.
+ */
+ if (!sc->opportunistic_compaction)
+ num_pages = ttm_backup_bytes_avail() >> PAGE_SHIFT;
read_lock(&shrinker->lock);
if (can_backup)
@@ -249,7 +256,14 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
- if (nr_scanned >= nr_to_scan || !can_backup)
+ /*
+ * Stop after the purge pass for opportunistic high-order reclaim:
+ * the subsequent backup/writeback pass works at native page order
+ * and is unlikely to free a contiguous high-order block, so doing
+ * it here would just churn working sets for no compaction benefit.
+ */
+ if (nr_scanned >= nr_to_scan || !can_backup ||
+ sc->opportunistic_compaction)
goto out;
/* If we didn't wake before, try to do it now if needed. */
@@ -0,0 +1,75 @@
diff --git c/drivers/gpu/drm/xe/xe_shrinker.c i/drivers/gpu/drm/xe/xe_shrinker.c
--- c/drivers/gpu/drm/xe/xe_shrinker.c
+++ i/drivers/gpu/drm/xe/xe_shrinker.c
@@ -170,27 +170,28 @@ static int xe_shrinker_walk(struct xe_shrinker *shrinker,
return ret;
}
+static inline bool xe_shrinker_can_backup(const struct shrink_control *sc)
+{
+ return (sc->gfp_mask & __GFP_FS) && !sc->opportunistic_compaction;
+}
+
static unsigned long
xe_shrinker_count(struct shrinker *shrink, struct shrink_control *sc)
{
struct xe_shrinker *shrinker = to_xe_shrinker(shrink);
unsigned long num_pages = 0;
- bool can_backup = !!(sc->gfp_mask & __GFP_FS);
/*
- * Skip accounting backup-able pages when this is an opportunistic
- * high-order pass: TTM backup work shrinks at native page granularity
- * and is unlikely to produce the contiguous block the caller wants,
- * so don't advertise it as reclaimable for this hint.
+ * Only advertise backup-able pages when the backup pass can run. TTM
+ * backup work shrinks at native page granularity and is unlikely to
+ * produce the contiguous block an opportunistic high-order caller
+ * wants, and it needs __GFP_FS.
*/
- if (!sc->opportunistic_compaction)
- num_pages = ttm_backup_bytes_avail() >> PAGE_SHIFT;
read_lock(&shrinker->lock);
-
- if (can_backup)
- num_pages = min_t(unsigned long, num_pages, shrinker->shrinkable_pages);
- else
- num_pages = 0;
+ if (xe_shrinker_can_backup(sc))
+ num_pages = min_t(unsigned long,
+ ttm_backup_bytes_avail() >> PAGE_SHIFT,
+ shrinker->shrinkable_pages);
num_pages += shrinker->purgeable_pages;
read_unlock(&shrinker->lock);
@@ -240,7 +241,7 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
};
bool runtime_pm;
bool purgeable;
- bool can_backup = !!(sc->gfp_mask & __GFP_FS);
+ bool can_backup = xe_shrinker_can_backup(sc);
nr_to_scan = sc->nr_to_scan;
@@ -257,13 +258,15 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
sc->nr_scanned = nr_scanned;
/*
- * Stop after the purge pass for opportunistic high-order reclaim:
- * the subsequent backup/writeback pass works at native page order
- * and is unlikely to free a contiguous high-order block, so doing
- * it here would just churn working sets for no compaction benefit.
+ * Stop after the purge pass when the backup pass may not run. The
+ * backup/writeback pass works at native page order and is unlikely to
+ * free a contiguous high-order block for an opportunistic caller, so
+ * doing it there would churn working sets for no compaction benefit.
+ *
+ * The per-walk runtime PM reference in __xe_shrinker_walk() is not
+ * affected. Only the outer backup-driven reference is.
*/
- if (nr_scanned >= nr_to_scan || !can_backup ||
- sc->opportunistic_compaction)
+ if (nr_scanned >= nr_to_scan || !can_backup)
goto out;
/* If we didn't wake before, try to do it now if needed. */
@@ -0,0 +1,12 @@
diff --git a/tools/bpf/bpftool/Makefile b/tools/bpf/bpftool/Makefile
--- a/tools/bpf/bpftool/Makefile
+++ b/tools/bpf/bpftool/Makefile
@@ -99,7 +99,7 @@ endif
HOST_LDFLAGS := $(LDFLAGS)
# Remove warnings for libbpf bootstrap build
-LIBBPF_BOOTSTRAP_CFLAGS := $(filter-out -W -Wall -Wextra -Wformat -Wformat-signedness,$(HOST_CFLAGS))
+LIBBPF_BOOTSTRAP_CFLAGS := $(filter-out -W -Wall -Wextra -Wformat%,$(HOST_CFLAGS))
INSTALL ?= install
RM ?= rm -f
+732
View File
@@ -0,0 +1,732 @@
# Maintainer: The Omarchy Authors
# Adopted from the Arch Linux linux package.
pkgbase=linux-omarchy-bore
pkgver=7.2.5
pkgrel=3
pkgdesc='Omarchy Linux (BORE CPU scheduler, ADIOS I/O scheduler)'
url='https://omarchy.org'
arch=(
x86_64
)
license=(GPL-2.0-only)
makedepends=(
bc
binutils
cpio
gettext
glibc
libelf
libgcc
openssl
pahole
perl
python
rust
rust-bindgen
rust-src
tar
xxhash
xz
zlib
zstd
)
options=(
!debug
!strip
)
_major=${pkgver%%.*}
if [[ $pkgver == *rc* ]]; then
_base=${pkgver%rc*}
_prev=${_base%.*}.$(( ${_base##*.} - 1 ))
_srcname=linux-${_prev%.0}
_rcpatch=patch-${pkgver//rc/-rc}
else
_srcname=linux-$pkgver
fi
source=(
https://cdn.kernel.org/pub/linux/kernel/v${_major}.x/${_srcname}.tar.{xz,sign}
${_rcpatch:+https://cdn.kernel.org/pub/linux/kernel/v${_major}.x/stable-review/${_rcpatch}.xz}
${_rcpatch:+https://cdn.kernel.org/pub/linux/kernel/v${_major}.x/stable-review/${_rcpatch}.sign}
0010-archlinux-base.patch{,.sig}
0110-bore-6.8.0.patch{,.sig}
0120-tlbpull.patch{,.sig}
0121-smp-preempt.patch{,.sig}
0130-sched-detach-tasks.patch{,.sig}
0131-sched-avg-idle.patch{,.sig}
0140-sched-always-inline.patch{,.sig}
0141-sched-urgent-fixes.patch{,.sig}
0142-sched-itmt-no-debugfs-dependency.patch{,.sig}
0143-sched-hybrid-cluster-balancing.patch{,.sig}
0144-sched-nohz-idle-core.patch{,.sig}
0150-adios-3.2.0.patch{,.sig}
0200-idle.patch{,.sig}
0210-pstate.patch{,.sig}
0211-amd-pstate-fixes.patch{,.sig}
0212-amd-pstate-epp-cache.patch{,.sig}
0250-zsmalloc.patch{,.sig}
0260-mglru-exec-protect.patch{,.sig}
0270-ksm-rmap-walk.patch{,.sig}
0280-mm-updates.patch{,.sig}
0290-zstd-bmi2-fallback-aliases.patch{,.sig}
0291-zstd-bmi2-cpu-feature-dispatch.patch{,.sig}
0292-crypto-zstd-defer-cstream-init.patch{,.sig}
0293-crypto-zstd-defer-dstream-init.patch{,.sig}
0295-af-alg-restrict.patch{,.sig}
0296-x86-mm-pmd-modify-keep-dirty-bit.patch{,.sig}
0300-btrfs.patch{,.sig}
0301-btrfs-fixes.patch{,.sig}
0302-btrfs-zstd-decompress-direct-to-page.patch{,.sig}
0310-fuse-eof-zeroing.patch{,.sig}
0311-fuse-perf.patch{,.sig}
0312-fuse-writethrough-uptodate.patch{,.sig}
0313-fuse-background-wakeup.patch{,.sig}
0350-drm-edid-populate-monitor-range-from-displayid-adaptive-sync.patch{,.sig}
0360-gpu-mem-cgroup.patch{,.sig}
0400-drm-i915-alpm-limit-pr-alpm-to-panel-replay.patch{,.sig}
0401-drm-i915-psr-exit-panel-replay-for-alpm-lag.patch{,.sig}
0402-psr2-early-transport-panels.patch{,.sig}
0411-drm-xe-display-no-stolen-framebuffers.patch{,.sig}
0420-safe-window.patch{,.sig}
0430-fbc.patch{,.sig}
0440-xe3-peak-bandwidth.patch{,.sig}
0450-amd-hdmi-vrr-allm.patch{,.sig}
0451-amd-vtem-tmds-links.patch{,.sig}
0452-amd-hdmi-frl-default.patch{,.sig}
0460-vesa-displayid-dsc-bpp.patch{,.sig}
0461-vesa-dsc-passthru-mode-match-fix.patch{,.sig}
0472-amdgpu-userq-post-reset-error.patch{,.sig}
0473-i915-ptl-cdclk-sanitize.patch{,.sig}
0474-amd-display-oled-vesa-backlight.patch{,.sig}
0475-revert-drm-i915-dp-as-sdp-vrr-or-pr.patch{,.sig}
0510-sound-updates.patch{,.sig}
0511-sound-updates-fixes.patch{,.sig}
0512-sound-fixes.patch{,.sig}
0513-xps13-sof-quirk.patch{,.sig}
0514-rt766-stream-config-type.patch{,.sig}
0516-hda-realtek-rog-strix-g733zw-speakers.patch{,.sig}
0517-asoc-amd-yc-acer-aspire-a314-23p.patch{,.sig}
0540-media-ipu-bridge-ivsc-no-cvs-lookup.patch{,.sig}
0541-cvs-nova-lake-acpi-id.patch{,.sig}
0542-media-cvs-wake-irq-without-claiming-gpio.patch{,.sig}
0560-input.patch{,.sig}
0565-i2c-asue140d-touchpad-100khz.patch{,.sig}
0566-hid-asus-no-keyboard-init-reports-to-touchpads.patch{,.sig}
0600-usb4stream-fixes.patch{,.sig}
0601-usb4stream-busy-poll.patch{,.sig}
0610-typec-cable-altmode-check.patch{,.sig}
0620-usb-string-sanitize.patch{,.sig}
0650-wireguard-tstamp-type.patch{,.sig}
0660-btusb-mediatek-mt7922-13d3-3625.patch{,.sig}
0661-rtw89-command-offload-source.patch{,.sig}
0662-iwlwifi-mld-skip-tx-when-firmware-dead.patch{,.sig}
0700-pci-target-speed-quirk.patch{,.sig}
0750-applesmc-cache-race.patch{,.sig}
0751-applesmc-key-backlight-workqueue-leak.patch{,.sig}
0770-acpi-pm-acer-swift3-sf314-56g-power-resource.patch{,.sig}
0800-platform-updates.patch{,.sig}
0801-amd-pmf-util-unbind-use-after-free.patch{,.sig}
0850-futex-wait-multiple.patch{,.sig}
0851-futex-wait-multiple-fixes.patch{,.sig}
0852-futex-wait-multiple-abi-fixes.patch{,.sig}
8201-xe-shrinker-return-freed-page-count.patch{,.sig}
8202-xe-shrinker-runtime-pm-for-non-system-memory.patch{,.sig}
8203-xe-shrinker-release-through-the-put-helper.patch{,.sig}
8204-mm-opportunistic-compaction.patch{,.sig}
8205-mm-hint-uses-allocation-order.patch{,.sig}
8206-mm-carry-order-and-hint-in-one-word.patch{,.sig}
8207-mm-classify-huge-page-allocations-as-failable.patch{,.sig}
8208-mm-thp-deferred-split-uses-hint.patch{,.sig}
8209-xe-shrinker-use-opportunistic-hint.patch{,.sig}
8210-xe-shrinker-single-backup-decision.patch{,.sig}
9999-bpftool-strip-wformat-bootstrap.patch{,.sig}
)
source_x86_64=(config.x86_64)
validpgpkeys=(
ABAF11C65A2970B130ABE3C479BE3E4300411886 # Linus Torvalds
647F28654894E3BD457199BE38DBBDC86092693E # Greg Kroah-Hartman
12D27D5D8C8E9BF1AABC7C7C7C64768D3DE334E7 # Krzysztof Wilczynski
)
b2sums=('48551bee71cd02815136fb8abe7da4464c2e17c89ef35cb0c0530c8b969fe12127ca97ab6c656ea8e5b29f2c4a8fe4cc143a626f4cca6b972a8105991e4c6905'
'SKIP'
'6d92fb81077232b8cd1ae500b3aabc71434792e750ea6126dc095e50d4278d3e42b3cfdc3c8bc693ea14b109f6adfa3a1ed095187fc5501962fbe8f049f867bc'
'SKIP'
'a32edb39b4ee9c0378239f4998f477fe5371e6931c188d660874b360ca6d3bdcfff71cf088bed071365627d87458eca0be625e154ca3d1e8150e5e0bdf64175c'
'SKIP'
'f6c00ac2400580dffe3605d2693809396e18185f0c59b21b86f1f22a1f8d6529c51e8cc2474bc40f74a96631f415da51521a8b35d5b096e9b9d164c2755fb091'
'SKIP'
'4651980c9d988ed73dc91fd440467de0a1de21efa3efae0fc006524c88af40173182ce6826c212e8b7faaaae1aa512fc755272ec4663c185533e65c936780db3'
'SKIP'
'57f88a64a8ec20794c008c9de70b3f01680ce9023925df8e0f2c7be69f2600e10276d54bd264f5b8b258e04d38cf483cbf83bb79df466f17fe4b6b33fec52a76'
'SKIP'
'050d38240747325e90bb86bb7c47a14573dcb297ee323b3d84f74df9ddf3691cdf32ac9e293bd237c9b09221065b002e297745f4b898a937b9347dacb2de7109'
'SKIP'
'd21fa7cb609188720365a8d95a98c3286f2f32a772309e66913c6b16bcf871ed28ea1ff01fc15ecbb47a02def1a557512126e39ac0daa956a1ba89878acf5de5'
'SKIP'
'cd744c8f861d2cfdfdc98ea6875419a298dcb5d478444b8e551a52fe7ce5279e2d977f8610783d79144be4927a8bf41b2f80113e6f7aa3ea02f54b9984446017'
'SKIP'
'3434a5bbcd2bd8b4e2013400bf4d771ad5c8295ec74f67ff7acaf917954a01d25931679a71ddb8995fd7703b0b9904db4e6fe5f384d3705cf4d7b80dbf5dba55'
'SKIP'
'c8b8bb1ec676378d862f64b23445f170f2e995f3d89f4ffc727c377128ace87417e50d58c39e62d62795924ab4c62f3defad64139e3d3b472b75f6431484e2dd'
'SKIP'
'c9c1be8f72cef610592b1815a81a34706b54f2ef4a188acbd31d98243129e2939fe0c2b679f428a67eac70b069c8a662c3629c598f89430930f5b8bb602aada8'
'SKIP'
'4f58fc77bc13fe9492fe4662e0a4b879a8a1616e271c2095959dccdffed9b0a866a4eb9dc3ac74fad9eed11c362e43a7bfa56d8ea6f571572a63ff4550c5dc23'
'SKIP'
'd1007ee452cdf92967ec7055f1f39a3007a69842040e12f5063a1d2f24fd6d1eeaff2357abf96d0e330541a4781f93fecbcb116a4c73371913582eb72b8f9466'
'SKIP'
'0d3309140e496ad552d1a15d4fbf04b5ad03877bc73ae689f57554e91b7458578f2c88b6bbd37dd9fdf822a8d48c2cd6f961ec735ae5887f96fa51d908d203ee'
'SKIP'
'13beae6f8d23a1f075410f50c09bc94f24f505571af1feb3cee58f732c0d649a9ce5939bc11642c4f3c8dfbd967939b2a0a205220abf06f10d559c4c43a946a3'
'SKIP'
'2f24f88d8f960148046c36222fdaaadfed2ca88c60240fa1d111430898192ca45aeaa93a3afa24ad02b11c4dc8c3904a92f17c0284396f2f43352ad934a35056'
'SKIP'
'9809e258eb684b80d508fa2f2fa270683b62bd1412ed05a970114aaa7d84c37c0beaea51f9ff8c6ee1bc7318c4aee3725e9a1131fefb51ba700801de6ac414c5'
'SKIP'
'f115e8c80d723f7c24e6b1638b801cd664bef20acb1d2f6e6f5612744b916278693011782228cd3f4d0518654b8d53309bc5b06bbe8ed722fab9995f369bf548'
'SKIP'
'31abdd15ae3f9709d6912c90319a0b62d8380e1ae1169575c0e8e419f8ce264a7cf7c72558a85719cd5a1cae7341007adb3e3936268999ebd5215c2ed288f22c'
'SKIP'
'4efb9ab5258aee01b07bc322b160d2f6e3571ff82c4f3e317cab152a919c67522629d56fb79d517da034373cd6924bc3a20c98073197c93590feb1a34e902358'
'SKIP'
'55951c59286df43f069302b0b1036424812f80b45c5ea13219fce1bfc3e831731432c216288d722e44493c736cbf89a3e44801b5a4f1f242c17b5a89b236a5ef'
'SKIP'
'737db39c2f7f46550de4d6d80659bd6c1fc2bf1ca42bd7bd6fc345fc90af995acd056051e12175ac6c957333629a017051d76d5359c871158c93f0a294fb8de5'
'SKIP'
'4d0646c8ad139add6abb62d75308d2d7b5b07afbc74a0f0297ffc2148f0db38abaea16b048704187da49fe3ddee8370b10a7a41d6e62b665650cd79e8c392b4e'
'SKIP'
'bd7bce1bf0ee24ce201fe0e5f0b8a6c829a84d224800547cf97a893326381e2e6eaed9c70529a6864b0c338276b3a307b14e26fba221e62263b90c02eb5e5f34'
'SKIP'
'ad2cfdc3d8de2193110325395e11015577f26ef1081307bb236ed73fbd109822dd44ea026a57c2e17233e1fb7982e78aa986b71c043f42887e6c5d7f51b84ea9'
'SKIP'
'19c3dd32bbe3ed7caec86e90a6af39b07eb99d5a36515cf061bb6979e5a403ca98a0e00e584a9f746330d663e5fc0868ba7fc98ab2df264bc863bdf22ef3dd5e'
'SKIP'
'ed2ee4849b715d80ba103fe090669062fc08c6864c64e8f8291f7f21092f0184ca5528bdd3a9318f077ef00cbe8cf28bacd1e7a9cb9f9ae1047becd930a0c42e'
'SKIP'
'f09e80ed25fdda9b2c79bd5ea3701985596d478b0b4e6948a1ea8fb789ef944e71640ae200442afcc23a9f0e07b34d83365e223b620baf776bbd864977e633a0'
'SKIP'
'a51e5f2a5e37c350fb4d715ece644c565a508970c3aeb1d28cb6cd03a0c7316a1935e028b64f20a03818926c611b50ae0ffd77d04c676d604d303beed0f96e1f'
'SKIP'
'25dcd2c4a4287f26c01419b7d6430629d2827b6efff473361c68d6ec44e94e6d2bb8ab8215dbae5da25781a668f571cfc28b60c5693eb4ed629bf80fe2f68ccf'
'SKIP'
'8d6eeb2330a3ca7c522cef77bf7336fac9444a0884d5d101c933abd3a31a3711529a31f9a181d4795148ecbcc24425750587314ce28e44c9ab2fd2de8365e853'
'SKIP'
'fd487cb02212ffdae5d8beb7d895819377686085b95658dcf597a534b017adcb65dfdf06e0276e7322a46dc28118803b05c246c22e7d3c361f1778ed9e12a3c6'
'SKIP'
'e18e5086c531a37837f6e7df72ce10a38289b02de4cea376772d45cccb5aabc628fe4f3f9607f646903a7cf59b02692703e6b26800604cf958136987df0b4d5d'
'SKIP'
'1b340a502dcf5a1680caf5320b3aa72aa4317e4cf1d877c1f6425e281a5668d5f092417e1f282371c0a211344f5c187fb3bdaf3f97c7e122951edf7e4c08ca58'
'SKIP'
'1124c5c3fd5104c886c841ead1ae9afb42683388b640b1a86523dad9919c6bd59a1f78b07ee5a213e0dca77c636a29676af0592d4ed1735ea35f282f6f23e46f'
'SKIP'
'505bf01a9d6b0926b557b673bb686d1a4ff73ee70539991e52272fbf58aad3ea25cf5ffc5c2d094a2d5e78b18a1a134f2ab530f595073a2d3eba9ff2c1418e1e'
'SKIP'
'b33b46c37f12f7650a67c3731b87ff84add1ac8623cc96b52223a1ef9946bcaa990ed7c7036674e228335f57ec9ed421ea4d6c1dd72d19b516215d9ac2c7c3ee'
'SKIP'
'b4bf4c4c0672bad91d7f8d309e741104918dd8ee8688a00d600379eba5828d1b8157953377c73d0b58050b4ff888c43013ede4e9c959be8d9c028a6143bf58ed'
'SKIP'
'e403bdc300284c4f0f4302e501b8ccfe5e9af04e73a6070eb6615322f5f40ed6edad53af2ed278741f61a4e3811b92faf1af02f113dbcc2a19bde335a55930d7'
'SKIP'
'5de29d0eb5264e24ab9228bea2ea4b549b3fe37ea67c045df9f83064f11c2f557b9641e772d28db59f8d258418c77435b66ff471c3695043f0f8275d5e7b1685'
'SKIP'
'3c2cd960103ad97ce35679f906b3c9be79a9744959c5dfd0ea338d8075b698790934694b2f71fde026b164c5246cba71ffe3cd5dfc1464da338136fbbaea2f1c'
'SKIP'
'2b33b11e48d8dd8d6743f26170fb99eae38541e3665e2448d91cedf729cca7c0278bfba1078ecbe88201eac4407ec5bd0292a97dbab9cfc67e42646712befc1f'
'SKIP'
'71934436987ecf6914e5c559445c2b7ee92a31c467552342094c93a59d32563aa9930e710556cc1e271eeb352705a9df9b3702482f87d045d132233cee4ab48b'
'SKIP'
'9fb1318a3a181644c67b2adcac0cb04d1b88dd0e632751967239d4e8d4698f193795d00194b0f68505964c84572b67168dcbed02ad5bb7abd41e406a62111489'
'SKIP'
'353b7e89a7044b7426db730d5b900eb347907aa0c20e307d2317925d6021a969a343649ede7a32e698dad60dd4472eb197c9578fb0e36fc354a60266fd680cd7'
'SKIP'
'803223224f22aca25dfe34d73e2bd49438f0cfa83b71baa545fa6fe93ee1a631b6b62c7324dc1c0d2d2ded94c36185a83af39b4f4a3362f554a1943ec2e2ec3e'
'SKIP'
'b0a7ab192066c472f3ebc39bc9afc9fef0528ff7c98bf67c5b9cb679cc6d93eab3d25e91d8d74c9c52f72803949a3ade7940e677b9a3114605b966df94da93bb'
'SKIP'
'824089f9f45e2ff9e788d874bb9294b9208b4486fd22c56cb59d4d88cb996ff5a81ac22f8ff136ddffa1e40330f02e53ab759ffea7e4535c11d4f8195069966b'
'SKIP'
'696309582c56fad530a0bf2be12bb744dedac11ed69642b14eade0a26baf1ce52b36074b89b963b40c9c6fbc4857cdc71b2a6c6d64afcbb2df716bbb57579986'
'SKIP'
'14212fca051d3e9a85f75616ff7f95bbd119ef8e5400dafd4447f70db2c4e9c03d971ec4fdfb4db2319d2613062d74f54dfd5d0a100ab0fa27ac6c05e16bd063'
'SKIP'
'7228e249d25a99f5106040b8ddb5705979b3a5af21ca60d0d43a9ecaf6eaccef6bde8ea69c896cffbf2b5963ec4f68987e69b184a8885c7f4c98d2b05fd59fbb'
'SKIP'
'a54ac1c6d43fe88dee08884ed2e5bad0a0b6d11f34a46dc9801e675860afb7641b57384248a144a621e66a73541d73da11f71671116f1cc7f21534526411439a'
'SKIP'
'2e486cabad0d45aeb760d031c7946e9887c710b29dbe9c16e5f4bae6a230031dba4d817f1ee25a0bbca849d1c71b2f84052ffb73c4a6c4350a6ea5ee83c5a40e'
'SKIP'
'23e2a5def526fa886d1f1b05607bfe6b1ecbb2d6009d023145adff9eadc3135f9078faca6638966b586e81e4ab9d48d23b5758504d5a043d5f46e7a701d185fa'
'SKIP'
'5d82f32f7c04b084530722b9848254aac289c2a754a10e16821083a871ec14065dd9ee83e72b8e9e8ba2915f3fbaf17b0874df724002eb08a5f4f84d1b6aecd8'
'SKIP'
'4928cc008dc91fddb89b154b89b1ce2bbff14d8186c0ec1f81971fb9d3b622475e93e42015916699c3442e2268680779f545d183112b57dedae96fc48a75844c'
'SKIP'
'f93494f5f3196cc35d017ff9719d146bf3e5e7bfe32a7f5e75d96468ce30ef367428ba83b1c99b665064d3edaec84cf3bf18211d8f7b4f118b3bdee495eb8ef9'
'SKIP'
'eb79554ef0d014b94bc72d4d559c2cc5419df96486acb3b6a09d5e8e9dc48db626ce0f90338c67fe873b2cc3c45c3f308c8080532b60f74a07c9e39e01d34197'
'SKIP'
'f657143cdd9c937d78315f4617392e60f7c46f5cec882e060f740f77a27dd4c045878cb66e568a6d9ae74e9c706ffb3052817db6f327c87b85ff2f46eea678e2'
'SKIP'
'0acbd87a09a62af860d14f342d61c82c29ec1711038d4abf4ae473df187fbda6aa49386d88928c5dc4c3379e097478b1d40d9ebe9360e64d02e59161164b2a3f'
'SKIP'
'bb06066e90132f0172333d46be54ca816eb7d3230a193a6cae765d8edc50df4a6ba9fb7c0647cd6befc8c3bbbe1d96c285ee202812477d7295c1130ea8cdd6f4'
'SKIP'
'f1a94986b4b3f310fecf981e3619f9787a0c8cb0b4fb3f39e5fe85ee3400e7e50278d41bf5468f65e5bd890358d2aabe28d3e11ee838ccbda2ac7af16be568d9'
'SKIP'
'44dcaad2ab326ebee62d3e63207a161dcb1b63bdb21412b5624ca7875bb6e99997528683a633d7dd526dc5ae9408e73bd066e2a1e5b34ba5c829162df2265346'
'SKIP'
'162a2c2104d5f036657ace6eef2ac2d533c0934376fab2521ff1d3fccd704e3d08fa4f6162e9100a844db2aeb57fa3ce214b8058bd227f38b058ab7606ee7816'
'SKIP'
'd50d5a69087a5ba6c416a6391232764c8281a8094645175fdb2cfc1bb011e33335ae4e96d79364681bea61edcb58ba9acb491e08c301d0667481759d0ee1e183'
'SKIP'
'539b1d0e764f0ace3e572ac49f20a790183155ddc0bcd3a8d040f44c9891dd36b1a0accb8c0288267afbd33e6ceadf6fc3a5e3db40491c695e0c1d3dd64af40c'
'SKIP'
'f5cae6eabf2066d5deef737a15967d5d11340de34b181f3b3d841023507fc58af2b68926eb04bb342e1adb857165400f862f2b35ffec4204a107b3b22a1571e7'
'SKIP'
'ab28a1f4f1756b8451840566d8100cb3d0b498b13d006593c72829b9783e94900bae78ab47c576d156e1d08c56602274b3f7d5bd885de17336623b23c3e4f255'
'SKIP'
'8481f23c33a84d8efb798f4b83f78011f74cd5a3dec428b177d7e600922836c1eec61c4238724038221d8ca2e2cdf20462c5911085da3461aa315f7dd19d96d8'
'SKIP'
'19ebb39f3e53ce77dc3522d15c88270c5ca3fdd76fa8bd417852d08a1d85b30bc6eed3e11c95392b694dfc59bb8b43e41d43df4954f00593c8fbe50bd9e199cc'
'SKIP'
'4dc42e9bfb5ee0f952bfd7befc9d333044aec11fc3a3852d4636f2e38ab139eb92c497eb3e30921b6ee8ae7d6e5834dc334ad25419bac2c875a9fd7eb7945f9b'
'SKIP'
'fea498b5c5ada6d7f57caf652b68f687081aae8552d2d04cd9e96764efd1b896208671f4f501210bb557710cca3ceb0c02ac2b5da6d4c04c210e87ec2d715384'
'SKIP'
'4a4af88fc8a8c37ae6d20b35d19e60a17672569d5489be120edbcc62ab3ba4bbc43b78cd0799b9fd73c84b089364320bcbad6fcb7128c2cb06d15ed984d6fc8c'
'SKIP'
'0959ca78ff31299adf1091f4aa1b32cd2e9066aba5e44f0658c91bcb93fe8714f7b671ffa2cd498ecbb6c6e643ac6571bfb193034af206599122fd7c2a3c23a8'
'SKIP'
'03ca740048e48c5b6948f972f7a826240cf51c9b5c8645d1b1c50366e78cbf45f5b45b7b47f87f231b50aab46ef2f8fd78bc53899ba1c319a19f9f61c7cecb90'
'SKIP'
'1b63569b589e994d1869e2f8eaf5d4ee225fe68862ac5c848e46544b827ca5fddb4bf060f5f39001608fd9bb284eec005f5f9b095307d3a008c9ea8957c2fad1'
'SKIP'
'073514dabc88f0206911eb27eb36157e425849221056438340b22ee7bfa6997b58ce7dc4470b09d7ee96f6374f117d91cc19d9032401b2355d1c0aba6fc75771'
'SKIP'
'd7dffb68fd22db7ef41d1ac193dd297fb6150f034f257a44e6cb0a4848dc4a94e328c2e26d5296a72993e2db75d09e3ac46456d337a5bae767fb266082d3cc66'
'SKIP'
'aca0527f5630671a319e7ecff958a3a64f29df73b427c67c1d29763cc1801fa82e96761c3a61c22d936913943a49261da64d490a0752ba9ffc8e0c1d92f99835'
'SKIP'
'803736a8e836560b3e50b7c0c8cf1f9ebb16b8642cc7bc6e5167d19037bd757431565fe4581776e50eb2693c49f778a24fecff41a553612e7e498f755b8f2f67'
'SKIP'
'4bc132d9c34deaae9b13d08dcff8ea83aa8f9ce4398531e0e0a80df30da3808815f3a587fe5627f6553b8e9c2aea6cc1749b752250a91c94f1cf8ba4f3eb5e8d'
'SKIP'
'2b007e58c0541c6ae71ac07cc8c0762343a50e1f3eb52db5631ce725c4f074f41c639fbf6d8695daa78ee1d6a8cb9a0c31d0a9d9ea2a3a10c05f7f6421dda184'
'SKIP'
'd3b2c0e4fd9f26a0eded375522478f8e9403232799d233eed4668920f8a9942ec6bafbc928f14459e4736f63a9a66d3ae2e9f70dc810145acb5fc522b13ecbd9'
'SKIP'
'c7b0be7a05a304e9993fe8ebccefa44944bf9d27dad9f7b3cc7111221bf50b2db71ca51cafc89dc03b237aef3aba7114543c8475cc393d13e61726e68898d3af'
'SKIP'
'4008c7443bc221a142a939deac86918bef8f09ff1ca99a899586b13120edb8007450daf0b9d143770a9f04d616fe96f527b6aac160be45e5cfa66c0da898039c'
'SKIP'
'029d9e899dd9e7bdbcacee34ee9b468e2e2a8f726e38d014891c8353f6afcc52a58c7d9289fb0487cce4d90c8d51f578c27b1dbcdab42de7308f50b7baf27877'
'SKIP'
'9ede433f4e01a5a313a133a8029d64250596582b93a33e434b8e2ee5bc53ba7f91947e1be2aaf57776028cc89e7ca62cc90067ee64650b3e503acad858777a8a'
'SKIP'
'9ac15b3542629c5333cc7d7f53b4f2995d807a31331b9e04d3da051be0ddde7e8c3cbc79837ffb49a4d364db1680cdc6d763465fa56bd4d26b16b765985babe1'
'SKIP'
'25fc5f03732a584c864ee2ad20a74e16777998b0b34f880bbe9d646cf1944bac1f276bbd62515475337eaa81c0f398d79888aca6dc18bcd06ebb7cee9e4a1d8e'
'SKIP'
'70e35154f9385156a11280294cc254cb314fd89653daedb46546a6bd64c550e3fc7e7a8df3e653c516687f0317bcc710fce82b4d649472333c501ed85fcd6cf6'
'SKIP'
'576fd826f28d945782840865f3e5b7c87d7f6f91f08e473ff0d06a37caa2f89cdd9cb4a7aafef5551363aab9c60e887ee794d5b6292874c12baf88c23ea5668e'
'SKIP'
'b69cf36cc5633e507866e67f59557646dda7d6ea436b145f3f3356c4b12e1300c94977a7d1214835f2476c94a30dd672076241b920d96d744fbc199539df506f'
'SKIP')
b2sums_x86_64=('66da8f81bc90bd7d60cd1a5fc3cb3e721dd0d6d9ec8b82a6f0948cc9652ddbf9bef4d029d21694b346fe18e87e1b14067ea265c354a3adb020861a15131e1232')
# https://www.kernel.org/pub/linux/kernel/v7.x/sha256sums.asc
sha256sums=('55ddf0df8325d9dad96fcff7bd93977d22e3f50af06527572af59b77c7632b78'
'SKIP'
'95f3e9209629044028373af987dc0e270a5a15acb8c8e49c5f05057220c75fe2'
'SKIP'
'daf0aaebff3cf4679d0bb8137ace6c9a2de62f75aa4655a0a02ddf759f3c7f26'
'SKIP'
'a9171e731d08a454a50174889af8936fab962b33c37c67169b0cdf21b0b80821'
'SKIP'
'413e9994d35bc722cfe45752ab60a1bd8454443170d80987e6a5be6b7b745940'
'SKIP'
'01b9087c0cebf17d0ffbbbdd6025c9937fcdfa00c8672d310786ca3c3ce6daf9'
'SKIP'
'5c776365d391a15c23afb3969183449852f795f42355629ac563abf7ae5e4735'
'SKIP'
'3a6aaf3b29eb5b4018c5703626311cd1c7a61bd7100789633efc6a79a9b3247b'
'SKIP'
'b28d9b548407f0da4280f4f9a6c0e92db6f3a32a35c7668471b92096eb7b093c'
'SKIP'
'9492d7db11fd327ff06a52cad945bf8ff6dfcabec72c435bf02b4432ee67475e'
'SKIP'
'66af19c43c068144699183a17b67bcf4df03578937c4cffc5d8b24e2e2828f7b'
'SKIP'
'267506a02ac723ea37adcc7c851051edbd1ed731c6aaea4c9e590aadc207855c'
'SKIP'
'07baad05112a47cb0aea81191502c87e496e13b4f23a6ec36ea65bb4219c35a8'
'SKIP'
'720720067a21e8c57d2619f4c0b46846ac060ec6db97088634d62ab8628c70a9'
'SKIP'
'3d1299d19aa9fbc96d25f67154009813e05e92340340953ac8355649cff837a4'
'SKIP'
'dac26fb42f5df71b30c46b3c5894c6016af5a283c57b45173386767ca53b1bf8'
'SKIP'
'66eb2c49dbf708cc781d411d0e44ed613480f0ef66d522272929d9cb7bb21e93'
'SKIP'
'b7ba949eea77e169aceb2c401d4eb83a7413aab6f15ee2a7ab1a96352f0aa12d'
'SKIP'
'341424b925d506869793686e22ea8f095a34048595ebebec409c0a7bbd346ca2'
'SKIP'
'7c25d0a53a115bc1f072ff0cfdd9aa88858e754ae8b15026a2582656e8bd9c43'
'SKIP'
'79799f12cdf42de1cf6c5d483896b439d77bc2e582e10c7215f5375a76612a72'
'SKIP'
'5e3b455024fab2855b590091c9d14b0d12d5363f3d4bf025b31b9aa88b65eba5'
'SKIP'
'abd8c9326cecc9bb85db4b3dc98ddcb66306b34a6d84fb05d0805aed537e2dc4'
'SKIP'
'60a9df54821cd98581e7476f8a7bb6cb0819762bdbed4d356332e41d164a8182'
'SKIP'
'929108ae0a8eb4782cafdd89e3d4426d02d86f2c2102ed99e760e31f0bc3794a'
'SKIP'
'e1226c836a9c2fb6daff109e56ba21c13fd198e109cbb181de920af04fe153e6'
'SKIP'
'8d60f7dd26ef904419aa966427a9af2f2c9a079bb092ea0b58274ab905f6047a'
'SKIP'
'258597006eb96e1fb185ec9303ceb48148eedce95608b57900383f6df59d2989'
'SKIP'
'f935409f9aeba3314adfad7db1b3d895557a07029a80d10969804e49c0e270ca'
'SKIP'
'd8e64b9da3beab7c33832c019da50d0977b39dc32ec1f649a7c3bc93a2fdb9a5'
'SKIP'
'1780f66b157de6d249301331e31b7505f6e6fa26686fa4c618623567e4b758ae'
'SKIP'
'2501c98aaea6a53eeaa18d696f60b677e710d0f2d89f7525e47ba806de1878b3'
'SKIP'
'e6b770d37e80509ed7ec5d8b59b4d2e5d6dba54ba659d7392de3b567f1eb6e54'
'SKIP'
'b8596b5b5b546fdc309e10326b0e2ccea5a0538d4919630ed2845cded78580aa'
'SKIP'
'5cc215215fed4247c2d6b9732deff586331a07d54c3b00538a81921093a3a1a1'
'SKIP'
'e303da14a3c8a15fd1cb45394138f03dd16e66d2ac0d1ebcd933977df937d771'
'SKIP'
'1f0c958b64ba48b8dc8b5548e1ad1934b8d4903fc16fec15014759e3681ea275'
'SKIP'
'47994d576008a377de612ed77e6b2e7bc6b24ad3a30a6d157646d380ae323142'
'SKIP'
'd9eb3717ba98e270b0a998f8dd90cbda072f63b6a6ea8f66763bde3ea64e591d'
'SKIP'
'1e2fddb2e0c5d184a2e8c774886cf9be2f2f292e4f05832b50e5b49c48a7bacc'
'SKIP'
'b9b8771b6a8bd7128f4344ad189e32c155abf72abd3ff646dcf5c78b08eb94c4'
'SKIP'
'f62837e8ee51070f8139f718e0e80a4ce0c0dea9ce2328c10c63403aca5b3d2e'
'SKIP'
'cae59282e3d1afd69f4f5dc00a180c712ef464c98b8fd7ccc9b03de944dc79b9'
'SKIP'
'5a74c8b2ba11e5876eeeb5ba530ae97e384b48943fa7a31e13396ecb82c5a8ce'
'SKIP'
'8b954ea37a2190022064c82688bc8cb1cf79a90598f078035be7d4041003cda9'
'SKIP'
'40ac31789399a5964c4fd820d4cdc7214add91109bbcf4a614619e5d8b367441'
'SKIP'
'707460d08c15f451d0a0e567c6e35522f3e1409bbe3bfbcf7160f33af06391b7'
'SKIP'
'eebbe50deab379d09b5149ef30769ad6dfa11109696934ee66b108ec406da942'
'SKIP'
'dab92dbc6eac02771c72801d94e6a53aabcf629a9cd79e85e0ce2ba44a65b06b'
'SKIP'
'30e2fc90c950869ab2fe1535b609f35dc1d46d6b6816de4afcbf687dbc224f0a'
'SKIP'
'1fdf8c4d8ee07c02883b827a11a84c1e5f70570545b8a4a232f6b59dd02049fa'
'SKIP'
'0057dff07a3ce9c7b7085edd1f912958e01ce8f7400151d55dfb74c0c6a084e2'
'SKIP'
'2762f72620781a672d9a21bcc591b862b04a756b271cd1009a7390eb1b77635e'
'SKIP'
'541325476efee83232ab8a6c62c14cf3c3c957c49cf053ed2869f7780f86d032'
'SKIP'
'8a8eb0934424ab6312f26e03a06be5d0401360ee60a20a87b11f7c3c6590103c'
'SKIP'
'f7ff13ea46359217d696c6ccd49bc3d74c441ac5893d5151d95138017b0d10aa'
'SKIP'
'd17493b3579c730d27dc723447175b70d991f50253c9fd419df5e923bdf7a244'
'SKIP'
'92971bb4dbde209cf170838cfcd75d8ac9978239821d3493c28cc12623c2bb60'
'SKIP'
'7245c6e2cb5c0b0ddf800894f087770cbf8ccc6f8b3f53b9a21c704e9dfa42c2'
'SKIP'
'135140bce40644079be7743b51f8618fc829a8868ac53137050f76c45eca4061'
'SKIP'
'e6ec816ae309445c6c11e3433c04d0c9100e16c0e23268226d13c19e7be49edb'
'SKIP'
'29c61ec984bd342113b3c4310a3d5fddfd6d7db83847cfb26b84cc730385320a'
'SKIP'
'53425998005a4115bdf6001b9ffbe62c66bb2db205994b2e13005a3417ea6cbb'
'SKIP'
'c28fca9ba015e96599627acf42e52737a6a846bdcbfd3d4af4147eda3109e12d'
'SKIP'
'5ce97dadb9eb1a8b0da41a8446ee74e01a5a7e7ab439bbea7a711318f04ae24f'
'SKIP'
'44f1c7bd18ea7f8772ce41e24dddd5f314e416ec8dfac6afe1ce8680c1bb4903'
'SKIP'
'0bb8ae5b6f23e13c60953eb5bbc8939e8f6c8faf2b216a20203af3b6f0217134'
'SKIP'
'907c92ef681154d10959a5f0cbe2a636688036990ba42e059af00cc2311b4be4'
'SKIP'
'534e68a989d36823f14a10299803f294eef39391619b4c329d78c20539d9fcd5'
'SKIP'
'516e6ec51ca0330989128a4eb2a8f6785e5d45d930899cbbbf37b8147ddfd854'
'SKIP'
'bda3d6842f15998ce3f3e33d0e6e420df9c9c2abcba13716af478720c5933753'
'SKIP'
'3e7e7dba1ca88930c2c3f0d5411ddcee8c2e12fca59d859e2383dccaf9a08022'
'SKIP'
'3a271bc6b0219152047fb2b78d9e695eb791b796ea6d8233096912c77d62a5c0'
'SKIP'
'4e80f2d05591175d4ed0a1a144509e97260e5593472aae586c07956215d99162'
'SKIP'
'8ebce8f38e38bd66879d040028c2448e5bc7dcde9df0ccee2b4221015591b162'
'SKIP'
'0b0e05b615ae6eb825bf7d7fa568f551b0dff6c2cee62c9285c1b0d63224ea1c'
'SKIP'
'2e1e2a0eb039ce61c1612bf6c7ab5227b99df053bcd5592da7758715dd333fda'
'SKIP'
'17de1f76da3db42dc3c7829a22953cd2c6de916999d809e3fb0bbb741753ec8f'
'SKIP'
'd90cc6570a47c8754e4c5d45834e0434c0a1fcabda9b8933b1fba84e3dabc083'
'SKIP'
'028bf538c870465e1752514725372fa9eef6b7178d9b8d1e2ecc15edc5a63cc8'
'SKIP'
'a92a84405ee9845738c6b52d8f0f0eca16eb40afe0ebe0f4e9e1168597675311'
'SKIP'
'44a0b10a6dc83465ea86f5fba936b2bfbeffb1b3f0b2e04013bfed99ba4f6c5a'
'SKIP'
'229b28cee6f2b8afb5888eed453ce7f82de033900383bd91957339007c9e6cbf'
'SKIP'
'c5d61db05b19fde06102b0b3dbd76e1989c8f07f3b7d175176bd2d73f1a71e5c'
'SKIP'
'5988d7a37b4b71a64ed09fea872f08515a3c7029d33fd053a6a1822bb811ddb5'
'SKIP'
'1ec4fb93700b0696ba2c2a371f218f13b32dcc745523c7ee26a5c3cd7af253fc'
'SKIP'
'c67fee35a42866dd3c9056ff099fc504cdad9530c107d959331b5572b21af818'
'SKIP'
'36cb306751c57de3e61c7a4cd18a7f131eef0dc1de15a4e28751d1e5b08bace8'
'SKIP'
'07683e522d6dea3d3ce658ee80bf4317f6c3caf248259da960bc4610f4d1e807'
'SKIP'
'66a0933d2d4f2edc8999b6fdf0fe10b5b607646fd6f1a1d534bf26270543f8f1'
'SKIP'
'c5279468d93e562f509a6b599a2b7ca7ee0032d3fe35314c5769fc6c9a9234ab'
'SKIP'
'c0111614ec44014cbf621eedf5a4ef302a855423bc1633ee692392950e971d0e'
'SKIP'
'1cbe1ddd3b37cd0c7e4bad4af08e681eda2f7a61698004322bb87f320aa95dbb'
'SKIP')
export KBUILD_BUILD_HOST=omarchy
export KBUILD_BUILD_USER=$pkgbase
export KBUILD_BUILD_TIMESTAMP="$(date -Ru${SOURCE_DATE_EPOCH:+d @$SOURCE_DATE_EPOCH})"
prepare() {
cd $_srcname
echo "Setting version..."
echo "-$pkgrel" > localversion.10-pkgrel
echo "${pkgbase#linux}" > localversion.20-pkgname
if [[ -n ${_rcpatch:-} ]]; then
echo "Applying patch $_rcpatch..."
patch -Np1 --fuzz=0 < "../$_rcpatch"
fi
local src
for src in "${source[@]}"; do
src="${src%%::*}"
src="${src##*/}"
src="${src%.zst}"
[[ $src = *.patch ]] || continue
echo "Applying patch $src..."
patch -Np1 --fuzz=0 < "../$src"
done
echo "Setting config..."
cp ../config.$CARCH .config
make olddefconfig
diff -u ../config.$CARCH .config || :
make -s kernelrelease > version
echo "Prepared $pkgbase version $(<version)"
}
build() {
cd $_srcname
MAKEFLAGS="-j$(nproc) $MAKEFLAGS"
make all
make -C tools/bpf/bpftool vmlinux.h feature-clang-bpf-co-re=1
}
_package() {
pkgdesc="The $pkgdesc kernel and modules"
depends=(
coreutils
initramfs
kmod
)
optdepends=(
"$pkgbase-headers: headers and scripts for building modules"
'linux-firmware: firmware images needed for some devices'
'scx-scheds: to use sched-ext schedulers'
'sof-firmware: firmware for Intel SOF audio'
'wireless-regdb: to set the correct wireless channels of your country'
)
provides=(
ADIOS-MODULE
KSMBD-MODULE
NTSYNC-MODULE
VIRTUALBOX-GUEST-MODULES
WIREGUARD-MODULE
)
cd $_srcname
local modulesdir="$pkgdir/usr/lib/modules/$(<version)"
echo "Installing boot image..."
# systemd expects to find the kernel here to allow hibernation
# https://github.com/systemd/systemd/commit/edda44605f06a41fb86b7ab8128dcf99161d2344
install -Dm644 "$(make -s image_name)" "$modulesdir/vmlinuz"
# Used by mkinitcpio to name the kernel
echo "$pkgbase" | install -Dm644 /dev/stdin "$modulesdir/pkgbase"
echo "Installing modules..."
ZSTD_CLEVEL=19 make INSTALL_MOD_PATH="$pkgdir/usr" INSTALL_MOD_STRIP=1 \
DEPMOD=/doesnt/exist modules_install # Suppress depmod
# remove build link
rm "$modulesdir"/build
}
_package-headers() {
pkgdesc="Headers and scripts for building modules for the $pkgdesc kernel"
depends=(
binutils
glibc
libelf
libgcc
openssl
pahole
xxhash
zlib
zstd
)
provides=(LINUX-HEADERS)
cd $_srcname
local builddir="$pkgdir/usr/lib/modules/$(<version)/build"
local karch
case $CARCH in
x86_64) karch=x86 ;;
*) echo "Unknown CARCH $CARCH"; exit 1 ;;
esac
echo "Installing build files..."
install -Dt "$builddir" -m644 .config Makefile Module.symvers System.map \
localversion.* version vmlinux tools/bpf/bpftool/vmlinux.h
install -Dt "$builddir/kernel" -m644 kernel/Makefile
install -Dt "$builddir/arch/$karch" -m644 arch/$karch/Makefile
cp -t "$builddir" -a scripts
ln -srt "$builddir" "$builddir/scripts/gdb/vmlinux-gdb.py"
if [[ $(scripts/config -s CONFIG_HAVE_STACK_VALIDATION) = y ]]; then
install -Dt "$builddir/tools/objtool" tools/objtool/objtool
fi
if [[ $(scripts/config -s CONFIG_DEBUG_INFO_BTF_MODULES) = y ]]; then
install -Dt "$builddir/tools/bpf/resolve_btfids" tools/bpf/resolve_btfids/resolve_btfids
fi
echo "Installing headers..."
cp -t "$builddir" -a include
cp -t "$builddir/arch/$karch" -a arch/$karch/include
install -Dt "$builddir/arch/$karch/kernel" -m644 arch/$karch/kernel/asm-offsets.s
install -Dt "$builddir/drivers/md" -m644 drivers/md/*.h
install -Dt "$builddir/net/mac80211" -m644 net/mac80211/*.h
# https://bugs.archlinux.org/task/13146
install -Dt "$builddir/drivers/media/i2c" -m644 drivers/media/i2c/msp3400-driver.h
# https://bugs.archlinux.org/task/20402
install -Dt "$builddir/drivers/media/usb/dvb-usb" -m644 drivers/media/usb/dvb-usb/*.h
install -Dt "$builddir/drivers/media/dvb-frontends" -m644 drivers/media/dvb-frontends/*.h
install -Dt "$builddir/drivers/media/tuners" -m644 drivers/media/tuners/*.h
# https://bugs.archlinux.org/task/71392
install -Dt "$builddir/drivers/iio/common/hid-sensors" -m644 drivers/iio/common/hid-sensors/*.h
echo "Installing KConfig files..."
find . -name 'Kconfig*' -exec install -Dm644 {} "$builddir/{}" \;
if [[ $(scripts/config -s CONFIG_RUST) = y ]]; then
echo "Installing Rust files..."
install -Dt "$builddir/rust" -m644 rust/*.rmeta
install -Dt "$builddir/rust" rust/*.so
fi
echo "Installing unstripped VDSO..."
make INSTALL_MOD_PATH="$pkgdir/usr" vdso_install \
link= # Suppress build-id symlinks
echo "Removing unneeded architectures..."
local arch
for arch in "$builddir"/arch/*/; do
[[ $arch = */$karch/ ]] && continue
echo "Removing $(basename "$arch")"
rm -r "$arch"
done
echo "Removing documentation..."
rm -r "$builddir/Documentation"
echo "Removing broken symlinks..."
find -L "$builddir" -type l -printf 'Removing %P\n' -delete
echo "Removing loose objects..."
find "$builddir" -type f -name '*.o' -printf 'Removing %P\n' -delete
echo "Stripping build tools..."
local file
while read -rd '' file; do
case "$(file -Sib "$file")" in
application/x-sharedlib\;*) # Libraries (.so)
strip -v $STRIP_SHARED "$file" ;;
application/x-archive\;*) # Libraries (.a)
strip -v $STRIP_STATIC "$file" ;;
application/x-executable\;*) # Binaries
strip -v $STRIP_BINARIES "$file" ;;
application/x-pie-executable\;*) # Relocatable binaries
strip -v $STRIP_SHARED "$file" ;;
esac
done < <(find "$builddir" -type f -perm -u+x ! -name vmlinux -print0)
echo "Stripping vmlinux..."
strip -v $STRIP_STATIC "$builddir/vmlinux"
echo "Adding symlink..."
mkdir -p "$pkgdir/usr/src"
ln -sr "$builddir" "$pkgdir/usr/src/$pkgbase"
}
pkgname=(
"$pkgbase"
"$pkgbase-headers"
)
for _p in "${pkgname[@]}"; do
eval "package_$_p() {
$(declare -f "_package${_p#$pkgbase}")
_package${_p#$pkgbase}
}"
done
# vim:set ts=8 sts=2 sw=2 et:
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,78 @@
-----BEGIN PGP PUBLIC KEY BLOCK-----
mQINBE58tdUBEADY5iQsoL4k8l06dNt+uP2lH8IPi14M51/tOHsW1ZNc8Iok0stH
+uA8w0LpN97UgNhsvXFEkIK2JjLalasUTiUoIeeTshD9t+ekFBx5a9SbLCFlBrDS
TwfieK2xalzomoL22N5ztj1XbdLWh6NRM6kKMeYvgAGo8p884WJk4pPIJK6G0wEw
e9/TG6ilRSLOtxyaF9yZ+FC1eOA1S47Ld2K25Y5GsQF5agwi7nES+9tVVBZp97kB
8IOvELeiSiY0xFXi60yfwIlK6x9dfcxsx5nCyrp2qdqQiPiMD0EJMiuA6wymoi5W
XtmfCpweTB8TvW8Y8uqrwYApzmDleBDTIDP0vCY1o9eftJcWWMkRKC9c7Ziy4nT6
TzmVkNXgqC8/BuOQbpU7I/1VCMoa6e+2a8jrgy5to4dGgu6xQ6jTxWbvgDeB6Hct
WGqf8f9s5lSpH8D8OZLDOXKolqnBd5YrJr0Qmpq4cCcIqwNCMbURtsTpbW/EdWl+
AKwnStXXLI5O6Hg+m4c3O8ZwbzcnAOgTJePm2Xoi71t9SbAZZx1/W7p6/57UGrXR
Q4WfiwpOPD0siF33yO2L7G7Gmm4zh8ieX8aS8guqfWFhuSsDta77F2FB9ozD9WN0
Z5tJowiy3Z1VkxvZjZH8IbcB05yBBBV47BJxrPnSuDT+w45yNTqZ6m4VYwARAQAB
tC9HcmVnIEtyb2FoLUhhcnRtYW4gPGdyZWdraEBsaW51eGZvdW5kYXRpb24ub3Jn
PokCTgQTAQgAOBYhBGR/KGVIlOO9RXGZvjjbvchgkmk+BQJaHvQRAhsDBQsJCAcC
BhUICQoLAgQWAgMBAh4BAheAAAoJEDjbvchgkmk+3/8P+gJ85fYDzXoy47y90FFi
PJqqtkZhf/VPMP5YOJzxCnGVh0CUwC2fGFV6SIU5V78Ede+gArocYq+LpTV4nJz5
SJZZxNBzuEW8t42juF6GZ9uB5SNlqYHUjWbM0bLpl1gut3pe9yJ7mQ2DaZUMYlav
D7sOAiKw/5pCyFLvY9a6ZJmp8QmPUU8Fb9kbbudxfjxgDrAwuVlnGU/I8YIZOHhX
s1hjBNagZCWcxawktDLPylifNOL5UtNuoLJRjsUVatAEjp+g1Xq2A8/t/mfi5K1p
juQaEr5fVzqhkPqt7UQbT1QuZghStYJ5QRunaYT1trvBXmrXKzebBKk85+nlh58g
fRNTyEt2eflNkU1XpFtNcCWo6rke/PZjtHb1CivHD/GhyogeGBfRAMRfmfNDZRZw
e5V+EBNI+RUexscvhVyTp0XhxgXdGy9KpSpWbuwGaQ+q9mVLrYRlNn1k3dnYaWxD
nk0x7xGCE59dd6vpckcD6t/SXujRwT4b0Ypw1jy3Ve3h8OTB5sP5SBpCA33DoQs9
ONbgtL3nX3XST7frXxBkfCD7D58gGCvFvZYAEd1MDGj3250UnBHUPGeVp7/+t/wH
MJ/E3rvb45RGYadd736i0vnJStPIae4M/bVG5qddRjU6mcpir5qYHAIrDz6QwWWF
2BvR7vqYKa36TGX7TORxuyfotCZHcmVnIEtyb2FoLUhhcnRtYW4gPGdyZWdraEBr
ZXJuZWwub3JnPokCTgQTAQgAOBYhBGR/KGVIlOO9RXGZvjjbvchgkmk+BQJaHvNA
AhsDBQsJCAcCBhUICQoLAgQWAgMBAh4BAheAAAoJEDjbvchgkmk+TLEQAJ1Ux/6n
//f2jEVBdWb13qYFBBxKJMNeTU9yPMedQAAhrt68IU1Bt8+/nmZLm1iXWOvPQ019
21i3HBxANnbTqEYYYWnQJJyROiyTuwY7HWlguQXlkxLa1mahVuFee6DHO+O8IGU8
IM+PHdEL08e629sIluu3WGmNXXJ307j47UBu3QFA67YQ7YBmChl7AHBcSpKSplgN
82tbAYtrm5ywYHM5uMFhmbw/DJpzLdFsnzRT9E7PKhH+q1MyPojGT4Oytj3D1QZr
hp8yZ+Zp8TQnleXeBczLfpQPduzurqVomZpWwIZLHCgBJRWmz7/M0kTDIndQle9L
VcJtJqasrRmgL3NsKrYYBw+jHnBe2hp8aq6W3DVaUmkSdshran9ZCaLCpxt62NAg
UkI/eg1sSljo1aeXmF33ymYIpxavW5CGUYKlqYRLUT7en6t/mFiYCwPD22KOdLSf
svVG+pr4UNsfSZdIF+W9/FLW7HJVZGMIldsrGFv4lOtqiXdbRafMtylYw/mU+xhu
9+NslRRrbi1TlWS/BH7ULYu9zKahApf1DFRcrx0PyvtlFleoDZa88uIbmcUO8GzZ
XEhejTv9vNnbmjgvYsRywFcJPkJ/TObfasvvSU9GZn6aU36Y7GYSUGjD1anLiUpr
0FKkruymqBdXHaXGJ44GZ8Hhd5ZMTavwEX7BtE1HcmVnIEtyb2FoLUhhcnRtYW4g
KExpbnV4IGtlcm5lbCBzdGFibGUgcmVsZWFzZSBzaWduaW5nIGtleSkgPGdyZWdA
a3JvYWguY29tPokCOAQTAQIAIgUCTny11QIbAwYLCQgHAwIGFQgCCQoLBBYCAwEC
HgECF4AACgkQONu9yGCSaT5fXBAAx2NfTb1IZ59eV3PKtqNG0qwQdq/62oSqNKlv
lp/JzkeynjeJ7ic1IOs/CTTv2+xoPkLNcNhOPz7uem/4aa/my9A0AEp5UsF6Lvdo
/Hy7Jxc++0EgW//TyvWcU9qd5qS/85VZf8I5pL9TZtHVwfIfLME+G8hkQx0+CWRJ
loLFG48lwi8khp+TsCRYv1tQei7G22xAY5s+53TssaC1MXyQT7aJBGhwnbspY2Ia
RMzsrX0msZn+Fn5WlxxMDxUmUACFMyKGJ+1F6VY01nWolT3G1udOnpee66qXHJo6
XnzkNhzeH8Vf3sMe0sXx8YkN682g1NFaa+el0SDcXZvB91pFkWnQaQSfac5gI4Ki
ShxAqePAH6Og+a/fhs5XdyYw0SN50O+yaSnqEDl7JkByXVKJiVVihDuEe5JZXkoI
O/eTN6uceF89ZQiO/dFn0Kcqc4vL7uuI6FDMRZK7mY7bjFxFW1VjspcxhT1NdR7S
FNrK8Glzd5FS67oTwSNB3CzkJ3ON/kOJ8JSxFEt1ZTc2ZpQujrFyTtbksWm3Yy63
kbpwxRoR6xgaGwtx0SdkkWDCcA+2GZymCjk5FFQkAhoEk0tu/n5fvHS7TTZui9a2
HMsyqmgTJzeU0eQJDgmb/ahzW0VgjHtABaJr40Q83M9upkZdHFXSZb7UHFYkAdH1
OxdvSFW5Ag0ETny11QEQALIiIb/niWy6M6GfBMt/2EBWpLuE+FYVeUQGpGhXD2rU
hOo9UpoxBD/Y5mc5OaJsVL3fySYQldVFOaT7Pu0J1N5FXIBckgtbT3eg+TGD9WIf
Jy6ZpWjBKf6K4frwTwRpLBKqZhcA/78KzxFHeRHjV4cEVZVNoRtVqLYuTlbdlkH6
G2YxgCioxAfqvsGjsg2ES7Xl6xz3uaBH1DFX7S2LXHkDHnloWOTaDRe/4h2VnFHf
76xsJCgt2seJp91kI8bhuR7CUrO5mkRMhnp/z9v6vc2qcMv8EMK62FiBaqENaKg5
6ag8Icujar1YwXG7oYhOuYiWxqGpJUwg5+h/HeYw5Q8ue0UwHPCUZR14pzQCKxag
RMibiufOlS6URbCcBG44ddFAt2vqqopIo069moxfqt6OGig59cYv7PSMfHX25dV0
1Ns+2R1eo7qiktkV+3CSSs/dUArcTxyovuadIAUaZAJ3XqsS3FGzZsPYMYNM9faZ
qOfF6mmGmCZRJMMESWuWjc8ZnVAv4luyD18vlsr/J9rO0t28s4PJyqJGozEXLBLt
saCVihxBHMY7QK/pC0jRniLpeniDDHY875TIiG3nrmtR84nnW9WNOG6tuaIcB6hD
/DmSr72rRoNEpCa/eT7XiCOymGHS5gWR+94R1+J1rQZbd1T8gSq/nQQluJII7oz7
ABEBAAGJAjYEGAEKACACGwwWIQRkfyhlSJTjvUVxmb44273IYJJpPgUCafiYQgAK
CRA4273IYJJpPhxAEACqoI90rwbASUFgZLeu3XnuXvkU991ZnH9Szvlsa1N4KBjK
9Rw+JgJoC26poHstLzfJXHvgbNnycgpyYB04/9P1W86d5Ab37QicRZEa25KIlTLU
CffBOdi/XgjichVUZca5sKgB3zloFSZxock3JtYFYRtrQThOMvjkDS3Dw85hwsDg
41BqdKT1GEDVV6nKWFNqyQAJekNHfYwD5ZK0GVP/59OsLRFofq5eX1HtbeG/FoER
cYseRs7I3FXDkWL7hIrkWrUNNjBRHnBZPs63Z+Tek555cGL31gPHTaobPC44Aiwr
lztjtQzRQf5nHBAUj/++4KNUr/yvtQnT80eUu+B2EpZs5JKHPo2Secs17dfUyB9Q
tdXWlKyVdWbPQfF7jqK3G9eWMroXKB7tX/hWY78mtt9dp9OZ2takkoI79cwVumeJ
n6sNIy1yY26IX2cJibYioy/X1k0Xgrjpq4gNxmxGzur3tmnLkmFa2V5yPpCjLxe4
EK0UYxCjUBM+faFOrQ1A8POYcfUgBJ5CMOVd/n1Jht2uvphLVeVx9MiC8JsW9z0J
0CAC86WSRfcxbvLvgX6cxzezFbDeci3FoAlvrTqIykxPKul3jw+Bk6+1Jx5siXtn
WZLEHgjYDN7F4eJ3yyFANs1zAKIKTlvzuI6A9VSb7KFyn9+GF+QCIgDfGjmr4Q==
=P+GO
-----END PGP PUBLIC KEY BLOCK-----
@@ -0,0 +1,38 @@
-----BEGIN PGP PUBLIC KEY BLOCK-----
mQENBE55CJIBCACkn+aOLmsaq1ejUcXCAOXkO3w7eiLqjR/ziTL2KZ30p7bxP8cT
UXvfM7fwE7EnqCCkji25x2xsoKXB8AlUswIEYUFCOupj2BOsVmJ/rKZW7fCvKTOK
+BguKjebDxNbgmif39bfSnHDWrW832f5HrYmZn7a/VySDQFdul8Gl/R6gs6PHJbg
jjt+K7Px6cQVMVNvY/VBWdvA1zckO/4h6gf3kWWZN+Wlq8wv/pxft8QzNFgweH9o
5bj4tnQ+wMCLCLiDsgEuVawoOAkg3dRMugIUoiKoBKw7b21q9Vjp4jezRvciC6Ys
4kGUSFG1ZjIn3MpY3f3xZ3yuYwrxQ8JcA7KTABEBAAG0JExpbnVzIFRvcnZhbGRz
IDx0b3J2YWxkc0BrZXJuZWwub3JnPokBTgQTAQgAOBYhBKuvEcZaKXCxMKvjxHm+
PkMAQRiGBQJaHxkTAhsDBQsJCAcCBhUICQoLAgQWAgMBAh4BAheAAAoJEHm+PkMA
QRiGzMcH/ieyxrsHR0ng3pi+qy1/sLiTT4WEBN53+1FsGWdP6/DCD3sprFdWDkkB
Dfh9vPCVzPqX7siZMJxw3+wOfjNnGBRiGj7mTE/1XeXJHDwFRyBEVa/bY8ExLKbv
Bf+xpiWOg2Myj5RYaOUBFbOEtfTPob0FtvfZvK3PXkjODTHhDH7QJT2zNPivHG+E
R5VyF1yJEpl10rDTM91NhEeV0n4wpfZkgL8a3JSzo9H2AJX3y35+Dk9wtNge440Z
SVWAnjwxhBLX2R0LUszRhU925c0vP2l20eFncBmAT0NKpn7v9a670WHv45PluG+S
KKktf6b5/BtfqpC3eV58I6FEtSVpM1u0LkxpbnVzIFRvcnZhbGRzIDx0b3J2YWxk
c0BsaW51eC1mb3VuZGF0aW9uLm9yZz6JATgEEwECACIFAk55CJICGwMGCwkIBwMC
BhUIAgkKCwQWAgMBAh4BAheAAAoJEHm+PkMAQRiGbpwH/2jMNyBq6SjFrltEwt6c
wOJak1lkjpP5IfFMemfKPH03jBv98Yb7nnVE/VofRQi0erPvzU9HPitzmq9Hdaz8
pTVD1nNiejn6MBHREY5T10U8J9Holn9S1G3CUvEUaBg+YEhHwWA8hhxFCIRcfz6N
PRkZH5zi9xdXBnjLrE3CpoZwVguwCT/25DuSqqJnviKiH+BOvJi/BnHSnjV1J71M
OpVabaTZKxQ1Qkwiyo7KRa/MrBV4Cw87MjF1jmja91wWNOuAwv1ST+aSaI038zcl
VqbFrc9gHkTeP3o5p8DG3Q7A1pE/yVLRUW+3jucKtiojylWaqxX7FD0RZtIuhNsU
ig+5AQ0ETnkIkgEIAN+ybgD0IlgKRPJ3eksafd+KORseBWwxUy3GH0yAg/4jZCsf
HZ7jpbRKzxNTKW1kE6ClSqehUsuXT5Vc1eh6079erN3y+JNxl6zZPC9v+5GNyc28
qSfNejt4wmwa/y86T7oQfgo77o8Gu/aO/xzOjw7jSDDR3u9p/hFVtsqzptxZzvs3
hVaiLS+0mar9qYZheaCUqOXOKVo38Vg5gkOhMEwKvZs9x3fINU/t8ckxOHq6KiLa
p5Bq87XP0ZJsCaMBwdLYhOFxAiEVtlzwyo3DvMplIahqqNELb71YDhpMq/Hu+42o
R3pqASCPLfO/0GUSdAGXJVhv7L7ng02ETSBmVOUAEQEAAYkBNgQYAQoAIAIbDBYh
BKuvEcZaKXCxMKvjxHm+PkMAQRiGBQJp+i1OAAoJEHm+PkMAQRiG1JMH/0Dnlg9q
tV0CBoaHWFR78YNAo2VpSVjUw6oINeEYvIRR5XoCUzUV6icrxDn1DIhELMxJ7KFp
ie1Ny6D1vzl2NjJagtlPTfi8TfwluxrH5Y/w6pEei/PDw/CawFa7GSkarARipF1H
Yq+DnG8hUm7HA+U6hEyssCPcpYVwjTfLrSB1PJd7OCo6qNzcuEX3b6bJ3et0Eqqt
N4EOg+Krvn+OLSGPFLmu2TBgTV9RKcFRztjHpkY/ipfRdAt+bYXqrAC4w0UZbw22
tOx4cPe2Ru/i6VpfQy9AmL9p7N8V4aaXmcAJtHYVeD0i9wr3scgqRnztKhyikjVm
2CQKI+uoDKHN45k=
=9Sy5
-----END PGP PUBLIC KEY BLOCK-----
@@ -0,0 +1,4 @@
{
"source": "local",
"skip_build": true
}
@@ -0,0 +1,56 @@
diff --git a/kernel/fork.c b/kernel/fork.c
index f0e2e131a9a5..7b611da9a27a 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -127,6 +127,12 @@
#include <kunit/visibility.h>
+#ifdef CONFIG_USER_NS
+static int unprivileged_userns_clone = 1;
+#else
+#define unprivileged_userns_clone 1
+#endif
+
/*
* Minimum number of threads to boot the kernel
*/
@@ -2093,6 +2099,11 @@ __latent_entropy struct task_struct *copy_process(
return ERR_PTR(-EPERM);
}
+ if ((clone_flags & CLONE_NEWUSER) && !unprivileged_userns_clone) {
+ if (!capable(CAP_SYS_ADMIN))
+ return ERR_PTR(-EPERM);
+ }
+
/*
* Force any signals received before this point to be delivered
* before the fork happens. Collect up signals sent to multiple
@@ -3163,6 +3174,10 @@ static int check_unshare_flags(unsigned long unshare_flags)
if (!current_is_single_threaded())
return -EINVAL;
}
+ if ((unshare_flags & CLONE_NEWUSER) && !unprivileged_userns_clone) {
+ if (!capable(CAP_SYS_ADMIN))
+ return -EPERM;
+ }
return 0;
}
@@ -3398,6 +3413,15 @@ static const struct ctl_table fork_sysctl_table[] = {
.mode = 0644,
.proc_handler = sysctl_max_threads,
},
+#ifdef CONFIG_USER_NS
+ {
+ .procname = "unprivileged_userns_clone",
+ .data = &unprivileged_userns_clone,
+ .maxlen = sizeof(int),
+ .mode = 0644,
+ .proc_handler = proc_dointvec,
+ },
+#endif
};
static int __init init_fork_sysctl(void)
Loaded 100 of 195 files, more files were not shown because too many files have changed in this diff. Show more