Add MuQSS kernel based on v7.2.8-5 with latest MuQSS from CK tree

Signed-off-by: Krzysztof Wilczyński <kwilczynski@omarchy.org>
This commit is contained in:
Krzysztof Wilczyński committed 2026-10-08 05:43:59 +09:00
1 parent 2d56129a55
commit 4b6cc8ba0f
304 files changed
+141852

No files matched your search

@@ -0,0 +1,4 @@
{
"source": "local",
"skip_build": true
}
@@ -0,0 +1,55 @@
diff --git a/kernel/fork.c b/kernel/fork.c
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -127,6 +127,12 @@
#include <kunit/visibility.h>
+#ifdef CONFIG_USER_NS
+static int unprivileged_userns_clone = 1;
+#else
+#define unprivileged_userns_clone 1
+#endif
+
/*
* Minimum number of threads to boot the kernel
*/
@@ -2092,6 +2098,11 @@ __latent_entropy struct task_struct *copy_process(
return ERR_PTR(-EPERM);
}
+ if ((clone_flags & CLONE_NEWUSER) && !unprivileged_userns_clone) {
+ if (!capable(CAP_SYS_ADMIN))
+ return ERR_PTR(-EPERM);
+ }
+
/*
* Force any signals received before this point to be delivered
* before the fork happens. Collect up signals sent to multiple
@@ -3165,6 +3176,10 @@ static int check_unshare_flags(unsigned long unshare_flags)
if (!current_is_single_threaded())
return -EINVAL;
}
+ if ((unshare_flags & CLONE_NEWUSER) && !unprivileged_userns_clone) {
+ if (!capable(CAP_SYS_ADMIN))
+ return -EPERM;
+ }
return 0;
}
@@ -3400,6 +3415,15 @@ static const struct ctl_table fork_sysctl_table[] = {
.mode = 0644,
.proc_handler = sysctl_max_threads,
},
+#ifdef CONFIG_USER_NS
+ {
+ .procname = "unprivileged_userns_clone",
+ .data = &unprivileged_userns_clone,
+ .maxlen = sizeof(int),
+ .mode = 0644,
+ .proc_handler = proc_dointvec,
+ },
+#endif
};
static int __init init_fork_sysctl(void)
@@ -0,0 +1,31 @@
diff --git a/Makefile b/Makefile
--- a/Makefile
+++ b/Makefile
@@ -935,6 +935,9 @@ KBUILD_RUSTFLAGS += -Copt-level=2
else ifdef CONFIG_CC_OPTIMIZE_FOR_SIZE
KBUILD_CFLAGS += -Os
KBUILD_RUSTFLAGS += -Copt-level=s
+else ifdef CONFIG_CC_OPTIMIZE_FOR_PERFORMANCE_O3
+KBUILD_CFLAGS += -O3
+KBUILD_RUSTFLAGS += -Copt-level=3
endif
# Always set `debug-assertions` and `overflow-checks` because their default
diff --git a/init/Kconfig b/init/Kconfig
--- a/init/Kconfig
+++ b/init/Kconfig
@@ -1622,6 +1622,14 @@ config CC_OPTIMIZE_FOR_SIZE
Choosing this option will pass "-Os" to your compiler resulting
in a smaller kernel.
+config CC_OPTIMIZE_FOR_PERFORMANCE_O3
+ bool "Optimize harder for performance (-O3)"
+ help
+ Build with the "-O3" compiler flag: more inlining, loop
+ unrolling and vectorization than -O2, at the cost of a larger
+ kernel image and larger modules. Rust code is built at
+ opt-level 3.
+
endchoice
config HAVE_LD_DEAD_CODE_DATA_ELIMINATION
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,582 @@
diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h
--- a/arch/x86/include/asm/pgtable.h
+++ b/arch/x86/include/asm/pgtable.h
@@ -50,7 +50,7 @@ void ptdump_walk_user_pgd_level_checkwx(void);
extern spinlock_t pgd_lock;
extern struct list_head pgd_list;
-extern struct mm_struct *pgd_page_get_mm(struct page *page);
+struct mm_struct *pgd_page_get_mm(struct ptdesc *pt);
extern pmdval_t early_pmd_flags;
diff --git a/arch/x86/include/asm/pgtable_types.h b/arch/x86/include/asm/pgtable_types.h
--- a/arch/x86/include/asm/pgtable_types.h
+++ b/arch/x86/include/asm/pgtable_types.h
@@ -512,7 +512,7 @@ static inline pgprot_t pgprot_large_2_4k(pgprot_t pgprot)
return __pgprot(protval_large_2_4k(pgprot_val(pgprot)));
}
-
+struct ptdesc;
typedef struct page *pgtable_t;
extern pteval_t __supported_pte_mask;
diff --git a/arch/x86/include/asm/tlbflush.h b/arch/x86/include/asm/tlbflush.h
--- a/arch/x86/include/asm/tlbflush.h
+++ b/arch/x86/include/asm/tlbflush.h
@@ -4,6 +4,7 @@
#include <linux/mm_types.h>
#include <linux/mmu_notifier.h>
+#include <linux/minmax.h>
#include <linux/sched.h>
#include <asm/barrier.h>
@@ -211,6 +212,12 @@ extern u16 invlpgb_count_max;
extern void initialize_tlbstate_and_flush(void);
+/*
+ * Keep stack-allocated flush_tlb_info cacheline aligned, but cap the
+ * alignment to avoid excessive stack usage on large-cacheline systems.
+ */
+#define FLUSH_TLB_INFO_ALIGN MIN(SMP_CACHE_BYTES, 64)
+
/*
* TLB flushing:
*
@@ -249,7 +256,7 @@ struct flush_tlb_info {
u8 stride_shift;
u8 freed_tables;
u8 trim_cpumask;
-};
+} __aligned(FLUSH_TLB_INFO_ALIGN);
void flush_tlb_local(void);
void flush_tlb_one_user(unsigned long addr);
diff --git a/arch/x86/kernel/kvm.c b/arch/x86/kernel/kvm.c
--- a/arch/x86/kernel/kvm.c
+++ b/arch/x86/kernel/kvm.c
@@ -663,8 +663,10 @@ static void kvm_flush_tlb_multi(const struct cpumask *cpumask,
u8 state;
int cpu;
struct kvm_steal_time *src;
- struct cpumask *flushmask = this_cpu_cpumask_var_ptr(__pv_cpu_mask);
+ struct cpumask *flushmask;
+ guard(preempt)();
+ flushmask = this_cpu_cpumask_var_ptr(__pv_cpu_mask);
cpumask_copy(flushmask, cpumask);
/*
* We have to call flush only on online vCPUs. And
diff --git a/arch/x86/mm/fault.c b/arch/x86/mm/fault.c
--- a/arch/x86/mm/fault.c
+++ b/arch/x86/mm/fault.c
@@ -275,17 +275,17 @@ void arch_sync_kernel_mappings(unsigned long start, unsigned long end)
for (addr = start & PMD_MASK;
addr >= TASK_SIZE_MAX && addr < VMALLOC_END;
addr += PMD_SIZE) {
- struct page *page;
+ struct ptdesc *ptdesc;
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
spinlock_t *pgt_lock;
/* the pgt_lock only for Xen */
- pgt_lock = &pgd_page_get_mm(page)->page_table_lock;
+ pgt_lock = &pgd_page_get_mm(ptdesc)->page_table_lock;
spin_lock(pgt_lock);
- vmalloc_sync_one(page_address(page), addr);
+ vmalloc_sync_one(ptdesc_address(ptdesc), addr);
spin_unlock(pgt_lock);
}
spin_unlock(&pgd_lock);
diff --git a/arch/x86/mm/init_64.c b/arch/x86/mm/init_64.c
--- a/arch/x86/mm/init_64.c
+++ b/arch/x86/mm/init_64.c
@@ -136,7 +136,7 @@ static void sync_global_pgds_l5(unsigned long start, unsigned long end)
for (addr = start; addr <= end; addr = ALIGN(addr + 1, PGDIR_SIZE)) {
const pgd_t *pgd_ref = pgd_offset_k(addr);
- struct page *page;
+ struct ptdesc *ptdesc;
/* Check for overflow */
if (addr < start)
@@ -146,13 +146,13 @@ static void sync_global_pgds_l5(unsigned long start, unsigned long end)
continue;
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
pgd_t *pgd;
spinlock_t *pgt_lock;
- pgd = (pgd_t *)page_address(page) + pgd_index(addr);
+ pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(addr);
/* the pgt_lock only for Xen */
- pgt_lock = &pgd_page_get_mm(page)->page_table_lock;
+ pgt_lock = &pgd_page_get_mm(ptdesc)->page_table_lock;
spin_lock(pgt_lock);
if (!pgd_none(*pgd_ref) && !pgd_none(*pgd))
@@ -174,7 +174,7 @@ static void sync_global_pgds_l4(unsigned long start, unsigned long end)
for (addr = start; addr <= end; addr = ALIGN(addr + 1, PGDIR_SIZE)) {
pgd_t *pgd_ref = pgd_offset_k(addr);
const p4d_t *p4d_ref;
- struct page *page;
+ struct ptdesc *ptdesc;
/*
* With folded p4d, pgd_none() is always false, we need to
@@ -187,15 +187,15 @@ static void sync_global_pgds_l4(unsigned long start, unsigned long end)
continue;
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
pgd_t *pgd;
p4d_t *p4d;
spinlock_t *pgt_lock;
- pgd = (pgd_t *)page_address(page) + pgd_index(addr);
+ pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(addr);
p4d = p4d_offset(pgd, addr);
/* the pgt_lock only for Xen */
- pgt_lock = &pgd_page_get_mm(page)->page_table_lock;
+ pgt_lock = &pgd_page_get_mm(ptdesc)->page_table_lock;
spin_lock(pgt_lock);
if (!p4d_none(*p4d_ref) && !p4d_none(*p4d))
diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c
--- a/arch/x86/mm/pat/set_memory.c
+++ b/arch/x86/mm/pat/set_memory.c
@@ -88,9 +88,8 @@ static unsigned long direct_pages_count[PG_LEVEL_NUM];
void update_page_count(int level, unsigned long pages)
{
/* Protect against CPA */
- spin_lock(&pgd_lock);
+ guard(spinlock)(&pgd_lock);
direct_pages_count[level] += pages;
- spin_unlock(&pgd_lock);
}
static void split_page_count(int level)
@@ -420,7 +419,7 @@ static void __cpa_collapse_large_pages(struct cpa_data *cpa)
int collapsed = 0;
int i;
- spin_lock(&cpa_lock);
+ guard(spinlock)(&cpa_lock);
if (cpa->flags & (CPA_PAGES_ARRAY | CPA_ARRAY)) {
for (i = 0; i < cpa->numpages; i++)
@@ -435,10 +434,8 @@ static void __cpa_collapse_large_pages(struct cpa_data *cpa)
collapsed += collapse_large_pages(addr, &pgtables);
}
- if (!collapsed) {
- spin_unlock(&cpa_lock);
+ if (!collapsed)
return;
- }
flush_tlb_all();
@@ -454,8 +451,6 @@ static void __cpa_collapse_large_pages(struct cpa_data *cpa)
else
pagetable_free(ptdesc);
}
-
- spin_unlock(&cpa_lock);
}
static void cpa_collapse_large_pages(struct cpa_data *cpa)
@@ -916,24 +911,23 @@ static void __set_pmd_pte(pte_t *kpte, unsigned long address, pte_t pte)
{
/* change init_mm */
set_pte_atomic(kpte, pte);
-#ifdef CONFIG_X86_32
- {
- struct page *page;
- list_for_each_entry(page, &pgd_list, lru) {
+ if (IS_ENABLED(CONFIG_X86_32)) {
+ struct ptdesc *ptdesc;
+
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
pgd_t *pgd;
p4d_t *p4d;
pud_t *pud;
pmd_t *pmd;
- pgd = (pgd_t *)page_address(page) + pgd_index(address);
+ pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(address);
p4d = p4d_offset(pgd, address);
pud = pud_offset(p4d, address);
pmd = pmd_offset(pud, address);
set_pte_atomic((pte_t *)pmd, pte);
}
}
-#endif
}
static pgprot_t pgprot_clear_protnone_bits(pgprot_t prot)
@@ -1103,16 +1097,11 @@ static int __should_split_large_page(pte_t *kpte, unsigned long address,
static int should_split_large_page(pte_t *kpte, unsigned long address,
struct cpa_data *cpa)
{
- int do_split;
-
if (cpa->force_split)
return 1;
- spin_lock(&pgd_lock);
- do_split = __should_split_large_page(kpte, address, cpa);
- spin_unlock(&pgd_lock);
-
- return do_split;
+ guard(spinlock)(&pgd_lock);
+ return __should_split_large_page(kpte, address, cpa);
}
static void split_set_pte(struct cpa_data *cpa, pte_t *pte, unsigned long pfn,
@@ -1162,16 +1151,14 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address,
bool nx, rw;
pte_t *tmp;
- spin_lock(&pgd_lock);
+ guard(spinlock)(&pgd_lock);
/*
* Check for races, another CPU might have split this page
* up for us already:
*/
tmp = _lookup_address_cpa(cpa, address, &level, &nx, &rw);
- if (tmp != kpte) {
- spin_unlock(&pgd_lock);
+ if (tmp != kpte)
return 1;
- }
paravirt_alloc_pte(&init_mm, page_to_pfn(base));
@@ -1204,7 +1191,6 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address,
break;
default:
- spin_unlock(&pgd_lock);
return 1;
}
@@ -1252,7 +1238,6 @@ __split_large_page(struct cpa_data *cpa, pte_t *kpte, unsigned long address,
* just split large page entry.
*/
flush_tlb_all();
- spin_unlock(&pgd_lock);
return 0;
}
@@ -1327,11 +1312,11 @@ static int collapse_pmd_page(pmd_t *pmd, unsigned long addr,
list_add(&page_ptdesc(pmd_page(old_pmd))->pt_list, pgtables);
if (IS_ENABLED(CONFIG_X86_32)) {
- struct page *page;
+ struct ptdesc *ptdesc;
/* Update all PGD tables to use the same large page */
- list_for_each_entry(page, &pgd_list, lru) {
- pgd_t *pgd = (pgd_t *)page_address(page) + pgd_index(addr);
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
+ pgd_t *pgd = (pgd_t *)ptdesc_address(ptdesc) + pgd_index(addr);
p4d_t *p4d = p4d_offset(pgd, addr);
pud_t *pud = pud_offset(p4d, addr);
pmd_t *pmd = pmd_offset(pud, addr);
@@ -1405,7 +1390,7 @@ static int collapse_pud_page(pud_t *pud, unsigned long addr,
*/
static int collapse_large_pages(unsigned long addr, struct list_head *pgtables)
{
- int collapsed = 0;
+ int collapsed;
pgd_t *pgd;
p4d_t *p4d;
pud_t *pud;
@@ -1413,26 +1398,24 @@ static int collapse_large_pages(unsigned long addr, struct list_head *pgtables)
addr &= PMD_MASK;
- spin_lock(&pgd_lock);
+ guard(spinlock)(&pgd_lock);
pgd = pgd_offset_k(addr);
if (pgd_none(*pgd))
- goto out;
+ return 0;
p4d = p4d_offset(pgd, addr);
if (p4d_none(*p4d))
- goto out;
+ return 0;
pud = pud_offset(p4d, addr);
if (!pud_present(*pud) || pud_leaf(*pud))
- goto out;
+ return 0;
pmd = pmd_offset(pud, addr);
if (!pmd_present(*pmd) || pmd_leaf(*pmd))
- goto out;
+ return 0;
collapsed = collapse_pmd_page(pmd, addr, pgtables);
if (collapsed)
collapsed += collapse_pud_page(pud, addr, pgtables);
-out:
- spin_unlock(&pgd_lock);
return collapsed;
}
diff --git a/arch/x86/mm/pgtable.c b/arch/x86/mm/pgtable.c
--- a/arch/x86/mm/pgtable.c
+++ b/arch/x86/mm/pgtable.c
@@ -74,9 +74,9 @@ static void pgd_set_mm(pgd_t *pgd, struct mm_struct *mm)
virt_to_ptdesc(pgd)->pt_mm = mm;
}
-struct mm_struct *pgd_page_get_mm(struct page *page)
+struct mm_struct *pgd_page_get_mm(struct ptdesc *pt)
{
- return page_ptdesc(page)->pt_mm;
+ return pt->pt_mm;
}
static void pgd_ctor(struct mm_struct *mm, pgd_t *pgd)
diff --git a/arch/x86/mm/tlb.c b/arch/x86/mm/tlb.c
--- a/arch/x86/mm/tlb.c
+++ b/arch/x86/mm/tlb.c
@@ -1373,35 +1373,19 @@ void flush_tlb_multi(const struct cpumask *cpumask,
*/
unsigned long tlb_single_page_flush_ceiling __read_mostly = 33;
-static DEFINE_PER_CPU_SHARED_ALIGNED(struct flush_tlb_info, flush_tlb_info);
-
-#ifdef CONFIG_DEBUG_VM
-static DEFINE_PER_CPU(unsigned int, flush_tlb_info_idx);
-#endif
-
-static struct flush_tlb_info *get_flush_tlb_info(struct mm_struct *mm,
- unsigned long start, unsigned long end,
- unsigned int stride_shift, bool freed_tables,
- u64 new_tlb_gen)
+static void init_flush_tlb_info(struct flush_tlb_info *info,
+ struct mm_struct *mm,
+ unsigned long start, unsigned long end,
+ unsigned int stride_shift, bool freed_tables,
+ u64 new_tlb_gen)
{
- struct flush_tlb_info *info = this_cpu_ptr(&flush_tlb_info);
-
-#ifdef CONFIG_DEBUG_VM
- /*
- * Ensure that the following code is non-reentrant and flush_tlb_info
- * is not overwritten. This means no TLB flushing is initiated by
- * interrupt handlers and machine-check exception handlers.
- */
- BUG_ON(this_cpu_inc_return(flush_tlb_info_idx) != 1);
-#endif
-
/*
* If the number of flushes is so large that a full flush
* would be faster, do a full flush.
*/
if ((end - start) >> stride_shift > tlb_single_page_flush_ceiling) {
- start = 0;
- end = TLB_FLUSH_ALL;
+ start = 0;
+ end = TLB_FLUSH_ALL;
}
info->start = start;
@@ -1412,32 +1396,21 @@ static struct flush_tlb_info *get_flush_tlb_info(struct mm_struct *mm,
info->new_tlb_gen = new_tlb_gen;
info->initiating_cpu = smp_processor_id();
info->trim_cpumask = 0;
-
- return info;
-}
-
-static void put_flush_tlb_info(void)
-{
-#ifdef CONFIG_DEBUG_VM
- /* Complete reentrancy prevention checks */
- barrier();
- this_cpu_dec(flush_tlb_info_idx);
-#endif
}
void flush_tlb_mm_range(struct mm_struct *mm, unsigned long start,
unsigned long end, unsigned int stride_shift,
bool freed_tables)
{
- struct flush_tlb_info *info;
+ struct flush_tlb_info info;
+ bool remote_flush = false;
int cpu = get_cpu();
u64 new_tlb_gen;
/* This is also a barrier that synchronizes with switch_mm(). */
new_tlb_gen = inc_mm_tlb_gen(mm);
- info = get_flush_tlb_info(mm, start, end, stride_shift, freed_tables,
- new_tlb_gen);
+ init_flush_tlb_info(&info, mm, start, end, stride_shift, freed_tables, new_tlb_gen);
/*
* flush_tlb_multi() is not optimized for the common case in which only
@@ -1445,20 +1418,24 @@ void flush_tlb_mm_range(struct mm_struct *mm, unsigned long start,
* flush_tlb_func_local() directly in this case.
*/
if (mm_global_asid(mm)) {
- broadcast_tlb_flush(info);
+ broadcast_tlb_flush(&info);
} else if (cpumask_any_but(mm_cpumask(mm), cpu) < nr_cpu_ids) {
- info->trim_cpumask = should_trim_cpumask(mm);
- flush_tlb_multi(mm_cpumask(mm), info);
- consider_global_asid(mm);
+ remote_flush = true;
} else if (mm == this_cpu_read(cpu_tlbstate.loaded_mm)) {
lockdep_assert_irqs_enabled();
local_irq_disable();
- flush_tlb_func(info);
+ flush_tlb_func(&info);
local_irq_enable();
}
- put_flush_tlb_info();
put_cpu();
+
+ if (remote_flush) {
+ info.trim_cpumask = should_trim_cpumask(mm);
+ flush_tlb_multi(mm_cpumask(mm), &info);
+ consider_global_asid(mm);
+ }
+
mmu_notifier_arch_invalidate_secondary_tlbs(mm, start, end);
}
@@ -1527,19 +1504,16 @@ static void kernel_tlb_flush_range(struct flush_tlb_info *info)
void flush_tlb_kernel_range(unsigned long start, unsigned long end)
{
- struct flush_tlb_info *info;
+ struct flush_tlb_info info;
guard(preempt)();
+ init_flush_tlb_info(&info, NULL, start, end, PAGE_SHIFT, false,
+ TLB_GENERATION_INVALID);
- info = get_flush_tlb_info(NULL, start, end, PAGE_SHIFT, false,
- TLB_GENERATION_INVALID);
-
- if (info->end == TLB_FLUSH_ALL)
- kernel_tlb_flush_all(info);
+ if (info.end == TLB_FLUSH_ALL)
+ kernel_tlb_flush_all(&info);
else
- kernel_tlb_flush_range(info);
-
- put_flush_tlb_info();
+ kernel_tlb_flush_range(&info);
}
/*
@@ -1707,12 +1681,12 @@ EXPORT_SYMBOL_FOR_KVM(__flush_tlb_all);
void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch)
{
- struct flush_tlb_info *info;
-
+ struct flush_tlb_info info;
+ bool remote_flush = false;
int cpu = get_cpu();
- info = get_flush_tlb_info(NULL, 0, TLB_FLUSH_ALL, 0, false,
- TLB_GENERATION_INVALID);
+ init_flush_tlb_info(&info, NULL, 0, TLB_FLUSH_ALL, 0, false,
+ TLB_GENERATION_INVALID);
/*
* flush_tlb_multi() is not optimized for the common case in which only
* a local TLB flush is needed. Optimize this use-case by calling
@@ -1722,18 +1696,20 @@ void arch_tlbbatch_flush(struct arch_tlbflush_unmap_batch *batch)
invlpgb_flush_all_nonglobals();
batch->unmapped_pages = false;
} else if (cpumask_any_but(&batch->cpumask, cpu) < nr_cpu_ids) {
- flush_tlb_multi(&batch->cpumask, info);
+ remote_flush = true;
} else if (cpumask_test_cpu(cpu, &batch->cpumask)) {
lockdep_assert_irqs_enabled();
local_irq_disable();
- flush_tlb_func(info);
+ flush_tlb_func(&info);
local_irq_enable();
}
- cpumask_clear(&batch->cpumask);
-
- put_flush_tlb_info();
put_cpu();
+
+ if (remote_flush)
+ flush_tlb_multi(&batch->cpumask, &info);
+
+ cpumask_clear(&batch->cpumask);
}
/*
diff --git a/arch/x86/xen/mmu_pv.c b/arch/x86/xen/mmu_pv.c
--- a/arch/x86/xen/mmu_pv.c
+++ b/arch/x86/xen/mmu_pv.c
@@ -836,15 +836,15 @@ static void xen_pgd_pin(struct mm_struct *mm)
*/
void xen_mm_pin_all(void)
{
- struct page *page;
+ struct ptdesc *ptdesc;
spin_lock(&init_mm.page_table_lock);
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
- if (!PagePinned(page)) {
- __xen_pgd_pin(&init_mm, (pgd_t *)page_address(page));
- SetPageSavePinned(page);
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
+ if (!PagePinned(ptdesc_page(ptdesc))) {
+ __xen_pgd_pin(&init_mm, (pgd_t *)ptdesc_address(ptdesc));
+ SetPageSavePinned(ptdesc_page(ptdesc));
}
}
@@ -947,16 +947,16 @@ static void xen_pgd_unpin(struct mm_struct *mm)
*/
void xen_mm_unpin_all(void)
{
- struct page *page;
+ struct ptdesc *ptdesc;
spin_lock(&init_mm.page_table_lock);
spin_lock(&pgd_lock);
- list_for_each_entry(page, &pgd_list, lru) {
- if (PageSavePinned(page)) {
- BUG_ON(!PagePinned(page));
- __xen_pgd_unpin(&init_mm, (pgd_t *)page_address(page));
- ClearPageSavePinned(page);
+ list_for_each_entry(ptdesc, &pgd_list, pt_list) {
+ if (PageSavePinned(ptdesc_page(ptdesc))) {
+ BUG_ON(!PagePinned(ptdesc_page(ptdesc)));
+ __xen_pgd_unpin(&init_mm, (pgd_t *)ptdesc_address(ptdesc));
+ ClearPageSavePinned(ptdesc_page(ptdesc));
}
}
Binary file not shown.
@@ -0,0 +1,514 @@
diff --git a/include/linux/sched.h b/include/linux/sched.h
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -832,6 +832,17 @@ struct kmap_ctrl {
#endif
};
+#if defined(CONFIG_SMP) && defined(CONFIG_PREEMPTION)
+struct task_ipi_mask {
+ union {
+ cpumask_t *ipi_mask_ptr;
+ unsigned long ipi_mask_val;
+ };
+};
+#else
+struct task_ipi_mask { };
+#endif
+
struct task_struct {
#ifdef CONFIG_THREAD_INFO_IN_TASK
/*
@@ -1442,6 +1453,7 @@ struct task_struct {
struct list_head perf_event_list;
struct perf_ctx_data __rcu *perf_ctx_data;
#endif
+ struct task_ipi_mask __private ipi_mask;
#ifdef CONFIG_DEBUG_PREEMPT
unsigned long preempt_disable_ip;
#endif
diff --git a/include/linux/smp.h b/include/linux/smp.h
--- a/include/linux/smp.h
+++ b/include/linux/smp.h
@@ -47,8 +47,7 @@ extern void __smp_call_single_queue(int cpu, struct llist_node *node);
/* total number of cpus in this system (may exceed NR_CPUS) */
extern unsigned int total_cpus;
-int smp_call_function_single(int cpuid, smp_call_func_t func, void *info,
- int wait);
+int smp_call_function_single(int cpuid, smp_call_func_t func, void *info, bool wait);
void on_each_cpu_cond_mask(smp_cond_func_t cond_func, smp_call_func_t func,
void *info, bool wait, const struct cpumask *mask);
@@ -239,6 +238,18 @@ static inline int get_boot_cpu_id(void)
#endif /* !SMP */
+#if defined(CONFIG_PREEMPTION) && defined(CONFIG_SMP)
+int smp_task_ipi_mask_alloc(struct task_struct *task);
+void smp_task_ipi_mask_free(struct task_struct *task);
+#else
+static inline int smp_task_ipi_mask_alloc(struct task_struct *task)
+{
+ return 0;
+}
+
+static inline void smp_task_ipi_mask_free(struct task_struct *task) { }
+#endif
+
/*
* raw_smp_processor_id() - get the current (unstable) CPU id
*
diff --git a/kernel/fork.c b/kernel/fork.c
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -543,6 +543,7 @@ void free_task(struct task_struct *tsk)
#endif
release_user_cpus_ptr(tsk);
scs_release(tsk);
+ smp_task_ipi_mask_free(tsk);
#ifndef CONFIG_THREAD_INFO_IN_TASK
/*
@@ -941,10 +942,14 @@ static struct task_struct *dup_task_struct(struct task_struct *orig, int node)
#endif
account_kernel_stack(tsk, 1);
- err = scs_prepare(tsk, node);
+ err = smp_task_ipi_mask_alloc(tsk);
if (err)
goto free_stack;
+ err = scs_prepare(tsk, node);
+ if (err)
+ goto free_ipi_mask;
+
#ifdef CONFIG_SECCOMP
/*
* We must handle setting up seccomp filters once we're under
@@ -1022,6 +1027,8 @@ static struct task_struct *dup_task_struct(struct task_struct *orig, int node)
#endif
return tsk;
+free_ipi_mask:
+ smp_task_ipi_mask_free(tsk);
free_stack:
exit_task_stack_account(tsk);
free_thread_stack(tsk);
diff --git a/kernel/scftorture.c b/kernel/scftorture.c
--- a/kernel/scftorture.c
+++ b/kernel/scftorture.c
@@ -348,6 +348,8 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
int ret = 0;
struct scf_check *scfcp = NULL;
struct scf_selector *scfsp = scf_sel_rand(trsp);
+ bool is_single = (scfsp->scfs_prim == SCF_PRIM_SINGLE ||
+ scfsp->scfs_prim == SCF_PRIM_SINGLE_RPC);
if (scfsp->scfs_prim == SCF_PRIM_SINGLE || scfsp->scfs_wait) {
scfcp = kmalloc_obj(*scfcp, GFP_ATOMIC);
@@ -364,8 +366,6 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
}
if (use_cpus_read_lock)
cpus_read_lock();
- else
- preempt_disable();
switch (scfsp->scfs_prim) {
case SCF_PRIM_RESCHED:
if (IS_BUILTIN(CONFIG_SCF_TORTURE_TEST)) {
@@ -411,13 +411,10 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
if (!ret) {
if (use_cpus_read_lock)
cpus_read_unlock();
- else
- preempt_enable();
+
wait_for_completion(&scfcp->scfc_completion);
if (use_cpus_read_lock)
cpus_read_lock();
- else
- preempt_disable();
} else {
scfp->n_single_rpc_ofl++;
scf_add_to_free_list(scfcp);
@@ -452,7 +449,7 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
scfcp->scfc_out = true;
}
if (scfcp && scfsp->scfs_wait) {
- if (WARN_ON_ONCE((num_online_cpus() > 1 || scfsp->scfs_prim == SCF_PRIM_SINGLE) &&
+ if (WARN_ON_ONCE(((use_cpus_read_lock && num_online_cpus() > 1) || is_single) &&
!scfcp->scfc_out)) {
pr_warn("%s: Memory-ordering failure, scfs_prim: %d.\n", __func__, scfsp->scfs_prim);
atomic_inc(&n_mb_out_errs); // Leak rather than trash!
@@ -463,8 +460,6 @@ static void scftorture_invoke_one(struct scf_statistics *scfp, struct torture_ra
}
if (use_cpus_read_lock)
cpus_read_unlock();
- else
- preempt_enable();
if (allocfail)
schedule_timeout_idle((1 + longwait) * HZ); // Let no-wait handlers complete.
else if (!(torture_random(trsp) & 0xfff))
diff --git a/kernel/smp.c b/kernel/smp.c
--- a/kernel/smp.c
+++ b/kernel/smp.c
@@ -16,6 +16,7 @@
#include <linux/init.h>
#include <linux/interrupt.h>
#include <linux/gfp.h>
+#include <linux/slab.h>
#include <linux/smp.h>
#include <linux/cpu.h>
#include <linux/sched.h>
@@ -63,7 +64,14 @@ int smpcfd_prepare_cpu(unsigned int cpu)
free_cpumask_var(cfd->cpumask);
return -ENOMEM;
}
- cfd->csd = alloc_percpu(call_single_data_t);
+
+ /*
+ * Allocate the per-CPU CSD the first time a CPU comes up. It is
+ * not freed when the CPU is offlined, so csd_lock_wait() can access
+ * it even when the CPU was offlined after preemption was re-enabled.
+ */
+ if (!cfd->csd)
+ cfd->csd = alloc_percpu(call_single_data_t);
if (!cfd->csd) {
free_cpumask_var(cfd->cpumask);
free_cpumask_var(cfd->cpumask_ipi);
@@ -79,7 +87,6 @@ int smpcfd_dead_cpu(unsigned int cpu)
free_cpumask_var(cfd->cpumask);
free_cpumask_var(cfd->cpumask_ipi);
- free_percpu(cfd->csd);
return 0;
}
@@ -323,6 +330,8 @@ static void __csd_lock_wait(call_single_data_t *csd)
int bug_id = 0;
u64 ts0, ts1;
+ guard(preempt)();
+
ts1 = ts0 = ktime_get_mono_fast_ns();
for (;;) {
if (csd_lock_wait_toolong(csd, ts0, &ts1, &bug_id, &nmessages))
@@ -659,17 +668,9 @@ void flush_smp_call_function_queue(void)
local_irq_restore(flags);
}
-/**
- * smp_call_function_single - Run a function on a specific CPU
- * @cpu: Specific target CPU for this function.
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait until function has completed on other CPUs.
- *
- * Returns: %0 on success, else a negative status code.
- */
-int smp_call_function_single(int cpu, smp_call_func_t func, void *info,
- int wait)
+static int __smp_call_function_single(int cpu, smp_call_func_t func,
+ void *info, const struct cpumask *mask,
+ bool wait)
{
call_single_data_t *csd;
call_single_data_t csd_stack = {
@@ -686,6 +687,14 @@ int smp_call_function_single(int cpu, smp_call_func_t func, void *info,
*/
this_cpu = get_cpu();
+ if (mask) {
+ /* Try for same CPU (cheapest) */
+ if (!cpumask_test_cpu(this_cpu, mask))
+ cpu = sched_numa_find_nth_cpu(mask, 0, cpu_to_node(this_cpu));
+ else
+ cpu = this_cpu;
+ }
+
/*
* Can deadlock when called with interrupts disabled.
* We allow cpu's that are not yet online though, as no one else can
@@ -718,13 +727,32 @@ int smp_call_function_single(int cpu, smp_call_func_t func, void *info,
err = generic_exec_single(cpu, csd);
+ /*
+ * @csd is stack-allocated when @wait is true. No concurrent access
+ * except from the IPI completion path, so we can re-enable preemption
+ * early to reduce latency.
+ */
+ put_cpu();
+
if (wait)
csd_lock_wait(csd);
- put_cpu();
-
return err;
}
+
+/**
+ * smp_call_function_single - Run a function on a specific CPU
+ * @cpu: Specific target CPU for this function.
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait until function has completed on other CPUs.
+ *
+ * Returns: %0 on success, else a negative status code.
+ */
+int smp_call_function_single(int cpu, smp_call_func_t func, void *info, bool wait)
+{
+ return __smp_call_function_single(cpu, func, info, NULL, wait);
+}
EXPORT_SYMBOL(smp_call_function_single);
/**
@@ -775,10 +803,10 @@ EXPORT_SYMBOL_GPL(smp_call_function_single_async);
/**
* smp_call_function_any - Run a function on any of the given cpus
- * @mask: The mask of cpus it can run on.
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait until function has completed.
+ * @mask: The mask of cpus it can run on.
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait until function has completed.
*
* Selection preference:
* 1) current cpu if in @mask
@@ -789,20 +817,54 @@ EXPORT_SYMBOL_GPL(smp_call_function_single_async);
int smp_call_function_any(const struct cpumask *mask,
smp_call_func_t func, void *info, int wait)
{
- unsigned int cpu;
- int ret;
-
- /* Try for same CPU (cheapest) */
- cpu = get_cpu();
- if (!cpumask_test_cpu(cpu, mask))
- cpu = sched_numa_find_nth_cpu(mask, 0, cpu_to_node(cpu));
-
- ret = smp_call_function_single(cpu, func, info, wait);
- put_cpu();
- return ret;
+ return __smp_call_function_single(-1, func, info, mask, wait);
}
EXPORT_SYMBOL_GPL(smp_call_function_any);
+static DEFINE_STATIC_KEY_FALSE(ipi_mask_inlined);
+
+#ifdef CONFIG_PREEMPTION
+
+int smp_task_ipi_mask_alloc(struct task_struct *task)
+{
+ if (static_branch_unlikely(&ipi_mask_inlined))
+ return 0;
+
+ ACCESS_PRIVATE(task, ipi_mask).ipi_mask_ptr =
+ kmalloc(cpumask_size(), GFP_KERNEL);
+ if (!ACCESS_PRIVATE(task, ipi_mask).ipi_mask_ptr)
+ return -ENOMEM;
+
+ return 0;
+}
+
+void smp_task_ipi_mask_free(struct task_struct *task)
+{
+ if (static_branch_unlikely(&ipi_mask_inlined))
+ return;
+
+ kfree(ACCESS_PRIVATE(task, ipi_mask).ipi_mask_ptr);
+}
+
+static cpumask_t *smp_task_ipi_mask(struct task_struct *cur)
+{
+ /*
+ * If cpumask_size() is smaller than or equal to the pointer
+ * size, it stashes the cpumask in the pointer itself to
+ * avoid extra memory allocations.
+ */
+ if (static_branch_unlikely(&ipi_mask_inlined))
+ return (cpumask_t *)&ACCESS_PRIVATE(cur, ipi_mask).ipi_mask_val;
+
+ return ACCESS_PRIVATE(cur, ipi_mask).ipi_mask_ptr;
+}
+#else
+static cpumask_t *smp_task_ipi_mask(struct task_struct *cur)
+{
+ return NULL;
+}
+#endif
+
/*
* Flags to be used as scf_flags argument of smp_call_function_many_cond().
*
@@ -817,13 +879,20 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
unsigned int scf_flags,
smp_cond_func_t cond_func)
{
- int cpu, last_cpu, this_cpu = smp_processor_id();
- struct call_function_data *cfd;
+ struct cpumask *cpumask, *task_mask;
bool wait = scf_flags & SCF_WAIT;
- int nr_cpus = 0;
+ struct call_function_data *cfd;
+ int cpu, last_cpu, this_cpu;
bool run_remote = false;
+ int nr_cpus = 0;
- lockdep_assert_preemption_disabled();
+ this_cpu = get_cpu();
+ cfd = this_cpu_ptr(&cfd_data);
+ task_mask = smp_task_ipi_mask(current);
+ if (task_mask)
+ cpumask = task_mask;
+ else
+ cpumask = cfd->cpumask;
/*
* Can deadlock when called with interrupts disabled.
@@ -845,16 +914,15 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
/* Check if we need remote execution, i.e., any CPU excluding this one. */
if (cpumask_any_and_but(mask, cpu_online_mask, this_cpu) < nr_cpu_ids) {
- cfd = this_cpu_ptr(&cfd_data);
- cpumask_and(cfd->cpumask, mask, cpu_online_mask);
- __cpumask_clear_cpu(this_cpu, cfd->cpumask);
+ cpumask_and(cpumask, mask, cpu_online_mask);
+ __cpumask_clear_cpu(this_cpu, cpumask);
cpumask_clear(cfd->cpumask_ipi);
- for_each_cpu(cpu, cfd->cpumask) {
+ for_each_cpu(cpu, cpumask) {
call_single_data_t *csd = per_cpu_ptr(cfd->csd, cpu);
if (cond_func && !cond_func(cpu, info)) {
- __cpumask_clear_cpu(cpu, cfd->cpumask);
+ __cpumask_clear_cpu(cpu, cpumask);
continue;
}
@@ -904,8 +972,18 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
local_irq_restore(flags);
}
+ /*
+ * The IPI work has been queued and dispatched. On PREEMPT kernels,
+ * tasks created through dup_task_struct() have task-local wait masks.
+ * The boot init_task can fall back to cfd->cpumask when the mask is
+ * not inlined, but other tasks still use task-local masks and cannot
+ * overwrite it. On !PREEMPT kernels, preempt_enable() cannot schedule
+ * another task, so the per-CPU mask remains protected.
+ */
+ put_cpu();
+
if (run_remote && wait) {
- for_each_cpu(cpu, cfd->cpumask) {
+ for_each_cpu(cpu, cpumask) {
call_single_data_t *csd;
csd = per_cpu_ptr(cfd->csd, cpu);
@@ -916,15 +994,14 @@ static void smp_call_function_many_cond(const struct cpumask *mask,
/**
* smp_call_function_many() - Run a function on a set of CPUs.
- * @mask: The set of cpus to run on (only runs on online subset).
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait (atomically) until function has completed
- * on other CPUs.
+ * @mask: The set of cpus to run on (only runs on online subset).
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait (atomically) until function has completed
+ * on other CPUs.
*
* You must not call this function with disabled interrupts or from a
- * hardware interrupt handler or from a bottom half handler. Preemption
- * must be disabled when calling this function.
+ * hardware interrupt handler or from a bottom half handler.
*
* @func is not called on the local CPU even if @mask contains it. Consider
* using on_each_cpu_cond_mask() instead if this is not desirable.
@@ -938,10 +1015,10 @@ EXPORT_SYMBOL(smp_call_function_many);
/**
* smp_call_function() - Run a function on all other CPUs.
- * @func: The function to run. This must be fast and non-blocking.
- * @info: An arbitrary pointer to pass to the function.
- * @wait: If true, wait (atomically) until function has completed
- * on other CPUs.
+ * @func: The function to run. This must be fast and non-blocking.
+ * @info: An arbitrary pointer to pass to the function.
+ * @wait: If true, wait (atomically) until function has completed
+ * on other CPUs.
*
* If @wait is true, then returns once @func has returned; otherwise
* it returns just before the target cpu calls @func.
@@ -951,9 +1028,8 @@ EXPORT_SYMBOL(smp_call_function_many);
*/
void smp_call_function(smp_call_func_t func, void *info, int wait)
{
- preempt_disable();
- smp_call_function_many(cpu_online_mask, func, info, wait);
- preempt_enable();
+ smp_call_function_many_cond(cpu_online_mask, func, info,
+ wait ? SCF_WAIT : 0, NULL);
}
EXPORT_SYMBOL(smp_call_function);
@@ -1019,6 +1095,9 @@ EXPORT_SYMBOL(nr_cpu_ids);
void __init setup_nr_cpu_ids(void)
{
set_nr_cpu_ids(find_last_bit(cpumask_bits(cpu_possible_mask), NR_CPUS) + 1);
+
+ if (IS_ENABLED(CONFIG_PREEMPTION) && cpumask_size() <= sizeof(unsigned long))
+ static_branch_enable(&ipi_mask_inlined);
}
/* Called by boot processor to activate the rest. */
@@ -1055,12 +1134,14 @@ void __init smp_init(void)
* @func: The function to run on all applicable CPUs.
* This must be fast and non-blocking.
* @info: An arbitrary pointer to pass to both functions.
- * @wait: If true, wait (atomically) until function has
- * completed on other CPUs.
+ * @wait: If true, wait until function has completed on other CPUs.
* @mask: The set of cpus to run on (only runs on online subset).
*
- * Preemption is disabled to protect against CPUs going offline but not online.
- * CPUs going online during the call will not be seen or sent an IPI.
+ * Target CPU selection and work queueing are done with preemption
+ * disabled. This protects against CPUs going offline, but not against
+ * CPUs coming online concurrently; newly online CPUs are not guaranteed
+ * to be seen or sent an IPI. If @wait is true, the final wait for remote
+ * completion happens after that preemption-disabled section.
*
* You must not call this function with disabled interrupts or
* from a hardware interrupt handler or from a bottom half handler.
@@ -1073,9 +1154,7 @@ void on_each_cpu_cond_mask(smp_cond_func_t cond_func, smp_call_func_t func,
if (wait)
scf_flags |= SCF_WAIT;
- preempt_disable();
smp_call_function_many_cond(mask, func, info, scf_flags, cond_func);
- preempt_enable();
}
EXPORT_SYMBOL(on_each_cpu_cond_mask);
diff --git a/kernel/up.c b/kernel/up.c
--- a/kernel/up.c
+++ b/kernel/up.c
@@ -9,8 +9,7 @@
#include <linux/smp.h>
#include <linux/hypervisor.h>
-int smp_call_function_single(int cpu, void (*func) (void *info), void *info,
- int wait)
+int smp_call_function_single(int cpu, void (*func)(void *info), void *info, bool wait)
{
unsigned long flags;
@@ -0,0 +1,21 @@
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -13350,12 +13350,15 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
* still unbalanced. ld_moved simply stays zero, so it is
* correctly treated as an imbalance.
*/
- env.loop_max = min(sysctl_sched_nr_migrate, busiest->nr_running);
-
more_balance:
rq_lock_irqsave(busiest, &rf);
update_rq_clock(busiest);
+ if (!env.loop_max)
+ env.loop_max = min(sysctl_sched_nr_migrate, busiest->cfs.h_nr_queued);
+ else
+ env.loop_max = min(env.loop_max, busiest->cfs.h_nr_queued);
+
/*
* cur_ld_moved - load moved in current iteration
* ld_moved - cumulative load moved across iterations
@@ -0,0 +1,367 @@
diff --git a/arch/arm/include/asm/mmu_context.h b/arch/arm/include/asm/mmu_context.h
--- a/arch/arm/include/asm/mmu_context.h
+++ b/arch/arm/include/asm/mmu_context.h
@@ -80,7 +80,7 @@ static inline void check_and_switch_context(struct mm_struct *mm,
#ifndef MODULE
#define finish_arch_post_lock_switch \
finish_arch_post_lock_switch
-static inline void finish_arch_post_lock_switch(void)
+static __always_inline void finish_arch_post_lock_switch(void)
{
struct mm_struct *mm = current->mm;
diff --git a/arch/riscv/include/asm/sync_core.h b/arch/riscv/include/asm/sync_core.h
--- a/arch/riscv/include/asm/sync_core.h
+++ b/arch/riscv/include/asm/sync_core.h
@@ -6,7 +6,7 @@
* RISC-V implements return to user-space through an xRET instruction,
* which is not core serializing.
*/
-static inline void sync_core_before_usermode(void)
+static __always_inline void sync_core_before_usermode(void)
{
asm volatile ("fence.i" ::: "memory");
}
diff --git a/arch/s390/include/asm/mmu_context.h b/arch/s390/include/asm/mmu_context.h
--- a/arch/s390/include/asm/mmu_context.h
+++ b/arch/s390/include/asm/mmu_context.h
@@ -93,7 +93,7 @@ static inline void switch_mm(struct mm_struct *prev, struct mm_struct *next,
}
#define finish_arch_post_lock_switch finish_arch_post_lock_switch
-static inline void finish_arch_post_lock_switch(void)
+static __always_inline void finish_arch_post_lock_switch(void)
{
struct task_struct *tsk = current;
struct mm_struct *mm = tsk->mm;
diff --git a/arch/sparc/include/asm/mmu_context_64.h b/arch/sparc/include/asm/mmu_context_64.h
--- a/arch/sparc/include/asm/mmu_context_64.h
+++ b/arch/sparc/include/asm/mmu_context_64.h
@@ -160,7 +160,7 @@ static inline void arch_start_context_switch(struct task_struct *prev)
}
#define finish_arch_post_lock_switch finish_arch_post_lock_switch
-static inline void finish_arch_post_lock_switch(void)
+static __always_inline void finish_arch_post_lock_switch(void)
{
/* Restore the state of MCDPER register for the new process
* just switched to.
diff --git a/arch/x86/include/asm/sync_core.h b/arch/x86/include/asm/sync_core.h
--- a/arch/x86/include/asm/sync_core.h
+++ b/arch/x86/include/asm/sync_core.h
@@ -93,7 +93,7 @@ static __always_inline void sync_core(void)
* to user-mode. x86 implements return to user-space through sysexit,
* sysrel, and sysretq, which are not core serializing.
*/
-static inline void sync_core_before_usermode(void)
+static __always_inline void sync_core_before_usermode(void)
{
/* With PTI, we unconditionally serialize before running user code. */
if (static_cpu_has(X86_FEATURE_PTI))
diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h
--- a/include/linux/perf_event.h
+++ b/include/linux/perf_event.h
@@ -1632,7 +1632,7 @@ static inline void perf_event_task_migrate(struct task_struct *task)
task->sched_migrated = 1;
}
-static inline void perf_event_task_sched_in(struct task_struct *prev,
+static __always_inline void perf_event_task_sched_in(struct task_struct *prev,
struct task_struct *task)
{
if (static_branch_unlikely(&perf_sched_events))
diff --git a/include/linux/sched/mm.h b/include/linux/sched/mm.h
--- a/include/linux/sched/mm.h
+++ b/include/linux/sched/mm.h
@@ -32,7 +32,7 @@ extern struct mm_struct *mm_alloc(void);
* See also <Documentation/mm/active_mm.rst> for an in-depth explanation
* of &mm_struct.mm_count vs &mm_struct.mm_users.
*/
-static inline void mmgrab(struct mm_struct *mm)
+static __always_inline void mmgrab(struct mm_struct *mm)
{
atomic_inc(&mm->mm_count);
}
@@ -44,7 +44,7 @@ static inline void smp_mb__after_mmgrab(void)
extern void __mmdrop(struct mm_struct *mm);
-static inline void mmdrop(struct mm_struct *mm)
+static __always_inline void mmdrop(struct mm_struct *mm)
{
/*
* The implicit full barrier implied by atomic_dec_and_test() is
@@ -71,27 +71,27 @@ static inline void __mmdrop_delayed(struct rcu_head *rhp)
* Invoked from finish_task_switch(). Delegates the heavy lifting on RT
* kernels via RCU.
*/
-static inline void mmdrop_sched(struct mm_struct *mm)
+static __always_inline void mmdrop_sched(struct mm_struct *mm)
{
/* Provides a full memory barrier. See mmdrop() */
if (atomic_dec_and_test(&mm->mm_count))
call_rcu(&mm->delayed_drop, __mmdrop_delayed);
}
#else
-static inline void mmdrop_sched(struct mm_struct *mm)
+static __always_inline void mmdrop_sched(struct mm_struct *mm)
{
mmdrop(mm);
}
#endif
/* Helpers for lazy TLB mm refcounting */
-static inline void mmgrab_lazy_tlb(struct mm_struct *mm)
+static __always_inline void mmgrab_lazy_tlb(struct mm_struct *mm)
{
if (IS_ENABLED(CONFIG_MMU_LAZY_TLB_REFCOUNT))
mmgrab(mm);
}
-static inline void mmdrop_lazy_tlb(struct mm_struct *mm)
+static __always_inline void mmdrop_lazy_tlb(struct mm_struct *mm)
{
if (IS_ENABLED(CONFIG_MMU_LAZY_TLB_REFCOUNT)) {
mmdrop(mm);
@@ -104,7 +104,7 @@ static inline void mmdrop_lazy_tlb(struct mm_struct *mm)
}
}
-static inline void mmdrop_lazy_tlb_sched(struct mm_struct *mm)
+static __always_inline void mmdrop_lazy_tlb_sched(struct mm_struct *mm)
{
if (IS_ENABLED(CONFIG_MMU_LAZY_TLB_REFCOUNT))
mmdrop_sched(mm);
@@ -128,12 +128,12 @@ static inline void mmdrop_lazy_tlb_sched(struct mm_struct *mm)
* See also <Documentation/mm/active_mm.rst> for an in-depth explanation
* of &mm_struct.mm_count vs &mm_struct.mm_users.
*/
-static inline void mmget(struct mm_struct *mm)
+static __always_inline void mmget(struct mm_struct *mm)
{
atomic_inc(&mm->mm_users);
}
-static inline bool mmget_not_zero(struct mm_struct *mm)
+static __always_inline bool mmget_not_zero(struct mm_struct *mm)
{
return atomic_inc_not_zero(&mm->mm_users);
}
@@ -532,7 +532,7 @@ enum {
#include <asm/membarrier.h>
#endif
-static inline void membarrier_mm_sync_core_before_usermode(struct mm_struct *mm)
+static __always_inline void membarrier_mm_sync_core_before_usermode(struct mm_struct *mm)
{
/*
* The atomic_read() below prevents CSE. The following should
diff --git a/include/linux/tick.h b/include/linux/tick.h
--- a/include/linux/tick.h
+++ b/include/linux/tick.h
@@ -175,7 +175,7 @@ extern cpumask_var_t tick_nohz_full_mask;
#ifdef CONFIG_NO_HZ_FULL
extern bool tick_nohz_full_running;
-static inline bool tick_nohz_full_enabled(void)
+static __always_inline bool tick_nohz_full_enabled(void)
{
if (!context_tracking_enabled())
return false;
@@ -299,7 +299,7 @@ static inline void __tick_nohz_task_switch(void) { }
static inline void tick_nohz_full_setup(cpumask_var_t cpumask) { }
#endif
-static inline void tick_nohz_task_switch(void)
+static __always_inline void tick_nohz_task_switch(void)
{
if (tick_nohz_full_enabled())
__tick_nohz_task_switch();
diff --git a/include/linux/vtime.h b/include/linux/vtime.h
--- a/include/linux/vtime.h
+++ b/include/linux/vtime.h
@@ -89,24 +89,24 @@ static __always_inline void vtime_account_guest_exit(void)
* For now vtime state is tied to context tracking. We might want to decouple
* those later if necessary.
*/
-static inline bool vtime_accounting_enabled(void)
+static __always_inline bool vtime_accounting_enabled(void)
{
return context_tracking_enabled();
}
-static inline bool vtime_accounting_enabled_cpu(int cpu)
+static __always_inline bool vtime_accounting_enabled_cpu(int cpu)
{
return vtime_generic_enabled_cpu(cpu);
}
-static inline bool vtime_accounting_enabled_this_cpu(void)
+static __always_inline bool vtime_accounting_enabled_this_cpu(void)
{
return vtime_generic_enabled_this_cpu();
}
extern void vtime_task_switch_generic(struct task_struct *prev);
-static inline void vtime_task_switch(struct task_struct *prev)
+static __always_inline void vtime_task_switch(struct task_struct *prev)
{
if (vtime_accounting_enabled_this_cpu())
vtime_task_switch_generic(prev);
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -5098,7 +5098,7 @@ static inline void prepare_task(struct task_struct *next)
WRITE_ONCE(next->on_cpu, 1);
}
-static inline void finish_task(struct task_struct *prev)
+static __always_inline void finish_task(struct task_struct *prev)
{
/*
* This must be the very last reference to @prev from this CPU. After
@@ -5142,7 +5142,7 @@ static void zap_balance_callbacks(struct rq *rq)
rq->balance_callback = found ? &balance_push_callback : NULL;
}
-static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
+static __always_inline void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
{
void (*func)(struct rq *rq);
struct balance_callback *next;
@@ -5177,7 +5177,7 @@ struct balance_callback balance_push_callback = {
.func = balance_push,
};
-static inline struct balance_callback *
+static __always_inline struct balance_callback *
__splice_balance_callbacks(struct rq *rq, bool split)
{
struct balance_callback *head = rq->balance_callback;
@@ -5251,7 +5251,7 @@ prepare_lock_switch(struct rq *rq, struct task_struct *next, struct rq_flags *rf
__acquire(__rq_lockp(this_rq()));
}
-static inline void finish_lock_switch(struct rq *rq)
+static __always_inline void finish_lock_switch(struct rq *rq)
__releases(__rq_lockp(rq))
{
/*
@@ -5285,7 +5285,7 @@ static inline void kmap_local_sched_out(void)
#endif
}
-static inline void kmap_local_sched_in(void)
+static __always_inline void kmap_local_sched_in(void)
{
#ifdef CONFIG_KMAP_LOCAL
if (unlikely(current->kmap_ctrl.idx))
@@ -5339,7 +5339,7 @@ prepare_task_switch(struct rq *rq, struct task_struct *prev,
* past. 'prev == current' is still correct but we need to recalculate this_rq
* because prev may have moved to another CPU.
*/
-static struct rq *finish_task_switch(struct task_struct *prev)
+static __always_inline struct rq *finish_task_switch(struct task_struct *prev)
__releases(__rq_lockp(this_rq()))
{
struct rq *rq = this_rq();
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1464,12 +1464,12 @@ static inline struct cpumask *sched_group_span(struct sched_group *sg);
DECLARE_STATIC_KEY_FALSE(__sched_core_enabled);
-static inline bool sched_core_enabled(struct rq *rq)
+static __always_inline bool sched_core_enabled(struct rq *rq)
{
return static_branch_unlikely(&__sched_core_enabled) && rq->core_enabled;
}
-static inline bool sched_core_disabled(void)
+static __always_inline bool sched_core_disabled(void)
{
return !static_branch_unlikely(&__sched_core_enabled);
}
@@ -1478,7 +1478,7 @@ static inline bool sched_core_disabled(void)
* Be careful with this function; not for general use. The return value isn't
* stable unless you actually hold a relevant rq->__lock.
*/
-static inline raw_spinlock_t *rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *rq_lockp(struct rq *rq)
{
if (sched_core_enabled(rq))
return &rq->core->__lock;
@@ -1486,7 +1486,7 @@ static inline raw_spinlock_t *rq_lockp(struct rq *rq)
return &rq->__lock;
}
-static inline raw_spinlock_t *__rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *__rq_lockp(struct rq *rq)
__returns_ctx_lock(rq_lockp(rq)) /* alias them */
{
if (rq->core_enabled)
@@ -1589,12 +1589,12 @@ static inline bool sched_core_disabled(void)
return true;
}
-static inline raw_spinlock_t *rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *rq_lockp(struct rq *rq)
{
return &rq->__lock;
}
-static inline raw_spinlock_t *__rq_lockp(struct rq *rq)
+static __always_inline raw_spinlock_t *__rq_lockp(struct rq *rq)
__returns_ctx_lock(rq_lockp(rq)) /* alias them */
{
return &rq->__lock;
@@ -1654,33 +1654,33 @@ extern void raw_spin_rq_lock_nested(struct rq *rq, int subclass)
extern bool raw_spin_rq_trylock(struct rq *rq)
__cond_acquires(true, __rq_lockp(rq));
-static inline void raw_spin_rq_lock(struct rq *rq)
+static __always_inline void raw_spin_rq_lock(struct rq *rq)
__acquires(__rq_lockp(rq))
{
raw_spin_rq_lock_nested(rq, 0);
}
-static inline void raw_spin_rq_unlock(struct rq *rq)
+static __always_inline void raw_spin_rq_unlock(struct rq *rq)
__releases(__rq_lockp(rq))
{
raw_spin_unlock(rq_lockp(rq));
}
-static inline void raw_spin_rq_lock_irq(struct rq *rq)
+static __always_inline void raw_spin_rq_lock_irq(struct rq *rq)
__acquires(__rq_lockp(rq))
{
local_irq_disable();
raw_spin_rq_lock(rq);
}
-static inline void raw_spin_rq_unlock_irq(struct rq *rq)
+static __always_inline void raw_spin_rq_unlock_irq(struct rq *rq)
__releases(__rq_lockp(rq))
{
raw_spin_rq_unlock(rq);
local_irq_enable();
}
-static inline unsigned long _raw_spin_rq_lock_irqsave(struct rq *rq)
+static __always_inline unsigned long _raw_spin_rq_lock_irqsave(struct rq *rq)
__acquires(__rq_lockp(rq))
{
unsigned long flags;
@@ -1691,7 +1691,7 @@ static inline unsigned long _raw_spin_rq_lock_irqsave(struct rq *rq)
return flags;
}
-static inline void raw_spin_rq_unlock_irqrestore(struct rq *rq, unsigned long flags)
+static __always_inline void raw_spin_rq_unlock_irqrestore(struct rq *rq, unsigned long flags)
__releases(__rq_lockp(rq))
{
raw_spin_rq_unlock(rq);
@@ -0,0 +1,97 @@
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -10575,17 +10575,40 @@ static enum llc_mig can_migrate_llc(int src_cpu, int dst_cpu,
return mig_llc;
}
+static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct *p)
+{
+ /*
+ * On asymmetric CPU capacity domains, do not let cache-aware
+ * balancing pull the task onto a destination CPU that cannot
+ * accommodate it. Doing so would turn the task into a misfit on
+ * the destination, trading a cache-locality gain for a capacity
+ * loss. If the task already does not fit its source CPU, the move
+ * cannot make things worse, so let the LLC preference decide.
+ */
+ if ((env->sd->flags & SD_ASYM_CPUCAPACITY) && p &&
+ !task_fits_cpu(p, env->dst_cpu) &&
+ task_fits_cpu(p, env->src_cpu))
+ return true;
+
+ return false;
+}
+
/*
* Check if task p can migrate from source LLC to
* destination LLC in terms of cache aware load balance.
*/
-static enum llc_mig can_migrate_llc_task(int src_cpu, int dst_cpu,
+static enum llc_mig can_migrate_llc_task(struct lb_env *env,
struct task_struct *p)
{
struct mm_struct *mm;
bool to_pref;
- int cpu;
+ int cpu, src_cpu, dst_cpu;
+ if (task_misfits_asym_cpu(env, p))
+ return mig_forbid;
+
+ src_cpu = env->src_cpu;
+ dst_cpu = env->dst_cpu;
mm = p->mm;
if (!mm)
return mig_unrestricted;
@@ -10642,6 +10665,14 @@ alb_break_llc(struct lb_env *env)
unsigned long util = 0;
struct task_struct *cur;
+ /*
+ * Migrating misfit tasks from current CPU
+ * to CPU with a better fit.
+ * Prioritize that over LLC preference.
+ */
+ if (env->migration_type == migrate_misfit)
+ return false;
+
if (env->src_rq->nr_running <= 1)
return true;
@@ -10649,7 +10680,8 @@ alb_break_llc(struct lb_env *env)
if (cur && cur->sched_class == &fair_sched_class)
util = task_util(cur);
- if (can_migrate_llc(env->src_cpu, env->dst_cpu,
+ if (task_misfits_asym_cpu(env, cur) ||
+ can_migrate_llc(env->src_cpu, env->dst_cpu,
util, false) == mig_forbid)
return true;
}
@@ -10689,8 +10721,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu))
return true;
- if (can_migrate_llc_task(env->src_cpu,
- env->dst_cpu, p) != mig_forbid)
+ if (can_migrate_llc_task(env, p) != mig_forbid)
return false;
return true;
@@ -11753,6 +11784,15 @@ static inline bool llc_balance(struct lb_env *env, struct sg_lb_stats *sgs,
if (env->sd->flags & SD_SHARE_LLC)
return false;
+ /*
+ * On asymmetric domains, group_misfit_task_load
+ * should be prioritized to move tasks to CPU that fit them
+ * over aggregating tasks to their preferred LLC.
+ */
+ if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
+ sgs->group_misfit_task_load)
+ return false;
+
/*
* Skip cache aware tagging if nr_balanced_failed is sufficiently high.
* Threshold of cache_nice_tries is set to 1 higher than nr_balance_failed
@@ -0,0 +1,88 @@
diff --git a/include/linux/sched/sd_flags.h b/include/linux/sched/sd_flags.h
--- a/include/linux/sched/sd_flags.h
+++ b/include/linux/sched/sd_flags.h
@@ -146,8 +146,7 @@ SD_FLAG(SD_ASYM_PACKING, SDF_NEEDS_GROUPS)
/*
* Prefer to place tasks in a sibling domain
*
- * Set up until domains start spanning NUMA nodes. Close to being a SHARED_CHILD
- * flag, but cleared below domains with SD_ASYM_CPUCAPACITY.
+ * Set up until domains start spanning NUMA nodes.
*
* NEEDS_GROUPS: Load balancing flag.
*/
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -11912,12 +11912,25 @@ static inline void update_sg_lb_stats(struct lb_env *env,
continue;
if (sd_flags & SD_ASYM_CPUCAPACITY) {
- /* Check for a misfit task on the cpu */
- if (sgs->group_misfit_task_load < rq->misfit_task_load) {
- sgs->group_misfit_task_load = rq->misfit_task_load;
-
+ if (rq->misfit_task_load) {
+ /*
+ * Always mark the root domain overloaded so big
+ * CPUs can pick up misfit tasks via newly idle
+ * balance.
+ */
if (balancing_at_rd)
*sg_overloaded = 1;
+
+ /*
+ * Only account misfit load if @dst_cpu can
+ * help; otherwise, the group may be classified
+ * as misfit_task and update_sd_pick_busiest()
+ * will skip it.
+ */
+ if (capacity_greater(capacity_of(env->dst_cpu),
+ group->sgc->max_capacity) &&
+ (sgs->group_misfit_task_load < rq->misfit_task_load))
+ sgs->group_misfit_task_load = rq->misfit_task_load;
}
} else if (env->idle && sched_reduced_capacity(rq, env->sd)) {
/* Check for a task running on a CPU with reduced capacity */
@@ -13030,9 +13043,24 @@ static struct rq *sched_balance_find_src_rq(struct lb_env *env,
* average load.
*/
if (env->sd->flags & SD_ASYM_CPUCAPACITY &&
- !capacity_greater(capacity_of(env->dst_cpu), capacity) &&
- nr_running == 1)
- continue;
+ nr_running == 1) {
+ bool cluster_equal_cap = static_branch_unlikely(&sched_cluster_active) &&
+ (get_actual_cpu_capacity(env->dst_cpu) ==
+ get_actual_cpu_capacity(i));
+ bool smt_degraded_cap = sched_smt_active() && !is_core_idle(i);
+
+ /*
+ * Busy SMT siblings reduce the capacity of CPU @i. Do
+ * not skip it in this case.
+ *
+ * CONFIG_SCHED_CLUSTER requires balancing load across
+ * clusters of identical capacity, accounting for
+ * hardware and cpufreq pressure.
+ */
+ if (!smt_degraded_cap && !cluster_equal_cap &&
+ !capacity_greater(capacity_of(env->dst_cpu), capacity))
+ continue;
+ }
/*
* Make sure we only pull tasks from a CPU of lower priority
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -1995,10 +1995,6 @@ sd_init(struct sched_domain_topology_level *tl,
/*
* Convert topological properties into behaviour.
*/
- /* Don't attempt to spread across CPUs of different capacities. */
- if ((sd->flags & SD_ASYM_CPUCAPACITY) && sd->child)
- sd->child->flags &= ~SD_PREFER_SIBLING;
-
if (sd->flags & SD_SHARE_CPUCAPACITY) {
sd->imbalance_pct = 110;
@@ -0,0 +1,77 @@
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -13934,29 +13934,62 @@ static inline int on_null_domain(struct rq *rq)
*/
static inline int find_new_ilb(void)
{
- int this_cpu = smp_processor_id();
- const struct cpumask *hk_mask;
- int ilb_cpu;
+ struct cpumask *ilb_cpus;
+ int ilb_cpu, fallback = -1;
- hk_mask = housekeeping_cpumask(HK_TYPE_KERNEL_NOISE);
+ lockdep_assert_irqs_disabled();
- for_each_cpu_and(ilb_cpu, nohz.idle_cpus_mask, hk_mask) {
- if (ilb_cpu == this_cpu)
+ /*
+ * Reuse the per-CPU select_rq_mask, which is protected from concurrent
+ * use on this CPU by having interrupts disabled.
+ */
+ ilb_cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
+ cpumask_and(ilb_cpus, nohz.idle_cpus_mask,
+ housekeeping_cpumask(HK_TYPE_KERNEL_NOISE));
+
+ for_each_cpu(ilb_cpu, ilb_cpus) {
+ if (!idle_cpu(ilb_cpu)) {
+ /*
+ * Once an idle fallback exists, a busy CPU proves that
+ * this core cannot be fully idle. Skip its siblings.
+ */
+ if (sched_smt_active() && fallback >= 0)
+ cpumask_andnot(ilb_cpus, ilb_cpus, cpu_smt_mask(ilb_cpu));
continue;
+ }
- if (idle_cpu(ilb_cpu))
- return ilb_cpu;
+ /*
+ * Running the idle load balancer on an idle sibling of a busy
+ * SMT core can reduce the capacity available to its sibling. Prefer
+ * a CPU whose entire core is idle, but retain the first idle CPU as
+ * a fallback so idle balancing can still make progress when no fully
+ * idle core exists.
+ */
+ if (sched_smt_active() && !is_core_idle(ilb_cpu)) {
+ if (fallback < 0)
+ fallback = ilb_cpu;
+
+ /*
+ * The core is not idle, so there is no need to check
+ * any of its other SMT siblings.
+ */
+ cpumask_andnot(ilb_cpus, ilb_cpus,
+ cpu_smt_mask(ilb_cpu));
+ continue;
+ }
+
+ return ilb_cpu;
}
- return -1;
+ return fallback;
}
/*
* Kick a CPU to do the NOHZ balancing, if it is time for it, via a cross-CPU
* SMP function call (IPI).
*
- * We pick the first idle CPU in the HK_TYPE_KERNEL_NOISE housekeeping set
- * (if there is one).
+ * Prefer a CPU on a fully idle core in the HK_TYPE_KERNEL_NOISE housekeeping
+ * set. Fall back to the first idle CPU when no fully idle core exists.
*/
static void kick_ilb(unsigned int flags)
{
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,104 @@
diff --git a/drivers/opp/core.c b/drivers/opp/core.c
--- a/drivers/opp/core.c
+++ b/drivers/opp/core.c
@@ -453,8 +453,8 @@ int dev_pm_opp_get_opp_count(struct device *dev)
_find_opp_table(dev);
if (IS_ERR(opp_table)) {
- dev_dbg(dev, "%s: OPP table not found (%ld)\n",
- __func__, PTR_ERR(opp_table));
+ dev_dbg(dev, "%s: OPP table not found (%pe)\n",
+ __func__, opp_table);
return PTR_ERR(opp_table);
}
@@ -611,8 +611,8 @@ _find_key(struct device *dev, unsigned long *key, int index, bool available,
_find_opp_table(dev);
if (IS_ERR(opp_table)) {
- dev_err(dev, "%s: OPP table not found (%ld)\n", __func__,
- PTR_ERR(opp_table));
+ dev_err(dev, "%s: OPP table not found (%pe)\n", __func__,
+ opp_table);
return ERR_CAST(opp_table);
}
@@ -722,8 +722,8 @@ struct dev_pm_opp *dev_pm_opp_find_key_exact(struct device *dev,
struct opp_table *opp_table __free(put_opp_table) = _find_opp_table(dev);
if (IS_ERR(opp_table)) {
- dev_err(dev, "%s: OPP table not found (%ld)\n", __func__,
- PTR_ERR(opp_table));
+ dev_err(dev, "%s: OPP table not found (%pe)\n", __func__,
+ opp_table);
return ERR_CAST(opp_table);
}
@@ -1036,8 +1036,8 @@ static int _set_opp_voltage(struct device *dev, struct regulator *reg,
/* Regulator not available for device */
if (IS_ERR(reg)) {
- dev_dbg(dev, "%s: regulator not available: %ld\n", __func__,
- PTR_ERR(reg));
+ dev_dbg(dev, "%s: regulator not available: %pe\n", __func__,
+ reg);
return 0;
}
@@ -1448,8 +1448,8 @@ int dev_pm_opp_set_rate(struct device *dev, unsigned long target_freq)
temp_freq = freq;
opp = _find_freq_ceil(opp_table, &temp_freq);
if (IS_ERR(opp)) {
- dev_err(dev, "%s: failed to find OPP for freq %lu (%ld)\n",
- __func__, freq, PTR_ERR(opp));
+ dev_err(dev, "%s: failed to find OPP for freq %lu (%pe)\n",
+ __func__, freq, opp);
return PTR_ERR(opp);
}
@@ -1581,6 +1581,8 @@ static struct opp_table *_update_opp_table_clk(struct device *dev,
struct opp_table *opp_table,
bool getclk)
{
+ int ret;
+
/*
* Return early if we don't need to get clk or we have already done it
* earlier.
@@ -1607,9 +1609,9 @@ static struct opp_table *_update_opp_table_clk(struct device *dev,
opp_table->clk = clk_get_optional(dev, NULL);
if (IS_ERR(opp_table->clk)) {
+ ret = dev_err_probe(dev, PTR_ERR(opp_table->clk), "Couldn't find clock\n");
dev_pm_opp_put_opp_table(opp_table);
- dev_err_probe(dev, PTR_ERR(opp_table->clk), "Couldn't find clock\n");
- return ERR_CAST(opp_table->clk);
+ return ERR_PTR(ret);
}
if (opp_table->clk)
@@ -2869,8 +2871,8 @@ static int _opp_set_availability(struct device *dev, unsigned long freq,
struct dev_pm_opp *opp __free(put_opp) = ERR_PTR(-ENODEV), *tmp_opp;
if (IS_ERR(opp_table)) {
- dev_warn(dev, "%s: Device OPP not found (%ld)\n", __func__,
- PTR_ERR(opp_table));
+ dev_warn(dev, "%s: Device OPP not found (%pe)\n", __func__,
+ opp_table);
return PTR_ERR(opp_table);
}
diff --git a/drivers/opp/of.c b/drivers/opp/of.c
--- a/drivers/opp/of.c
+++ b/drivers/opp/of.c
@@ -1345,8 +1345,8 @@ int of_get_required_opp_performance_state(struct device_node *np, int index)
_find_table_of_opp_np(required_np);
if (IS_ERR(opp_table)) {
- pr_err("%s: Failed to find required OPP table %pOF: %ld\n",
- __func__, np, PTR_ERR(opp_table));
+ pr_err("%s: Failed to find required OPP table %pOF: %pe\n",
+ __func__, np, opp_table);
return PTR_ERR(opp_table);
}
@@ -0,0 +1,54 @@
diff --git a/kernel/power/hibernate.c b/kernel/power/hibernate.c
--- a/kernel/power/hibernate.c
+++ b/kernel/power/hibernate.c
@@ -408,9 +408,18 @@ int hibernation_snapshot(int platform_mode)
if (error)
goto Close;
+ error = dpm_prepare(PMSG_FREEZE);
+ if (error)
+ goto Complete;
+
+ /* Preallocate image memory before freezing kernel threads and shutting down devices. */
+ error = hibernate_preallocate_memory();
+ if (error)
+ goto Complete;
+
error = freeze_kernel_threads();
if (error)
- goto Close;
+ goto Cleanup;
if (hibernation_test(TEST_FREEZER)) {
@@ -422,15 +431,6 @@ int hibernation_snapshot(int platform_mode)
goto Thaw;
}
- error = dpm_prepare(PMSG_FREEZE);
- if (error)
- goto Complete;
-
- /* Preallocate image memory before shutting down devices. */
- error = hibernate_preallocate_memory();
- if (error)
- goto Complete;
-
console_suspend_all();
pm_restrict_gfp_mask();
@@ -464,10 +464,12 @@ int hibernation_snapshot(int platform_mode)
platform_end(platform_mode);
return error;
- Complete:
- dpm_complete(PMSG_RECOVER);
Thaw:
thaw_kernel_threads();
+ Cleanup:
+ swsusp_free();
+ Complete:
+ dpm_complete(PMSG_RECOVER);
goto Close;
}
@@ -0,0 +1,61 @@
diff --git a/drivers/cpufreq/cpufreq.c b/drivers/cpufreq/cpufreq.c
--- a/drivers/cpufreq/cpufreq.c
+++ b/drivers/cpufreq/cpufreq.c
@@ -2590,8 +2590,8 @@ static void cpufreq_update_pressure(struct cpufreq_policy *policy)
cpu = cpumask_first(policy->related_cpus);
max_freq = arch_scale_freq_ref(cpu);
- if (!max_freq)
- max_freq = policy->cpuinfo.max_freq;
+ if (!max_freq && cpufreq_driver->scale_freq_ref)
+ max_freq = cpufreq_driver->scale_freq_ref(policy);
capped_freq = policy->max;
diff --git a/drivers/cpufreq/intel_pstate.c b/drivers/cpufreq/intel_pstate.c
--- a/drivers/cpufreq/intel_pstate.c
+++ b/drivers/cpufreq/intel_pstate.c
@@ -1175,6 +1175,14 @@ static bool hybrid_clear_max_perf_cpu(void)
return ret;
}
+static unsigned int intel_pstate_scale_freq_ref(struct cpufreq_policy *policy)
+{
+ if (READ_ONCE(all_cpu_data[policy->cpu]->capacity_perf))
+ return policy->cpuinfo.max_freq;
+
+ return 0;
+}
+
static void intel_pstate_update_freq_limits(struct cpudata *cpu)
{
int scaling = cpu->pstate.scaling;
@@ -3092,6 +3100,7 @@ static struct cpufreq_driver intel_pstate = {
.offline = intel_pstate_cpu_offline,
.online = intel_pstate_cpu_online,
.update_limits = intel_pstate_update_limits,
+ .scale_freq_ref = intel_pstate_scale_freq_ref,
.name = "intel_pstate",
};
@@ -3415,6 +3424,7 @@ static struct cpufreq_driver intel_cpufreq = {
.suspend = intel_cpufreq_suspend,
.resume = intel_pstate_resume,
.update_limits = intel_pstate_update_limits,
+ .scale_freq_ref = intel_pstate_scale_freq_ref,
.name = "intel_cpufreq",
};
diff --git a/include/linux/cpufreq.h b/include/linux/cpufreq.h
--- a/include/linux/cpufreq.h
+++ b/include/linux/cpufreq.h
@@ -420,6 +420,9 @@ struct cpufreq_driver {
/* Will be called after the driver is fully initialized */
void (*ready)(struct cpufreq_policy *policy);
+ /* Return the capacity reference frequency for policy. */
+ unsigned int (*scale_freq_ref)(struct cpufreq_policy *policy);
+
struct freq_attr **attr;
/* platform specific boost support code */
@@ -0,0 +1,95 @@
diff --git a/arch/x86/include/asm/cpufeatures.h b/arch/x86/include/asm/cpufeatures.h
--- a/arch/x86/include/asm/cpufeatures.h
+++ b/arch/x86/include/asm/cpufeatures.h
@@ -471,6 +471,8 @@
#define X86_FEATURE_AUTOIBRS (20*32+ 8) /* Automatic IBRS */
#define X86_FEATURE_NO_SMM_CTL_MSR (20*32+ 9) /* SMM_CTL MSR is not present */
+#define X86_FEATURE_L2_TLB_SIZE_X32 (20*32+14) /* L2 TLB sizes are encoded as multiples of 32 */
+
#define X86_FEATURE_GP_ON_USER_CPUID (20*32+17) /* User CPUID faulting */
#define X86_FEATURE_PREFETCHI (20*32+20) /* Prefetch Data/Instruction to Cache Level */
diff --git a/arch/x86/kernel/cpu/amd.c b/arch/x86/kernel/cpu/amd.c
--- a/arch/x86/kernel/cpu/amd.c
+++ b/arch/x86/kernel/cpu/amd.c
@@ -1190,7 +1190,7 @@ static unsigned int amd_size_cache(struct cpuinfo_x86 *c, unsigned int size)
static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
{
- u32 ebx, eax, ecx, edx;
+ u32 ebx, eax, ecx, edx, shift, tmp;
u16 mask = 0xfff;
if (c->x86 < 0xf)
@@ -1199,10 +1199,12 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
if (c->extended_cpuid_level < 0x80000006)
return;
+ shift = !!cpu_has(c, X86_FEATURE_L2_TLB_SIZE_X32) * 5;
+
cpuid(0x80000006, &eax, &ebx, &ecx, &edx);
- tlb_lld_4k = (ebx >> 16) & mask;
- tlb_lli_4k = ebx & mask;
+ tlb_lld_4k = ((ebx >> 16) & mask) << shift;
+ tlb_lli_4k = (ebx & mask) << shift;
/*
* K8 doesn't have 2M/4M entries in the L2 TLB so read out the L1 TLB
@@ -1214,16 +1216,18 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
}
/* Handle DTLB 2M and 4M sizes, fall back to L1 if L2 is disabled */
- if (!((eax >> 16) & mask))
+ tmp = ((eax >> 16) & mask) << shift;
+ if (!tmp)
tlb_lld_2m = (cpuid_eax(0x80000005) >> 16) & 0xff;
else
- tlb_lld_2m = (eax >> 16) & mask;
+ tlb_lld_2m = tmp;
/* a 4M entry uses two 2M entries */
tlb_lld_4m = tlb_lld_2m >> 1;
/* Handle ITLB 2M and 4M sizes, fall back to L1 if L2 is disabled */
- if (!(eax & mask)) {
+ tmp = (eax & mask) << shift;
+ if (!tmp) {
/* Erratum 658 */
if (c->x86 == 0x15 && c->x86_model <= 0x1f) {
tlb_lli_2m = 1024;
@@ -1231,8 +1235,9 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c)
cpuid(0x80000005, &eax, &ebx, &ecx, &edx);
tlb_lli_2m = eax & 0xff;
}
- } else
- tlb_lli_2m = eax & mask;
+ } else {
+ tlb_lli_2m = tmp;
+ }
tlb_lli_4m = tlb_lli_2m >> 1;
diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c
--- a/arch/x86/kernel/cpu/common.c
+++ b/arch/x86/kernel/cpu/common.c
@@ -857,7 +857,7 @@ static void get_model_name(struct cpuinfo_x86 *c)
void cpu_detect_cache_sizes(struct cpuinfo_x86 *c)
{
- unsigned int n, dummy, ebx, ecx, edx, l2size;
+ unsigned int n, dummy, ebx, ecx, edx, l2size, shift __maybe_unused;
n = c->extended_cpuid_level;
@@ -877,7 +877,9 @@ void cpu_detect_cache_sizes(struct cpuinfo_x86 *c)
l2size = ecx >> 16;
#ifdef CONFIG_X86_64
+ shift = !!cpu_has(c, X86_FEATURE_L2_TLB_SIZE_X32) * 5;
c->x86_tlbsize += ((ebx >> 16) & 0xfff) + (ebx & 0xfff);
+ c->x86_tlbsize <<= shift;
#else
/* do processor-specific cache resizing */
if (this_cpu->legacy_cache_size)
@@ -0,0 +1,396 @@
diff --git a/mm/zsmalloc.c b/mm/zsmalloc.c
--- a/mm/zsmalloc.c
+++ b/mm/zsmalloc.c
@@ -21,6 +21,10 @@
* pool->lock
* class->lock
* zspage->lock
+ *
+ * When ZS_OBJ_CLASS_BITS > 0, zs_free() skips pool->lock; it picks
+ * the size_class from obj's encoded class_idx and serializes against
+ * page migration via class->lock.
*/
#include <linux/module.h>
@@ -67,8 +71,8 @@
#define MAX_POSSIBLE_PHYSMEM_BITS MAX_PHYSMEM_BITS
#else
/*
- * If this definition of MAX_PHYSMEM_BITS is used, OBJ_INDEX_BITS will just
- * be PAGE_SHIFT
+ * If this definition of MAX_PHYSMEM_BITS is used, ZS_OBJ_PFN_SHIFT will
+ * just be PAGE_SHIFT
*/
#define MAX_POSSIBLE_PHYSMEM_BITS BITS_PER_LONG
#endif
@@ -88,8 +92,23 @@
#define OBJ_TAG_BITS 1
#define OBJ_TAG_MASK OBJ_ALLOCATED_TAG
-#define OBJ_INDEX_BITS (BITS_PER_LONG - _PFN_BITS)
-#define OBJ_INDEX_MASK ((_AC(1, UL) << OBJ_INDEX_BITS) - 1)
+/*
+ * obj is encoded as [PFN | class_idx | obj_idx] within an unsigned long:
+ *
+ * |<-- _PFN_BITS -->|<-- ZS_OBJ_CLASS_BITS -->|<-- ZS_OBJ_IDX_BITS -->|
+ * +-----------------+-------------------------+-----------------------+
+ * | PFN | class_idx | obj_idx |
+ * +-----------------+-------------------------+-----------------------+
+ * MSB ^ LSB
+ * |
+ * +-- ZS_OBJ_PFN_SHIFT
+ *
+ * Encoding class_idx into obj lets zs_free() locate the size_class
+ * without holding pool->lock; class_idx is invariant across page
+ * migration (only PFN changes), so a lockless read of the obj value
+ * always yields a valid class_idx.
+ */
+#define ZS_OBJ_PFN_SHIFT (BITS_PER_LONG - _PFN_BITS)
#define HUGE_BITS 1
#define FULLNESS_BITS 4
@@ -98,9 +117,61 @@
#define ZS_MAX_PAGES_PER_ZSPAGE (_AC(CONFIG_ZSMALLOC_CHAIN_SIZE, UL))
+/*
+ * Bits to index a page within a zspage = ceil(log2(ZS_MAX_PAGES_PER_ZSPAGE)).
+ * Computed at preprocessor time, for use in #if below. Kconfig
+ * restricts ZSMALLOC_CHAIN_SIZE to [4, 16].
+ */
+#if ZS_MAX_PAGES_PER_ZSPAGE <= 4
+#define ZS_PAGES_PER_ZSPAGE_BITS 2
+#elif ZS_MAX_PAGES_PER_ZSPAGE <= 8
+#define ZS_PAGES_PER_ZSPAGE_BITS 3
+#elif ZS_MAX_PAGES_PER_ZSPAGE <= 16
+#define ZS_PAGES_PER_ZSPAGE_BITS 4
+#else
+#error "ZSMALLOC_CHAIN_SIZE out of expected range [4,16]"
+#endif
+
+/*
+ * Bits to index an object within a single PAGE_SIZE at the smallest
+ * possible object size: log2(PAGE_SIZE / 32) = PAGE_SHIFT - 5.
+ * 32 is the hard floor of ZS_MIN_ALLOC_SIZE.
+ */
+#define ZS_OBJS_PER_PAGE_BITS (PAGE_SHIFT - 5)
+
+/*
+ * Bits to index any object in the densest possible zspage. Below this,
+ * ZS_MIN_ALLOC_SIZE is auto-raised by the MAX(32, ...) formula -- still
+ * correct, but objects are coarser.
+ */
+#define ZS_OBJS_PER_ZSPAGE_BITS \
+ (ZS_PAGES_PER_ZSPAGE_BITS + ZS_OBJS_PER_PAGE_BITS)
+
+/*
+ * Encode class_idx only when obj has spare bits; otherwise
+ * ZS_OBJ_CLASS_BITS folds to 0 (32-bit, or 64-bit UML/fallback).
+ */
+#if BITS_PER_LONG >= 64 && \
+ ZS_OBJ_PFN_SHIFT >= (CLASS_BITS + 1) + ZS_OBJS_PER_ZSPAGE_BITS
+#define ZS_OBJ_CLASS_BITS (CLASS_BITS + 1)
+#else
+#define ZS_OBJ_CLASS_BITS 0
+#endif
+#define ZS_OBJ_CLASS_MASK ((_AC(1, UL) << ZS_OBJ_CLASS_BITS) - 1)
+
+#define ZS_OBJ_IDX_BITS (ZS_OBJ_PFN_SHIFT - ZS_OBJ_CLASS_BITS)
+#define ZS_OBJ_IDX_MASK ((_AC(1, UL) << ZS_OBJ_IDX_BITS) - 1)
+
+/*
+ * Belt-and-suspenders: the #if above already guarantees this when
+ * class_idx is enabled. Catches future tweaks that bypass it.
+ */
+static_assert(ZS_OBJ_IDX_BITS >= ZS_PAGES_PER_ZSPAGE_BITS,
+ "zsmalloc: ZS_MIN_ALLOC_SIZE would exceed ZS_MAX_ALLOC_SIZE");
+
/* ZS_MIN_ALLOC_SIZE must be multiple of ZS_ALIGN */
#define ZS_MIN_ALLOC_SIZE \
- MAX(32, (ZS_MAX_PAGES_PER_ZSPAGE << PAGE_SHIFT >> OBJ_INDEX_BITS))
+ MAX(32, (ZS_MAX_PAGES_PER_ZSPAGE << PAGE_SHIFT >> ZS_OBJ_IDX_BITS))
/* each chunk includes extra space to keep handle */
#define ZS_MAX_ALLOC_SIZE PAGE_SIZE
@@ -396,10 +467,13 @@ static void cache_free_zspage(struct zspage *zspage)
kmem_cache_free(zspage_cachep, zspage);
}
-/* class->lock(which owns the handle) synchronizes races */
+/*
+ * Pairs with READ_ONCE() in handle_to_obj(): zs_free() may read the
+ * handle locklessly, so prevent store tearing here.
+ */
static void record_obj(unsigned long handle, unsigned long obj)
{
- *(unsigned long *)handle = obj;
+ WRITE_ONCE(*(unsigned long *)handle, obj);
}
static inline bool __maybe_unused is_first_zpdesc(struct zpdesc *zpdesc)
@@ -725,33 +799,36 @@ static struct zpdesc *get_next_zpdesc(struct zpdesc *zpdesc)
static void obj_to_location(unsigned long obj, struct zpdesc **zpdesc,
unsigned int *obj_idx)
{
- *zpdesc = pfn_zpdesc(obj >> OBJ_INDEX_BITS);
- *obj_idx = (obj & OBJ_INDEX_MASK);
+ *zpdesc = pfn_zpdesc(obj >> ZS_OBJ_PFN_SHIFT);
+ *obj_idx = (obj & ZS_OBJ_IDX_MASK);
}
static void obj_to_zpdesc(unsigned long obj, struct zpdesc **zpdesc)
{
- *zpdesc = pfn_zpdesc(obj >> OBJ_INDEX_BITS);
+ *zpdesc = pfn_zpdesc(obj >> ZS_OBJ_PFN_SHIFT);
}
/**
- * location_to_obj - get obj value encoded from (<zpdesc>, <obj_idx>)
+ * location_to_obj - encode (<zpdesc>, <obj_idx>, <class_idx>) into obj value
* @zpdesc: zpdesc object resides in zspage
* @obj_idx: object index
+ * @class_idx: size class index; ignored when ZS_OBJ_CLASS_BITS == 0
*/
-static unsigned long location_to_obj(struct zpdesc *zpdesc, unsigned int obj_idx)
+static unsigned long location_to_obj(struct zpdesc *zpdesc, unsigned int obj_idx,
+ unsigned int class_idx)
{
unsigned long obj;
- obj = zpdesc_pfn(zpdesc) << OBJ_INDEX_BITS;
- obj |= obj_idx & OBJ_INDEX_MASK;
+ obj = zpdesc_pfn(zpdesc) << ZS_OBJ_PFN_SHIFT;
+ obj |= (unsigned long)(class_idx & ZS_OBJ_CLASS_MASK) << ZS_OBJ_IDX_BITS;
+ obj |= obj_idx & ZS_OBJ_IDX_MASK;
return obj;
}
static unsigned long handle_to_obj(unsigned long handle)
{
- return *(unsigned long *)handle;
+ return READ_ONCE(*(unsigned long *)handle);
}
static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj,
@@ -805,13 +882,26 @@ static int trylock_zspage(struct zspage *zspage)
return 0;
}
-static void __free_zspage(struct zs_pool *pool, struct size_class *class,
- struct zspage *zspage)
+/*
+ * Three free helpers, kept apart here:
+ *
+ * __free_zspage_lockless(): bare core; walks zpdescs and returns pages
+ * to the buddy allocator. Caller owns all zpdesc locks and has
+ * removed the zspage from its class list. Used by zs_free() outside
+ * class->lock so the buddy-side work does not stall the class.
+ *
+ * __free_zspage(): __free_zspage_lockless() + per-class accounting,
+ * under class->lock. Used by async_free_zspage(), the worker for
+ * zspages whose trylock_zspage() failed.
+ *
+ * free_zspage(): full wrapper - trylock zpdescs, remove from class
+ * list, call __free_zspage(); kicks deferred free on contention.
+ * Used by compaction.
+ */
+static inline void __free_zspage_lockless(struct zspage *zspage)
{
struct zpdesc *zpdesc, *next;
- assert_spin_locked(&class->lock);
-
VM_BUG_ON(get_zspage_inuse(zspage));
VM_BUG_ON(zspage->fullness != ZS_INUSE_RATIO_0);
@@ -827,7 +917,13 @@ static void __free_zspage(struct zs_pool *pool, struct size_class *class,
} while (zpdesc != NULL);
cache_free_zspage(zspage);
+}
+static void __free_zspage(struct zs_pool *pool, struct size_class *class,
+ struct zspage *zspage)
+{
+ assert_spin_locked(&class->lock);
+ __free_zspage_lockless(zspage);
class_stat_sub(class, ZS_OBJS_ALLOCATED, class->objs_per_zspage);
atomic_long_sub(class->pages_per_zspage, &pool->pages_allocated);
}
@@ -1280,7 +1376,7 @@ static unsigned long obj_malloc(struct zs_pool *pool,
kunmap_local(vaddr);
mod_zspage_inuse(zspage, 1);
- obj = location_to_obj(m_zpdesc, obj);
+ obj = location_to_obj(m_zpdesc, obj, zspage->class);
record_obj(handle, obj);
return obj;
@@ -1383,37 +1479,97 @@ static void obj_free(int class_size, unsigned long obj)
mod_zspage_inuse(zspage, -1);
}
+#if (ZS_OBJ_CLASS_BITS > 0) || defined(CONFIG_COMPACTION)
+/* Folds to 0 when ZS_OBJ_CLASS_BITS == 0; no ifdef needed at callers. */
+static unsigned int obj_to_class_idx(unsigned long obj)
+{
+ return (obj >> ZS_OBJ_IDX_BITS) & ZS_OBJ_CLASS_MASK;
+}
+#endif
+
+/*
+ * Resolve @handle to its zspage / size_class and acquire class->lock.
+ *
+ * When class_idx is encoded in obj (ZS_OBJ_CLASS_BITS > 0), it is
+ * invariant under page migration, so the handle can be read locklessly
+ * to pick the size_class. Once class->lock is held migration is
+ * blocked and the handle is re-read to obtain a stable PFN.
+ *
+ * Otherwise (32-bit, or 64-bit fallback paths like UML where the
+ * encoding is disabled), fall back to pool->lock for the lookup.
+ */
+#if ZS_OBJ_CLASS_BITS > 0
+static inline void obj_class_get_and_lock(struct zs_pool *pool, unsigned long handle,
+ unsigned long *objp, struct zspage **zspagep,
+ struct size_class **classp)
+ __acquires(&(*classp)->lock)
+{
+ struct zpdesc *f_zpdesc;
+ unsigned long obj;
+
+ obj = handle_to_obj(handle);
+ *classp = pool->size_class[obj_to_class_idx(obj)];
+ spin_lock(&(*classp)->lock);
+ /* Re-read under class->lock: PFN is now stable vs migration. */
+ obj = handle_to_obj(handle);
+ obj_to_zpdesc(obj, &f_zpdesc);
+ *zspagep = get_zspage(f_zpdesc);
+ *objp = obj;
+}
+#else
+static inline void obj_class_get_and_lock(struct zs_pool *pool, unsigned long handle,
+ unsigned long *objp, struct zspage **zspagep,
+ struct size_class **classp)
+ __acquires(&(*classp)->lock)
+{
+ struct zpdesc *f_zpdesc;
+ unsigned long obj;
+
+ read_lock(&pool->lock);
+ obj = handle_to_obj(handle);
+ obj_to_zpdesc(obj, &f_zpdesc);
+ *zspagep = get_zspage(f_zpdesc);
+ *classp = zspage_class(pool, *zspagep);
+ spin_lock(&(*classp)->lock);
+ read_unlock(&pool->lock);
+ *objp = obj;
+}
+#endif
+
void zs_free(struct zs_pool *pool, unsigned long handle)
{
struct zspage *zspage;
- struct zpdesc *f_zpdesc;
unsigned long obj;
struct size_class *class;
int fullness;
+ struct zspage *zspage_to_free = NULL;
if (IS_ERR_OR_NULL((void *)handle))
return;
- /*
- * The pool->lock protects the race with zpage's migration
- * so it's safe to get the page from handle.
- */
- read_lock(&pool->lock);
- obj = handle_to_obj(handle);
- obj_to_zpdesc(obj, &f_zpdesc);
- zspage = get_zspage(f_zpdesc);
- class = zspage_class(pool, zspage);
- spin_lock(&class->lock);
- read_unlock(&pool->lock);
+ obj_class_get_and_lock(pool, handle, &obj, &zspage, &class);
class_stat_sub(class, ZS_OBJS_INUSE, 1);
obj_free(class->size, obj);
fullness = fix_fullness_group(class, zspage);
- if (fullness == ZS_INUSE_RATIO_0)
- free_zspage(pool, class, zspage);
+ if (fullness == ZS_INUSE_RATIO_0) {
+ if (trylock_zspage(zspage)) {
+ remove_zspage(class, zspage);
+ class_stat_sub(class, ZS_OBJS_ALLOCATED,
+ class->objs_per_zspage);
+ zspage_to_free = zspage;
+ } else {
+ kick_deferred_free(pool);
+ }
+ }
spin_unlock(&class->lock);
+
+ if (zspage_to_free) {
+ __free_zspage_lockless(zspage_to_free);
+ atomic_long_sub(class->pages_per_zspage, &pool->pages_allocated);
+ }
cache_free_handle(handle);
}
EXPORT_SYMBOL_GPL(zs_free);
@@ -1646,9 +1802,6 @@ static void lock_zspage(struct zspage *zspage)
}
zspage_read_unlock(zspage);
}
-#endif /* CONFIG_COMPACTION */
-
-#ifdef CONFIG_COMPACTION
static void replace_sub_page(struct size_class *class, struct zspage *zspage,
struct zpdesc *newzpdesc, struct zpdesc *oldzpdesc)
@@ -1715,8 +1868,8 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
pool = zspage->pool;
/*
- * The pool migrate_lock protects the race between zpage migration
- * and zs_free.
+ * The pool migrate_lock protects against races between zpage migration
+ * and zs_free(), but only when ZS_OBJ_CLASS_BITS does not apply.
*/
write_lock(&pool->lock);
class = zspage_class(pool, zspage);
@@ -1764,7 +1917,8 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
old_obj = handle_to_obj(handle);
obj_to_location(old_obj, &dummy, &obj_idx);
- new_obj = (unsigned long)location_to_obj(newzpdesc, obj_idx);
+ new_obj = location_to_obj(newzpdesc, obj_idx,
+ obj_to_class_idx(old_obj));
record_obj(handle, new_obj);
}
}
@@ -1775,9 +1929,9 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
* Since we complete the data copy and set up new zspage structure,
* it's okay to release migration_lock.
*/
- write_unlock(&pool->lock);
- spin_unlock(&class->lock);
zspage_write_unlock(zspage);
+ spin_unlock(&class->lock);
+ write_unlock(&pool->lock);
zpdesc_get(newzpdesc);
if (zpdesc_zone(newzpdesc) != zpdesc_zone(zpdesc)) {
@@ -1894,8 +2048,9 @@ static unsigned long __zs_compact(struct zs_pool *pool,
unsigned long pages_freed = 0;
/*
- * protect the race between zpage migration and zs_free
- * as well as zpage allocation/free
+ * Protect against races between zpage migration and zs_free()
+ * (only when ZS_OBJ_CLASS_BITS does not apply), as well as
+ * zpage allocation and free.
*/
write_lock(&pool->lock);
spin_lock(&class->lock);
@@ -0,0 +1,313 @@
diff --git a/include/linux/rmap.h b/include/linux/rmap.h
--- a/include/linux/rmap.h
+++ b/include/linux/rmap.h
@@ -843,7 +843,7 @@ static inline int folio_try_share_anon_rmap_pmd(struct folio *folio,
* Called from mm/vmscan.c to handle paging out
*/
int folio_referenced(struct folio *, int is_locked,
- struct mem_cgroup *memcg, vm_flags_t *vm_flags);
+ struct mem_cgroup *memcg, vma_flags_t *vma_flags);
void try_to_migrate(struct folio *folio, enum ttu_flags flags);
void try_to_unmap(struct folio *, enum ttu_flags flags);
@@ -975,10 +975,9 @@ struct anon_vma *folio_lock_anon_vma_read(const struct folio *folio,
#define anon_vma_prepare(vma) (0)
static inline int folio_referenced(struct folio *folio, int is_locked,
- struct mem_cgroup *memcg,
- vm_flags_t *vm_flags)
+ struct mem_cgroup *memcg, vma_flags_t *vma_flags)
{
- *vm_flags = 0;
+ vma_flags_clear_all(vma_flags);
return 0;
}
diff --git a/mm/rmap.c b/mm/rmap.c
--- a/mm/rmap.c
+++ b/mm/rmap.c
@@ -907,7 +907,7 @@ pmd_t *mm_find_pmd(struct mm_struct *mm, unsigned long address)
struct folio_referenced_arg {
int mapcount;
int referenced;
- vm_flags_t vm_flags;
+ vma_flags_t vma_flags;
struct mem_cgroup *memcg;
};
@@ -926,7 +926,7 @@ static bool folio_referenced_one(struct folio *folio,
address = pvmw.address;
nr = 1;
- if (vma->vm_flags & VM_LOCKED) {
+ if (vma_test(vma, VMA_LOCKED_BIT)) {
ptes++;
pra->mapcount--;
@@ -947,7 +947,7 @@ static bool folio_referenced_one(struct folio *folio,
/* Restore the mlock which got missed */
mlock_vma_folio(folio, vma);
page_vma_mapped_walk_done(&pvmw);
- pra->vm_flags |= VM_LOCKED;
+ vma_flags_set(&pra->vma_flags, VMA_LOCKED_BIT);
return false; /* To break the loop */
}
@@ -1015,8 +1015,11 @@ static bool folio_referenced_one(struct folio *folio,
referenced++;
if (referenced) {
+ vma_flags_t vma_flags = vma->flags;
+
pra->referenced++;
- pra->vm_flags |= vma->vm_flags & ~VM_LOCKED;
+ vma_flags_clear(&vma_flags, VMA_LOCKED_BIT);
+ vma_flags_set_mask(&pra->vma_flags, vma_flags);
}
if (!pra->mapcount)
@@ -1054,7 +1057,7 @@ static bool invalid_folio_referenced_vma(struct vm_area_struct *vma, void *arg)
* @folio: The folio to test.
* @is_locked: Caller holds lock on the folio.
* @memcg: target memory cgroup
- * @vm_flags: A combination of all the vma->vm_flags which referenced the folio.
+ * @vma_flags: A combination of all the vma->flags which referenced the folio.
*
* Quick test_and_clear_referenced for all mappings of a folio,
*
@@ -1062,7 +1065,7 @@ static bool invalid_folio_referenced_vma(struct vm_area_struct *vma, void *arg)
* the function bailed out due to rmap lock contention.
*/
int folio_referenced(struct folio *folio, int is_locked,
- struct mem_cgroup *memcg, vm_flags_t *vm_flags)
+ struct mem_cgroup *memcg, vma_flags_t *vma_flags)
{
bool we_locked = false;
struct folio_referenced_arg pra = {
@@ -1078,7 +1081,7 @@ int folio_referenced(struct folio *folio, int is_locked,
};
VM_WARN_ON_ONCE_FOLIO(folio_is_zone_device(folio), folio);
- *vm_flags = 0;
+ vma_flags_clear_all(vma_flags);
if (!pra.mapcount)
return 0;
@@ -1092,7 +1095,7 @@ int folio_referenced(struct folio *folio, int is_locked,
}
rmap_walk(folio, &rwc);
- *vm_flags = pra.vm_flags;
+ vma_flags_set_mask(vma_flags, pra.vma_flags);
if (we_locked)
folio_unlock(folio);
diff --git a/mm/vmscan.c b/mm/vmscan.c
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -262,6 +262,12 @@ static bool writeback_throttling_sane(struct scan_control *sc)
}
#endif
+static inline bool is_exec_file_folio(const struct folio *folio,
+ const vma_flags_t *vma_flags)
+{
+ return vma_flags_test(vma_flags, VMA_EXEC_BIT) && folio_is_file_lru(folio);
+}
+
static void set_task_reclaim_state(struct task_struct *task,
struct reclaim_state *rs)
{
@@ -830,10 +836,16 @@ enum folio_references {
* with PG_active set. In contrast, the aging (page table walk) path uses
* folio_update_gen().
*/
-static bool lru_gen_set_refs(struct folio *folio)
+static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags)
{
/* see the comment on LRU_REFS_FLAGS */
if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) {
+ /* Activate file-backed executable folios after first usage. */
+ if (is_exec_file_folio(folio, vma_flags)) {
+ set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset));
+ return true;
+ }
+
set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
return false;
}
@@ -846,7 +858,7 @@ static bool lru_gen_set_refs(struct folio *folio)
return true;
}
#else
-static bool lru_gen_set_refs(struct folio *folio)
+static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags)
{
return false;
}
@@ -856,16 +868,16 @@ static enum folio_references folio_check_references(struct folio *folio,
struct scan_control *sc)
{
int referenced_ptes, referenced_folio;
- vm_flags_t vm_flags;
+ vma_flags_t vma_flags;
referenced_ptes = folio_referenced(folio, 1, sc->target_mem_cgroup,
- &vm_flags);
+ &vma_flags);
/*
* The supposedly reclaimable folio was found to be in a VM_LOCKED vma.
* Let the folio, now marked Mlocked, be moved to the unevictable list.
*/
- if (vm_flags & VM_LOCKED)
+ if (vma_flags_test(&vma_flags, VMA_LOCKED_BIT))
return FOLIOREF_ACTIVATE;
/*
@@ -881,7 +893,7 @@ static enum folio_references folio_check_references(struct folio *folio,
if (!referenced_ptes)
return FOLIOREF_RECLAIM;
- return lru_gen_set_refs(folio) ? FOLIOREF_ACTIVATE : FOLIOREF_KEEP;
+ return lru_gen_set_refs(folio, &vma_flags) ? FOLIOREF_ACTIVATE : FOLIOREF_KEEP;
}
referenced_folio = folio_test_clear_referenced(folio);
@@ -909,7 +921,7 @@ static enum folio_references folio_check_references(struct folio *folio,
/*
* Activate file-backed executable folios after first usage.
*/
- if ((vm_flags & VM_EXEC) && folio_is_file_lru(folio))
+ if (is_exec_file_folio(folio, &vma_flags))
return FOLIOREF_ACTIVATE;
return FOLIOREF_KEEP;
@@ -2067,7 +2079,7 @@ static void shrink_active_list(unsigned long nr_to_scan,
{
unsigned long nr_taken;
unsigned long nr_scanned;
- vm_flags_t vm_flags;
+ vma_flags_t vma_flags;
LIST_HEAD(l_hold); /* The folios which were snipped off */
LIST_HEAD(l_active);
LIST_HEAD(l_inactive);
@@ -2111,7 +2123,7 @@ static void shrink_active_list(unsigned long nr_to_scan,
/* Referenced or rmap lock contention: rotate */
if (folio_referenced(folio, 0, sc->target_mem_cgroup,
- &vm_flags) != 0) {
+ &vma_flags) != 0) {
/*
* Identify referenced, file-backed active folios and
* give them one more trip around the active list. So
@@ -2121,7 +2133,7 @@ static void shrink_active_list(unsigned long nr_to_scan,
* IO, plus JVM can create lots of anon VM_EXEC folios,
* so we ignore them here.
*/
- if ((vm_flags & VM_EXEC) && folio_is_file_lru(folio)) {
+ if (is_exec_file_folio(folio, &vma_flags)) {
nr_rotated += folio_nr_pages(folio);
list_add(&folio->lru, &l_active);
continue;
@@ -3190,14 +3202,19 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv)
******************************************************************************/
/* promote pages accessed through page tables */
-static int folio_update_gen(struct folio *folio, int gen)
+static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma_flags)
{
unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f);
VM_WARN_ON_ONCE(gen >= MAX_NR_GENS);
- /* see the comment on LRU_REFS_FLAGS */
- if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) {
+ /*
+ * See the comment on LRU_REFS_FLAGS, and activate file-backed
+ * executable folios after first usage to avoid typical IO
+ * thrashing from reclaiming.
+ */
+ if (!folio_test_referenced(folio) && !folio_test_workingset(folio) &&
+ !is_exec_file_folio(folio, vma_flags)) {
set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced));
return -1;
}
@@ -3430,8 +3447,8 @@ static bool suitable_to_scan(int total, int young)
return young * n >= total;
}
-static void walk_update_folio(struct lru_gen_mm_walk *walk, struct folio *folio,
- int new_gen, bool dirty)
+static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma,
+ struct folio *folio, int new_gen, bool dirty)
{
int old_gen;
@@ -3444,10 +3461,10 @@ static void walk_update_folio(struct lru_gen_mm_walk *walk, struct folio *folio,
folio_mark_dirty(folio);
if (walk) {
- old_gen = folio_update_gen(folio, new_gen);
+ old_gen = folio_update_gen(folio, new_gen, &vma->flags);
if (old_gen >= 0 && old_gen != new_gen)
update_batch_size(walk, folio, old_gen, new_gen);
- } else if (lru_gen_set_refs(folio)) {
+ } else if (lru_gen_set_refs(folio, &vma->flags)) {
old_gen = folio_lru_gen(folio);
if (old_gen >= 0 && old_gen != new_gen)
folio_activate(folio);
@@ -3520,7 +3537,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
continue;
if (last != folio) {
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, args->vma, last, gen, dirty);
last = folio;
dirty = false;
@@ -3533,7 +3550,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end,
walk->mm_stats[MM_LEAF_YOUNG] += nr;
}
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, args->vma, last, gen, dirty);
last = NULL;
if (i < PTRS_PER_PTE && get_next_vma(PMD_MASK, PAGE_SIZE, args, &start, &end))
@@ -3611,7 +3628,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
goto next;
if (last != folio) {
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
last = folio;
dirty = false;
@@ -3625,7 +3642,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area
i = i > MIN_LRU_BATCH ? 0 : find_next_bit(bitmap, MIN_LRU_BATCH, i) + 1;
} while (i <= MIN_LRU_BATCH);
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
lazy_mmu_mode_disable();
spin_unlock(ptl);
@@ -4260,7 +4277,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
continue;
if (last != folio) {
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
last = folio;
dirty = false;
@@ -4272,7 +4289,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr)
young += nr;
}
- walk_update_folio(walk, last, gen, dirty);
+ walk_update_folio(walk, vma, last, gen, dirty);
lazy_mmu_mode_disable();
@@ -0,0 +1,161 @@
diff --git a/mm/ksm.c b/mm/ksm.c
--- a/mm/ksm.c
+++ b/mm/ksm.c
@@ -195,22 +195,28 @@ struct ksm_stable_node {
* @node: rb node of this rmap_item in the unstable tree
* @head: pointer to stable_node heading this list in the stable tree
* @hlist: link into hlist of rmap_items hanging off that stable_node
- * @age: number of scan iterations since creation
- * @remaining_skips: how many scans to skip
+ * @age: number of scan iterations since creation (unstable node)
+ * @remaining_skips: how many scans to skip (unstable node)
+ * @linear_page_index: the original page's index before merged by KSM (stable node)
*/
struct ksm_rmap_item {
struct ksm_rmap_item *rmap_list;
union {
- struct anon_vma *anon_vma; /* when stable */
+ struct anon_vma *anon_vma; /* for reverse mapping, when stable */
#ifdef CONFIG_NUMA
int nid; /* when node of unstable tree */
#endif
};
struct mm_struct *mm;
unsigned long address; /* + low bits used for flags below */
- unsigned int oldchecksum; /* when unstable */
- rmap_age_t age;
- rmap_age_t remaining_skips;
+ union {
+ struct {
+ unsigned int oldchecksum;
+ rmap_age_t age;
+ rmap_age_t remaining_skips;
+ }; /* when unstable */
+ unsigned long linear_page_index; /* for reverse mapping, when stable */
+ };
union {
struct rb_node node; /* when node of unstable tree */
struct { /* when listed from stable tree */
@@ -776,6 +782,11 @@ static struct vm_area_struct *find_mergeable_vma(struct mm_struct *mm,
return vma;
}
+/*
+ * break_cow: actively break COW, replacing the KSM page by a fresh anonymous
+ * page. This is called when rmap_item has not yet become stable, but page
+ * has been merged.
+ */
static void break_cow(struct ksm_rmap_item *rmap_item)
{
struct mm_struct *mm = rmap_item->mm;
@@ -787,6 +798,11 @@ static void break_cow(struct ksm_rmap_item *rmap_item)
* to undo, we also need to drop a reference to the anon_vma.
*/
put_anon_vma(rmap_item->anon_vma);
+ /*
+ * Reset linear_page_index that might overlay age-related
+ * information. (it's still unstable node)
+ */
+ rmap_item->linear_page_index = 0;
mmap_read_lock(mm);
vma = find_mergeable_vma(mm, addr);
@@ -899,6 +915,8 @@ static void remove_node_from_stable_tree(struct ksm_stable_node *stable_node)
VM_BUG_ON(stable_node->rmap_hlist_len <= 0);
stable_node->rmap_hlist_len--;
put_anon_vma(rmap_item->anon_vma);
+ /* Reset linear_page_index that might overlay age-related information. */
+ rmap_item->linear_page_index = 0;
rmap_item->address &= PAGE_MASK;
cond_resched();
}
@@ -1052,6 +1070,8 @@ static void remove_rmap_item_from_tree(struct ksm_rmap_item *rmap_item)
stable_node->rmap_hlist_len--;
put_anon_vma(rmap_item->anon_vma);
+ /* Reset linear_page_index that might overlay age-related information. */
+ rmap_item->linear_page_index = 0;
rmap_item->head = NULL;
rmap_item->address &= PAGE_MASK;
@@ -1598,8 +1618,15 @@ static int try_to_merge_with_ksm_page(struct ksm_rmap_item *rmap_item,
/* Unstable nid is in union with stable anon_vma: remove first */
remove_rmap_item_from_tree(rmap_item);
- /* Must get reference to anon_vma while still holding mmap_lock */
+ /*
+ * We can consider the VMA only while still holding the mmap lock,
+ * so lock, so reference the anon_vma and calculate the linear
+ * page index early, before stable_tree_append(). If anything goes
+ * wrong that prevents the rmap_item from being added to the
+ * stable_tree, break_cow() will clean it up.
+ */
rmap_item->anon_vma = vma->anon_vma;
+ rmap_item->linear_page_index = linear_page_index(vma, rmap_item->address);
get_anon_vma(vma->anon_vma);
out:
mmap_read_unlock(mm);
@@ -2458,6 +2485,13 @@ static bool should_skip_rmap_item(struct folio *folio,
if (folio_test_ksm(folio))
return false;
+ /*
+ * There is no age information in stable-tree nodes. We might end up
+ * here without a KSM page for example after COW.
+ */
+ if (rmap_item->address & STABLE_FLAG)
+ return false;
+
age = rmap_item->age;
if (age != U8_MAX)
rmap_item->age++;
@@ -3173,6 +3207,7 @@ void rmap_walk_ksm(struct folio *folio, struct rmap_walk_control *rwc)
hlist_for_each_entry(rmap_item, &stable_node->hlist, hlist) {
/* Ignore the stable/unstable/sqnr flags */
const unsigned long addr = rmap_item->address & PAGE_MASK;
+ const unsigned long index = rmap_item->linear_page_index;
struct anon_vma *anon_vma = rmap_item->anon_vma;
struct anon_vma_chain *vmac;
struct vm_area_struct *vma;
@@ -3186,8 +3221,18 @@ void rmap_walk_ksm(struct folio *folio, struct rmap_walk_control *rwc)
anon_vma_lock_read(anon_vma);
}
+ /*
+ * Currently, KSM folios are always small folios, so it's
+ * sufficient to search for a single page. We can simply use
+ * the linear_page_index of the original de-duplicate
+ * anonymous page that we remembered in the rmap_item while
+ * de-duplicating. Note that mremap() always de-duplicates KSM
+ * folios: so if there was mremap() in our parent or our child,
+ * we wouldn't have the KSM folio mapped in these processes
+ * anymore.
+ */
anon_vma_interval_tree_foreach(vmac, &anon_vma->rb_root,
- 0, ULONG_MAX) {
+ index, index) {
cond_resched();
vma = vmac->vma;
@@ -3243,17 +3288,17 @@ void collect_procs_ksm(const struct folio *folio, const struct page *page,
rcu_read_lock();
for_each_process(tsk) {
struct anon_vma_chain *vmac;
- unsigned long addr;
+ const unsigned long addr = rmap_item->address & PAGE_MASK;
+ const unsigned long index = rmap_item->linear_page_index;
struct task_struct *t =
task_early_kill(tsk, force_early);
if (!t)
continue;
- anon_vma_interval_tree_foreach(vmac, &av->rb_root, 0,
- ULONG_MAX)
+ anon_vma_interval_tree_foreach(vmac, &av->rb_root, index,
+ index)
{
vma = vmac->vma;
if (vma->vm_mm == t->mm) {
- addr = rmap_item->address & PAGE_MASK;
add_to_kill_ksm(t, page, vma, to_kill,
addr);
}
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,35 @@
diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c
--- a/mm/pgtable-generic.c
+++ b/mm/pgtable-generic.c
@@ -438,12 +438,30 @@ static void kernel_pgtable_work_func(struct work_struct *work)
__pagetable_free(pt);
}
+static void schedule_kernel_pgtable_free(void)
+{
+ schedule_work(&kernel_pgtable_work.work);
+}
+
void pagetable_free_kernel(struct ptdesc *pt)
{
spin_lock(&kernel_pgtable_work.lock);
list_add(&pt->pt_list, &kernel_pgtable_work.list);
spin_unlock(&kernel_pgtable_work.lock);
- schedule_work(&kernel_pgtable_work.work);
+ /*
+ * The workqueue may not exist yet while the system is booting.
+ * kernel_pgtable_drain_early() schedules the work once it does.
+ */
+ if (system_state != SYSTEM_BOOTING)
+ schedule_kernel_pgtable_free();
}
+
+static int __init kernel_pgtable_drain_early(void)
+{
+ /* Free the kernel page tables queued while booting. */
+ schedule_kernel_pgtable_free();
+ return 0;
+}
+core_initcall(kernel_pgtable_drain_early);
#endif
@@ -0,0 +1,97 @@
diff --git a/lib/zstd/common/entropy_common.c b/lib/zstd/common/entropy_common.c
--- a/lib/zstd/common/entropy_common.c
+++ b/lib/zstd/common/entropy_common.c
@@ -202,6 +202,8 @@ BMI2_TARGET_ATTRIBUTE static size_t FSE_readNCount_body_bmi2(
{
return FSE_readNCount_body(normalizedCounter, maxSVPtr, tableLogPtr, headerBuffer, hbSize);
}
+#else
+#define FSE_readNCount_body_bmi2 FSE_readNCount_body_default
#endif
size_t FSE_readNCount_bmi2(
@@ -323,6 +325,8 @@ static BMI2_TARGET_ATTRIBUTE size_t HUF_readStats_body_bmi2(BYTE* huffWeight, si
{
return HUF_readStats_body(huffWeight, hwSize, rankStats, nbSymbolsPtr, tableLogPtr, src, srcSize, workSpace, wkspSize, 1);
}
+#else
+#define HUF_readStats_body_bmi2 HUF_readStats_body_default
#endif
size_t HUF_readStats_wksp(BYTE* huffWeight, size_t hwSize, U32* rankStats,
diff --git a/lib/zstd/common/fse_decompress.c b/lib/zstd/common/fse_decompress.c
--- a/lib/zstd/common/fse_decompress.c
+++ b/lib/zstd/common/fse_decompress.c
@@ -300,6 +300,8 @@ BMI2_TARGET_ATTRIBUTE static size_t FSE_decompress_wksp_body_bmi2(void* dst, siz
{
return FSE_decompress_wksp_body(dst, dstCapacity, cSrc, cSrcSize, maxLog, workSpace, wkspSize, 1);
}
+#else
+#define FSE_decompress_wksp_body_bmi2 FSE_decompress_wksp_body_default
#endif
size_t FSE_decompress_wksp_bmi2(void* dst, size_t dstCapacity, const void* cSrc, size_t cSrcSize, unsigned maxLog, void* workSpace, size_t wkspSize, int bmi2)
diff --git a/lib/zstd/compress/zstd_compress_sequences.c b/lib/zstd/compress/zstd_compress_sequences.c
--- a/lib/zstd/compress/zstd_compress_sequences.c
+++ b/lib/zstd/compress/zstd_compress_sequences.c
@@ -415,6 +415,10 @@ ZSTD_encodeSequences_bmi2(
sequences, nbSeq, longOffsets);
}
+#else
+
+#define ZSTD_encodeSequences_bmi2 ZSTD_encodeSequences_default
+
#endif
size_t ZSTD_encodeSequences(
diff --git a/lib/zstd/decompress/huf_decompress.c b/lib/zstd/decompress/huf_decompress.c
--- a/lib/zstd/decompress/huf_decompress.c
+++ b/lib/zstd/decompress/huf_decompress.c
@@ -700,6 +700,8 @@ size_t HUF_decompress4X1_usingDTable_internal_bmi2(void* dst, size_t dstSize, vo
size_t cSrcSize, HUF_DTable const* DTable) {
return HUF_decompress4X1_usingDTable_internal_body(dst, dstSize, cSrc, cSrcSize, DTable);
}
+#else
+#define HUF_decompress4X1_usingDTable_internal_bmi2 HUF_decompress4X1_usingDTable_internal_default
#endif
static
@@ -1503,6 +1505,8 @@ size_t HUF_decompress4X2_usingDTable_internal_bmi2(void* dst, size_t dstSize, vo
size_t cSrcSize, HUF_DTable const* DTable) {
return HUF_decompress4X2_usingDTable_internal_body(dst, dstSize, cSrc, cSrcSize, DTable);
}
+#else
+#define HUF_decompress4X2_usingDTable_internal_bmi2 HUF_decompress4X2_usingDTable_internal_default
#endif
static
diff --git a/lib/zstd/decompress/zstd_decompress_block.c b/lib/zstd/decompress/zstd_decompress_block.c
--- a/lib/zstd/decompress/zstd_decompress_block.c
+++ b/lib/zstd/decompress/zstd_decompress_block.c
@@ -622,6 +622,8 @@ BMI2_TARGET_ATTRIBUTE static void ZSTD_buildFSETable_body_bmi2(ZSTD_seqSymbol* d
ZSTD_buildFSETable_body(dt, normalizedCounter, maxSymbolValue,
baseValue, nbAdditionalBits, tableLog, wksp, wkspSize);
}
+#else
+#define ZSTD_buildFSETable_body_bmi2 ZSTD_buildFSETable_body_default
#endif
void ZSTD_buildFSETable(ZSTD_seqSymbol* dt,
@@ -1934,6 +1936,16 @@ ZSTD_decompressSequencesLong_bmi2(ZSTD_DCtx* dctx,
}
#endif /* ZSTD_FORCE_DECOMPRESS_SEQUENCES_SHORT */
+#else
+
+#ifndef ZSTD_FORCE_DECOMPRESS_SEQUENCES_LONG
+#define ZSTD_decompressSequences_bmi2 ZSTD_decompressSequences_default
+#define ZSTD_decompressSequencesSplitLitBuffer_bmi2 ZSTD_decompressSequencesSplitLitBuffer_default
+#endif
+#ifndef ZSTD_FORCE_DECOMPRESS_SEQUENCES_SHORT
+#define ZSTD_decompressSequencesLong_bmi2 ZSTD_decompressSequencesLong_default
+#endif
+
#endif /* DYNAMIC_BMI2 */
#ifndef ZSTD_FORCE_DECOMPRESS_SEQUENCES_LONG
@@ -0,0 +1,413 @@
diff --git a/lib/zstd/common/compiler.h b/lib/zstd/common/compiler.h
--- a/lib/zstd/common/compiler.h
+++ b/lib/zstd/common/compiler.h
@@ -14,6 +14,7 @@
#include <linux/types.h>
+#include "zstd_deps.h"
#include "portability_macros.h"
/*-*******************************************************
@@ -96,6 +97,17 @@
*/
#define BMI2_TARGET_ATTRIBUTE TARGET_ATTRIBUTE("lzcnt,bmi,bmi2")
+#if !DYNAMIC_BMI2
+# define ZSTD_USE_BMI2(bmi2) 0
+# define ZSTD_SET_BMI2(state, value) do { } while (0)
+#elif defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+# define ZSTD_USE_BMI2(bmi2) cpu_feature_enabled(X86_FEATURE_BMI2)
+# define ZSTD_SET_BMI2(state, value) do { } while (0)
+#else
+# define ZSTD_USE_BMI2(bmi2) (bmi2)
+# define ZSTD_SET_BMI2(state, value) do { (state) = (value); } while (0)
+#endif
+
/* prefetch
* can be disabled, by declaring NO_PREFETCH build macro */
#if ( (__GNUC__ >= 4) || ( (__GNUC__ == 3) && (__GNUC_MINOR__ >= 1) ) )
diff --git a/lib/zstd/common/entropy_common.c b/lib/zstd/common/entropy_common.c
--- a/lib/zstd/common/entropy_common.c
+++ b/lib/zstd/common/entropy_common.c
@@ -16,6 +16,10 @@
/* *************************************
* Dependencies
***************************************/
+#include "zstd_deps.h"
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "mem.h"
#include "error_private.h" /* ERR_*, ERROR */
#define FSE_STATIC_LINKING_ONLY /* FSE_MIN_TABLELOG */
@@ -210,11 +214,9 @@ size_t FSE_readNCount_bmi2(
short* normalizedCounter, unsigned* maxSVPtr, unsigned* tableLogPtr,
const void* headerBuffer, size_t hbSize, int bmi2)
{
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
return FSE_readNCount_body_bmi2(normalizedCounter, maxSVPtr, tableLogPtr, headerBuffer, hbSize);
}
-#endif
(void)bmi2;
return FSE_readNCount_body_default(normalizedCounter, maxSVPtr, tableLogPtr, headerBuffer, hbSize);
}
@@ -335,11 +337,9 @@ size_t HUF_readStats_wksp(BYTE* huffWeight, size_t hwSize, U32* rankStats,
void* workSpace, size_t wkspSize,
int flags)
{
-#if DYNAMIC_BMI2
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
return HUF_readStats_body_bmi2(huffWeight, hwSize, rankStats, nbSymbolsPtr, tableLogPtr, src, srcSize, workSpace, wkspSize);
}
-#endif
(void)flags;
return HUF_readStats_body_default(huffWeight, hwSize, rankStats, nbSymbolsPtr, tableLogPtr, src, srcSize, workSpace, wkspSize);
}
diff --git a/lib/zstd/common/fse_decompress.c b/lib/zstd/common/fse_decompress.c
--- a/lib/zstd/common/fse_decompress.c
+++ b/lib/zstd/common/fse_decompress.c
@@ -24,6 +24,9 @@
#include "fse.h"
#include "error_private.h"
#include "zstd_deps.h" /* ZSTD_memcpy */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "bits.h" /* ZSTD_highbit32 */
@@ -306,11 +309,9 @@ BMI2_TARGET_ATTRIBUTE static size_t FSE_decompress_wksp_body_bmi2(void* dst, siz
size_t FSE_decompress_wksp_bmi2(void* dst, size_t dstCapacity, const void* cSrc, size_t cSrcSize, unsigned maxLog, void* workSpace, size_t wkspSize, int bmi2)
{
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
return FSE_decompress_wksp_body_bmi2(dst, dstCapacity, cSrc, cSrcSize, maxLog, workSpace, wkspSize);
}
-#endif
(void)bmi2;
return FSE_decompress_wksp_body_default(dst, dstCapacity, cSrc, cSrcSize, maxLog, workSpace, wkspSize);
}
diff --git a/lib/zstd/common/zstd_deps.h b/lib/zstd/common/zstd_deps.h
--- a/lib/zstd/common/zstd_deps.h
+++ b/lib/zstd/common/zstd_deps.h
@@ -26,6 +26,11 @@
#ifndef ZSTD_DEPS_COMMON
#define ZSTD_DEPS_COMMON
+#if defined(__KERNEL__) && defined(CONFIG_X86) && \
+ !defined(__DISABLE_EXPORTS)
+#define ZSTD_USE_KERNEL_CPU_FEATURES
+#endif
+
#include <linux/limits.h>
#include <linux/stddef.h>
diff --git a/lib/zstd/compress/huf_compress.c b/lib/zstd/compress/huf_compress.c
--- a/lib/zstd/compress/huf_compress.c
+++ b/lib/zstd/compress/huf_compress.c
@@ -22,6 +22,9 @@
* Includes
****************************************************************/
#include "../common/zstd_deps.h" /* ZSTD_memcpy, ZSTD_memset */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "../common/compiler.h"
#include "../common/bitstream.h"
#include "hist.h"
@@ -1138,9 +1141,10 @@ HUF_compress1X_usingCTable_internal(void* dst, size_t dstSize,
const void* src, size_t srcSize,
const HUF_CElt* CTable, const int flags)
{
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
return HUF_compress1X_usingCTable_internal_bmi2(dst, dstSize, src, srcSize, CTable);
}
+ (void)flags;
return HUF_compress1X_usingCTable_internal_default(dst, dstSize, src, srcSize, CTable);
}
diff --git a/lib/zstd/compress/zstd_compress.c b/lib/zstd/compress/zstd_compress.c
--- a/lib/zstd/compress/zstd_compress.c
+++ b/lib/zstd/compress/zstd_compress.c
@@ -102,7 +102,7 @@ static void ZSTD_initCCtx(ZSTD_CCtx* cctx, ZSTD_customMem memManager)
assert(cctx != NULL);
ZSTD_memset(cctx, 0, sizeof(*cctx));
cctx->customMem = memManager;
- cctx->bmi2 = ZSTD_cpuSupportsBmi2();
+ ZSTD_SET_BMI2(cctx->bmi2, ZSTD_cpuSupportsBmi2());
{ size_t const err = ZSTD_CCtx_reset(cctx, ZSTD_reset_parameters);
assert(!ZSTD_isError(err));
(void)err;
@@ -142,7 +142,7 @@ ZSTD_CCtx* ZSTD_initStaticCCtx(void* workspace, size_t workspaceSize)
cctx->blockState.nextCBlock = (ZSTD_compressedBlockState_t*)ZSTD_cwksp_reserve_object(&cctx->workspace, sizeof(ZSTD_compressedBlockState_t));
cctx->tmpWorkspace = ZSTD_cwksp_reserve_object(&cctx->workspace, TMP_WORKSPACE_SIZE);
cctx->tmpWkspSize = TMP_WORKSPACE_SIZE;
- cctx->bmi2 = ZSTD_cpuid_bmi2(ZSTD_cpuid());
+ ZSTD_SET_BMI2(cctx->bmi2, ZSTD_cpuid_bmi2(ZSTD_cpuid()));
return cctx;
}
@@ -4042,7 +4042,7 @@ ZSTD_compressSeqStore_singleBlock(ZSTD_CCtx* zc,
op + ZSTD_blockHeaderSize, dstCapacity - ZSTD_blockHeaderSize,
srcSize,
zc->tmpWorkspace, zc->tmpWkspSize /* statically allocated in resetCCtx */,
- zc->bmi2);
+ ZSTD_CCtx_get_bmi2(zc));
FORWARD_IF_ERROR(cSeqsSize, "ZSTD_entropyCompressSeqStore failed!");
if (!zc->isFirstBlock &&
@@ -4332,7 +4332,7 @@ ZSTD_compressBlock_internal(ZSTD_CCtx* zc,
dst, dstCapacity,
srcSize,
zc->tmpWorkspace, zc->tmpWkspSize /* statically allocated in resetCCtx */,
- zc->bmi2);
+ ZSTD_CCtx_get_bmi2(zc));
if (frame &&
/* We don't want to emit our first block as a RLE even if it qualifies because
@@ -6796,7 +6796,7 @@ ZSTD_compressSequences_internal(ZSTD_CCtx* cctx,
op + ZSTD_blockHeaderSize /* Leave space for block header */, dstCapacity - ZSTD_blockHeaderSize,
blockSize,
cctx->tmpWorkspace, cctx->tmpWkspSize /* statically allocated in resetCCtx */,
- cctx->bmi2);
+ ZSTD_CCtx_get_bmi2(cctx));
FORWARD_IF_ERROR(compressedSeqsSize, "Compressing sequences of block failed");
DEBUGLOG(5, "Compressed sequences size: %zu", compressedSeqsSize);
@@ -7321,7 +7321,7 @@ ZSTD_compressSequencesAndLiterals_internal(ZSTD_CCtx* cctx,
&cctx->blockState.prevCBlock->entropy, &cctx->blockState.nextCBlock->entropy,
&cctx->appliedParams,
cctx->tmpWorkspace, cctx->tmpWkspSize /* statically allocated in resetCCtx */,
- cctx->bmi2);
+ ZSTD_CCtx_get_bmi2(cctx));
FORWARD_IF_ERROR(compressedSeqsSize, "Compressing sequences of block failed");
/* note: the spec forbids for any compressed block to be larger than maximum block size */
if (compressedSeqsSize > cctx->blockSizeMax) compressedSeqsSize = 0;
diff --git a/lib/zstd/compress/zstd_compress_internal.h b/lib/zstd/compress/zstd_compress_internal.h
--- a/lib/zstd/compress/zstd_compress_internal.h
+++ b/lib/zstd/compress/zstd_compress_internal.h
@@ -537,6 +537,15 @@ struct ZSTD_CCtx_s {
size_t extSeqBufCapacity;
};
+MEM_STATIC int ZSTD_CCtx_get_bmi2(const struct ZSTD_CCtx_s *cctx) {
+#if DYNAMIC_BMI2 && !defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+ return cctx->bmi2;
+#else
+ (void)cctx;
+ return 0;
+#endif
+}
+
typedef enum { ZSTD_dtlm_fast, ZSTD_dtlm_full } ZSTD_dictTableLoadMethod_e;
typedef enum { ZSTD_tfp_forCCtx, ZSTD_tfp_forCDict } ZSTD_tableFillPurpose_e;
diff --git a/lib/zstd/compress/zstd_compress_sequences.c b/lib/zstd/compress/zstd_compress_sequences.c
--- a/lib/zstd/compress/zstd_compress_sequences.c
+++ b/lib/zstd/compress/zstd_compress_sequences.c
@@ -12,6 +12,10 @@
/*-*************************************
* Dependencies
***************************************/
+#include "../common/zstd_deps.h"
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "zstd_compress_sequences.h"
/*
@@ -429,15 +433,13 @@ size_t ZSTD_encodeSequences(
SeqDef const* sequences, size_t nbSeq, int longOffsets, int bmi2)
{
DEBUGLOG(5, "ZSTD_encodeSequences: dstCapacity = %u", (unsigned)dstCapacity);
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
return ZSTD_encodeSequences_bmi2(dst, dstCapacity,
CTable_MatchLength, mlCodeTable,
CTable_OffsetBits, ofCodeTable,
CTable_LitLength, llCodeTable,
sequences, nbSeq, longOffsets);
}
-#endif
(void)bmi2;
return ZSTD_encodeSequences_default(dst, dstCapacity,
CTable_MatchLength, mlCodeTable,
diff --git a/lib/zstd/compress/zstd_compress_superblock.c b/lib/zstd/compress/zstd_compress_superblock.c
--- a/lib/zstd/compress/zstd_compress_superblock.c
+++ b/lib/zstd/compress/zstd_compress_superblock.c
@@ -684,6 +684,6 @@ size_t ZSTD_compressSuperBlock(ZSTD_CCtx* zc,
&zc->appliedParams,
dst, dstCapacity,
src, srcSize,
- zc->bmi2, lastBlock,
+ ZSTD_CCtx_get_bmi2(zc), lastBlock,
zc->tmpWorkspace, zc->tmpWkspSize /* statically allocated in resetCCtx */);
}
diff --git a/lib/zstd/decompress/huf_decompress.c b/lib/zstd/decompress/huf_decompress.c
--- a/lib/zstd/decompress/huf_decompress.c
+++ b/lib/zstd/decompress/huf_decompress.c
@@ -17,6 +17,9 @@
* Dependencies
****************************************************************/
#include "../common/zstd_deps.h" /* ZSTD_memcpy, ZSTD_memset */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "../common/compiler.h"
#include "../common/bitstream.h" /* BIT_* */
#include "../common/fse.h" /* to compress headers */
@@ -113,9 +116,10 @@ typedef size_t (*HUF_DecompressUsingDTableFn)(void *dst, size_t dstSize,
static size_t fn(void* dst, size_t dstSize, void const* cSrc, \
size_t cSrcSize, HUF_DTable const* DTable, int flags) \
{ \
- if (flags & HUF_flags_bmi2) { \
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) { \
return fn##_bmi2(dst, dstSize, cSrc, cSrcSize, DTable); \
} \
+ (void)flags; \
return fn##_default(dst, dstSize, cSrc, cSrcSize, DTable); \
}
@@ -899,18 +903,16 @@ static size_t HUF_decompress4X1_usingDTable_internal(void* dst, size_t dstSize,
HUF_DecompressUsingDTableFn fallbackFn = HUF_decompress4X1_usingDTable_internal_default;
HUF_DecompressFastLoopFn loopFn = HUF_decompress4X1_usingDTable_internal_fast_c_loop;
-#if DYNAMIC_BMI2
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
fallbackFn = HUF_decompress4X1_usingDTable_internal_bmi2;
# if ZSTD_ENABLE_ASM_X86_64_BMI2
if (!(flags & HUF_flags_disableAsm)) {
loopFn = HUF_decompress4X1_usingDTable_internal_fast_asm_loop;
}
# endif
- } else {
+ } else if (DYNAMIC_BMI2) {
return fallbackFn(dst, dstSize, cSrc, cSrcSize, DTable);
}
-#endif
#if ZSTD_ENABLE_ASM_X86_64_BMI2 && defined(__BMI2__)
if (!(flags & HUF_flags_disableAsm)) {
@@ -1723,18 +1725,16 @@ static size_t HUF_decompress4X2_usingDTable_internal(void* dst, size_t dstSize,
HUF_DecompressUsingDTableFn fallbackFn = HUF_decompress4X2_usingDTable_internal_default;
HUF_DecompressFastLoopFn loopFn = HUF_decompress4X2_usingDTable_internal_fast_c_loop;
-#if DYNAMIC_BMI2
- if (flags & HUF_flags_bmi2) {
+ if (ZSTD_USE_BMI2(flags & HUF_flags_bmi2)) {
fallbackFn = HUF_decompress4X2_usingDTable_internal_bmi2;
# if ZSTD_ENABLE_ASM_X86_64_BMI2
if (!(flags & HUF_flags_disableAsm)) {
loopFn = HUF_decompress4X2_usingDTable_internal_fast_asm_loop;
}
# endif
- } else {
+ } else if (DYNAMIC_BMI2) {
return fallbackFn(dst, dstSize, cSrc, cSrcSize, DTable);
}
-#endif
#if ZSTD_ENABLE_ASM_X86_64_BMI2 && defined(__BMI2__)
if (!(flags & HUF_flags_disableAsm)) {
diff --git a/lib/zstd/decompress/zstd_decompress.c b/lib/zstd/decompress/zstd_decompress.c
--- a/lib/zstd/decompress/zstd_decompress.c
+++ b/lib/zstd/decompress/zstd_decompress.c
@@ -259,9 +259,7 @@ static void ZSTD_initDCtx_internal(ZSTD_DCtx* dctx)
dctx->noForwardProgress = 0;
dctx->oversizedDuration = 0;
dctx->isFrameDecompression = 1;
-#if DYNAMIC_BMI2
- dctx->bmi2 = ZSTD_cpuSupportsBmi2();
-#endif
+ ZSTD_SET_BMI2(dctx->bmi2, ZSTD_cpuSupportsBmi2());
dctx->ddictSet = NULL;
ZSTD_DCtx_resetParameters(dctx);
#ifdef FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION
diff --git a/lib/zstd/decompress/zstd_decompress_block.c b/lib/zstd/decompress/zstd_decompress_block.c
--- a/lib/zstd/decompress/zstd_decompress_block.c
+++ b/lib/zstd/decompress/zstd_decompress_block.c
@@ -16,6 +16,9 @@
* Dependencies
*********************************************************/
#include "../common/zstd_deps.h" /* ZSTD_memcpy, ZSTD_memmove, ZSTD_memset */
+#if defined(ZSTD_USE_KERNEL_CPU_FEATURES)
+#include <asm/cpufeature.h>
+#endif
#include "../common/compiler.h" /* prefetch */
#include "../common/cpu.h" /* bmi2 */
#include "../common/mem.h" /* low level memory routines */
@@ -631,13 +634,11 @@ void ZSTD_buildFSETable(ZSTD_seqSymbol* dt,
const U32* baseValue, const U8* nbAdditionalBits,
unsigned tableLog, void* wksp, size_t wkspSize, int bmi2)
{
-#if DYNAMIC_BMI2
- if (bmi2) {
+ if (ZSTD_USE_BMI2(bmi2)) {
ZSTD_buildFSETable_body_bmi2(dt, normalizedCounter, maxSymbolValue,
baseValue, nbAdditionalBits, tableLog, wksp, wkspSize);
return;
}
-#endif
(void)bmi2;
ZSTD_buildFSETable_body_default(dt, normalizedCounter, maxSymbolValue,
baseValue, nbAdditionalBits, tableLog, wksp, wkspSize);
@@ -1955,11 +1956,9 @@ ZSTD_decompressSequences(ZSTD_DCtx* dctx, void* dst, size_t maxDstSize,
const ZSTD_longOffset_e isLongOffset)
{
DEBUGLOG(5, "ZSTD_decompressSequences");
-#if DYNAMIC_BMI2
- if (ZSTD_DCtx_get_bmi2(dctx)) {
+ if (ZSTD_USE_BMI2(ZSTD_DCtx_get_bmi2(dctx))) {
return ZSTD_decompressSequences_bmi2(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
-#endif
return ZSTD_decompressSequences_default(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
static size_t
@@ -1968,11 +1967,9 @@ ZSTD_decompressSequencesSplitLitBuffer(ZSTD_DCtx* dctx, void* dst, size_t maxDst
const ZSTD_longOffset_e isLongOffset)
{
DEBUGLOG(5, "ZSTD_decompressSequencesSplitLitBuffer");
-#if DYNAMIC_BMI2
- if (ZSTD_DCtx_get_bmi2(dctx)) {
+ if (ZSTD_USE_BMI2(ZSTD_DCtx_get_bmi2(dctx))) {
return ZSTD_decompressSequencesSplitLitBuffer_bmi2(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
-#endif
return ZSTD_decompressSequencesSplitLitBuffer_default(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
#endif /* ZSTD_FORCE_DECOMPRESS_SEQUENCES_LONG */
@@ -1991,11 +1988,9 @@ ZSTD_decompressSequencesLong(ZSTD_DCtx* dctx,
const ZSTD_longOffset_e isLongOffset)
{
DEBUGLOG(5, "ZSTD_decompressSequencesLong");
-#if DYNAMIC_BMI2
- if (ZSTD_DCtx_get_bmi2(dctx)) {
+ if (ZSTD_USE_BMI2(ZSTD_DCtx_get_bmi2(dctx))) {
return ZSTD_decompressSequencesLong_bmi2(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
-#endif
return ZSTD_decompressSequencesLong_default(dctx, dst, maxDstSize, seqStart, seqSize, nbSeq, isLongOffset);
}
#endif /* ZSTD_FORCE_DECOMPRESS_SEQUENCES_SHORT */
diff --git a/lib/zstd/decompress/zstd_decompress_internal.h b/lib/zstd/decompress/zstd_decompress_internal.h
--- a/lib/zstd/decompress/zstd_decompress_internal.h
+++ b/lib/zstd/decompress/zstd_decompress_internal.h
@@ -204,7 +204,7 @@ struct ZSTD_DCtx_s
}; /* typedef'd to ZSTD_DCtx within "zstd.h" */
MEM_STATIC int ZSTD_DCtx_get_bmi2(const struct ZSTD_DCtx_s *dctx) {
-#if DYNAMIC_BMI2
+#if DYNAMIC_BMI2 && !defined(ZSTD_USE_KERNEL_CPU_FEATURES)
return dctx->bmi2;
#else
(void)dctx;
@@ -0,0 +1,44 @@
diff --git a/crypto/zstd.c b/crypto/zstd.c
--- a/crypto/zstd.c
+++ b/crypto/zstd.c
@@ -96,6 +96,7 @@ static int zstd_compress_one(struct acomp_req *req, struct zstd_ctx *ctx,
static int zstd_compress(struct acomp_req *req)
{
+ bool stream_initialized = false;
struct crypto_acomp_stream *s;
unsigned int pos, scur, dcur;
unsigned int total_out = 0;
@@ -115,12 +116,6 @@ static int zstd_compress(struct acomp_req *req)
if (ret)
goto out;
- ctx->cctx = zstd_init_cstream(&ctx->params, 0, ctx->wksp, ctx->wksp_size);
- if (!ctx->cctx) {
- ret = -EINVAL;
- goto out;
- }
-
do {
dcur = acomp_walk_next_dst(&walk);
if (!dcur) {
@@ -142,6 +137,19 @@ static int zstd_compress(struct acomp_req *req)
goto out;
}
+ if (!stream_initialized) {
+ ctx->cctx = zstd_init_cstream(&ctx->params, 0,
+ ctx->wksp, ctx->wksp_size);
+ if (!ctx->cctx) {
+ /* Release in the reverse of the map order. */
+ acomp_walk_done_src(&walk, 0);
+ acomp_walk_done_dst(&walk, 0);
+ ret = -EINVAL;
+ goto out;
+ }
+ stream_initialized = true;
+ }
+
if (scur) {
inbuf.pos = 0;
inbuf.src = walk.src.virt.addr;
@@ -0,0 +1,44 @@
diff --git a/crypto/zstd.c b/crypto/zstd.c
--- a/crypto/zstd.c
+++ b/crypto/zstd.c
@@ -215,6 +215,7 @@ static int zstd_decompress_one(struct acomp_req *req, struct zstd_ctx *ctx,
static int zstd_decompress(struct acomp_req *req)
{
+ bool stream_initialized = false;
struct crypto_acomp_stream *s;
unsigned int total_out = 0;
unsigned int scur, dcur;
@@ -232,12 +233,6 @@ static int zstd_decompress(struct acomp_req *req)
if (ret)
goto out;
- ctx->dctx = zstd_init_dstream(ZSTD_MAX_SIZE, ctx->wksp, ctx->wksp_size);
- if (!ctx->dctx) {
- ret = -EINVAL;
- goto out;
- }
-
do {
scur = acomp_walk_next_src(&walk);
if (scur) {
@@ -263,6 +258,19 @@ static int zstd_decompress(struct acomp_req *req)
goto out;
}
+ if (!stream_initialized) {
+ ctx->dctx = zstd_init_dstream(ZSTD_MAX_SIZE, ctx->wksp,
+ ctx->wksp_size);
+ if (!ctx->dctx) {
+ /* Release in the reverse of the map order. */
+ acomp_walk_done_dst(&walk, 0);
+ acomp_walk_done_src(&walk, 0);
+ ret = -EINVAL;
+ goto out;
+ }
+ stream_initialized = true;
+ }
+
outbuf.pos = 0;
outbuf.dst = (u8 *)walk.dst.virt.addr;
outbuf.size = dcur;
@@ -0,0 +1,363 @@
--- a/crypto/af_alg.c
+++ b/crypto/af_alg.c
@@ -8,6 +8,7 @@
*/
#include <linux/atomic.h>
+#include <linux/capability.h>
#include <crypto/if_alg.h>
#include <linux/crypto.h>
#include <linux/init.h>
@@ -22,10 +23,28 @@
#include <linux/sched/signal.h>
#include <linux/security.h>
#include <linux/string.h>
+#include <linux/sysctl.h>
+#include <linux/user_namespace.h>
#include <keys/user-type.h>
#include <keys/trusted-type.h>
#include <keys/encrypted-type.h>
+static int af_alg_restrict = 1;
+
+static const struct ctl_table af_alg_table[] = {
+ {
+ .procname = "af_alg_restrict",
+ .data = &af_alg_restrict,
+ .maxlen = sizeof(int),
+ .mode = 0644,
+ .proc_handler = proc_dointvec_minmax,
+ .extra1 = SYSCTL_ZERO,
+ .extra2 = SYSCTL_TWO,
+ },
+};
+
+static struct ctl_table_header *af_alg_header;
+
struct alg_type_list {
const struct af_alg_type *type;
struct list_head list;
@@ -110,6 +129,43 @@
}
EXPORT_SYMBOL_GPL(af_alg_unregister_type);
+static bool af_alg_capable(void)
+{
+ return ns_capable_noaudit(&init_user_ns, CAP_NET_ADMIN) ||
+ capable(CAP_SYS_ADMIN);
+}
+
+int af_alg_check_restriction(const char *name,
+ const struct af_alg_allowlist_entry allowlist[])
+{
+ int level = READ_ONCE(af_alg_restrict);
+
+ if (level == 0)
+ return 0;
+ if (level == 1) {
+ for (const struct af_alg_allowlist_entry *ent = allowlist;
+ ent->name; ent++) {
+ if (strcmp(name, ent->name) == 0) {
+ if ((ent->flags & AF_ALG_UNPRIVILEGED) ||
+ af_alg_capable())
+ return 0;
+ /* List contains at most one entry per name. */
+ break;
+ }
+ }
+ }
+ /*
+ * Use -ENOENT (the error code for "algorithm not found") instead of
+ * -EACCES or -EPERM, for the highest chance of correctly triggering
+ * fallback code paths in userspace programs.
+ *
+ * Don't log a warning, since it would be noisy. iwd tries to bind a
+ * bunch of algorithms that it never uses.
+ */
+ return -ENOENT;
+}
+EXPORT_SYMBOL_GPL(af_alg_check_restriction);
+
static void alg_do_release(const struct af_alg_type *type, void *private)
{
if (!type)
@@ -506,6 +562,9 @@
struct sock *sk;
int err;
+ if (READ_ONCE(af_alg_restrict) == 2)
+ return -EAFNOSUPPORT;
+
if (sock->type != SOCK_SEQPACKET)
return -ESOCKTNOSUPPORT;
if (protocol != 0)
@@ -1222,27 +1281,32 @@
static int __init af_alg_init(void)
{
- int err = proto_register(&alg_proto, 0);
+ int err;
+
+ af_alg_header = register_sysctl("crypto", af_alg_table);
+ err = proto_register(&alg_proto, 0);
if (err)
- goto out;
+ goto out_unregister_sysctl;
err = sock_register(&alg_family);
- if (err != 0)
+ if (err)
goto out_unregister_proto;
-out:
- return err;
+ return 0;
out_unregister_proto:
proto_unregister(&alg_proto);
- goto out;
+out_unregister_sysctl:
+ unregister_sysctl_table(af_alg_header);
+ return err;
}
static void __exit af_alg_exit(void)
{
sock_unregister(PF_ALG);
proto_unregister(&alg_proto);
+ unregister_sysctl_table(af_alg_header);
}
module_init(af_alg_init);
--- a/crypto/algif_aead.c
+++ b/crypto/algif_aead.c
@@ -34,6 +34,11 @@
#include <linux/net.h>
#include <net/sock.h>
+static const struct af_alg_allowlist_entry aead_allowlist[] = {
+ { "ccm(aes)" }, /* bluez */
+ {},
+};
+
static inline bool aead_sufficient_data(struct sock *sk)
{
struct alg_sock *ask = alg_sk(sk);
@@ -344,6 +349,12 @@
static void *aead_bind(const char *name)
{
+ int err;
+
+ err = af_alg_check_restriction(name, aead_allowlist);
+ if (err)
+ return ERR_PTR(err);
+
return crypto_alloc_aead(name, 0, AF_ALG_CRYPTOAPI_MASK);
}
--- a/crypto/algif_hash.c
+++ b/crypto/algif_hash.c
@@ -16,6 +16,24 @@
#include <linux/net.h>
#include <net/sock.h>
+static const struct af_alg_allowlist_entry hash_allowlist[] = {
+ { "cmac(aes)" }, /* iwd, bluez */
+ { "hmac(md5)" }, /* iwd */
+ { "hmac(sha1)" }, /* iwd */
+ { "hmac(sha224)" }, /* iwd */
+ { "hmac(sha256)" }, /* iwd */
+ { "hmac(sha384)" }, /* iwd */
+ { "hmac(sha512)" }, /* iwd, sha512hmac */
+ { "md4" }, /* iwd */
+ { "md5" }, /* iwd */
+ { "sha1", AF_ALG_UNPRIVILEGED }, /* iwd, iproute2 < 7.0 */
+ { "sha224" }, /* iwd */
+ { "sha256" }, /* iwd */
+ { "sha384" }, /* iwd */
+ { "sha512" }, /* iwd */
+ {},
+};
+
struct hash_ctx {
struct af_alg_sgl sgl;
@@ -382,6 +400,12 @@
static void *hash_bind(const char *name)
{
+ int err;
+
+ err = af_alg_check_restriction(name, hash_allowlist);
+ if (err)
+ return ERR_PTR(err);
+
return crypto_alloc_ahash(name, 0, AF_ALG_CRYPTOAPI_MASK);
}
--- a/crypto/algif_rng.c
+++ b/crypto/algif_rng.c
@@ -50,6 +50,10 @@
MODULE_AUTHOR("Stephan Mueller <smueller@chronox.de>");
MODULE_DESCRIPTION("User-space interface for random number generators");
+static const struct af_alg_allowlist_entry rng_allowlist[] = {
+ {},
+};
+
struct rng_ctx {
#define MAXSIZE 128
unsigned int len;
@@ -201,6 +205,11 @@
{
struct rng_parent_ctx *pctx;
struct crypto_rng *rng;
+ int err;
+
+ err = af_alg_check_restriction(name, rng_allowlist);
+ if (err)
+ return ERR_PTR(err);
pctx = kzalloc_obj(*pctx);
if (!pctx)
--- a/crypto/algif_skcipher.c
+++ b/crypto/algif_skcipher.c
@@ -35,6 +35,24 @@
#include <linux/string.h>
#include <net/sock.h>
+static const struct af_alg_allowlist_entry skcipher_allowlist[] = {
+ { "adiantum(xchacha12,aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "adiantum(xchacha20,aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "cbc(aes)" }, /* iwd */
+ { "cbc(des)" }, /* iwd */
+ { "cbc(des3_ede)" }, /* iwd */
+ { "cbc(paes)" }, /* caam and others */
+ { "ctr(aes)" }, /* iwd */
+ { "ecb(aes)" }, /* iwd, bluez */
+ { "ecb(des)" }, /* iwd */
+ { "hctr2(aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "xts(aes)", AF_ALG_UNPRIVILEGED }, /* cryptsetup benchmark */
+ { "xts(camellia)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "xts(serpent)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ { "xts(twofish)", AF_ALG_UNPRIVILEGED }, /* cryptsetup */
+ {},
+};
+
static int skcipher_sendmsg(struct socket *sock, struct msghdr *msg,
size_t size)
{
@@ -311,6 +329,11 @@
static void *skcipher_bind(const char *name)
{
u32 mask = AF_ALG_CRYPTOAPI_MASK;
+ int err;
+
+ err = af_alg_check_restriction(name, skcipher_allowlist);
+ if (err)
+ return ERR_PTR(err);
if (strcmp(name, "cbc(paes)") == 0)
mask = 0;
--- a/include/crypto/if_alg.h
+++ b/include/crypto/if_alg.h
@@ -8,6 +8,7 @@
#ifndef _CRYPTO_IF_ALG_H
#define _CRYPTO_IF_ALG_H
+#include <linux/bits.h>
#include <linux/compiler.h>
#include <linux/completion.h>
#include <linux/if_alg.h>
@@ -121,7 +122,7 @@
* @iv: IV for cipher operation
* @state: Existing state for continuing operation
* @aead_assoclen: Length of AAD for AEAD cipher operations
- * @completion: Work queue for synchronous operation
+ * @wait: For waiting for completion of async crypto ops
* @used: TX bytes sent to kernel. This variable is used to
* ensure that user space cannot cause the kernel
* to allocate too much memory in sendmsg operation.
@@ -161,9 +162,20 @@
unsigned int inflight;
};
+/* Flags for af_alg_allowlist_entry::flags: */
+#define AF_ALG_UNPRIVILEGED BIT(0) /* Unprivileged use is allowed */
+
+struct af_alg_allowlist_entry {
+ const char *name;
+ u32 flags;
+};
+
int af_alg_register_type(const struct af_alg_type *type);
int af_alg_unregister_type(const struct af_alg_type *type);
+int af_alg_check_restriction(const char *name,
+ const struct af_alg_allowlist_entry allowlist[]);
+
int af_alg_release(struct socket *sock);
void af_alg_release_parent(struct sock *sk);
int af_alg_accept(struct sock *sk, struct socket *newsock,
@@ -177,10 +189,11 @@
}
/**
- * Size of available buffer for sending data from user space to kernel.
+ * af_alg_sndbuf - Size of available buffer for sending data from user space to kernel.
*
- * @sk socket of connection to user space
- * @return number of bytes still available
+ * @sk: socket of connection to user space
+ *
+ * Returns: number of bytes still available
*/
static inline int af_alg_sndbuf(struct sock *sk)
{
@@ -192,10 +205,11 @@
}
/**
- * Can the send buffer still be written to?
+ * af_alg_writable - Can the send buffer still be written to?
+ *
+ * @sk: socket of connection to user space
*
- * @sk socket of connection to user space
- * @return true => writable, false => not writable
+ * Returns: true => writable, false => not writable
*/
static inline bool af_alg_writable(struct sock *sk)
{
@@ -203,10 +217,11 @@
}
/**
- * Size of available buffer used by kernel for the RX user space operation.
+ * af_alg_rcvbuf - Size of available buffer used by kernel for the RX user space operation.
*
- * @sk socket of connection to user space
- * @return number of bytes still available
+ * @sk: socket of connection to user space
+ *
+ * Returns: number of bytes still available
*/
static inline int af_alg_rcvbuf(struct sock *sk)
{
@@ -218,10 +233,11 @@
}
/**
- * Can the RX buffer still be written to?
+ * af_alg_readable - Can the RX buffer still be read from?
+ *
+ * @sk: socket of connection to user space
*
- * @sk socket of connection to user space
- * @return true => writable, false => not writable
+ * Returns: true => readable, false => not readable
*/
static inline bool af_alg_readable(struct sock *sk)
{
File diff suppressed because it is too large. Load diff
Binary file not shown.
@@ -0,0 +1,92 @@
diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c
--- a/fs/btrfs/dev-replace.c
+++ b/fs/btrfs/dev-replace.c
@@ -627,7 +627,7 @@ static int btrfs_dev_replace_start(struct btrfs_fs_info *fs_info,
ret = mark_block_group_to_copy(fs_info, src_device);
if (ret)
- return ret;
+ goto leave;
down_write(&dev_replace->rwsem);
dev_replace->replace_task = current;
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -384,6 +384,7 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags &= ~BTRFS_INODE_COMPRESS;
inode_flags |= BTRFS_INODE_NOCOMPRESS;
} else if (fsflags & FS_COMPR_FL) {
+ enum btrfs_compression_type comp_type;
if (IS_SWAPFILE(&inode->vfs_inode))
return -ETXTBSY;
@@ -391,9 +392,23 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags |= BTRFS_INODE_COMPRESS;
inode_flags &= ~BTRFS_INODE_NOCOMPRESS;
- comp = btrfs_compress_type2str(fs_info->compress_type);
- if (!comp || comp[0] == 0)
- comp = btrfs_compress_type2str(BTRFS_COMPRESS_ZLIB);
+ /*
+ * Keep the algorithm recorded in the compression property,
+ * otherwise changing an unrelated attribute would reset it to
+ * the mount default, since FS_IOC_SETFLAGS callers write back
+ * the whole flag set they got from FS_IOC_GETFLAGS and that
+ * includes FS_COMPR_FL for any inode carrying the property.
+ *
+ * Inodes with the compress flag set but no property keep using
+ * the mount default, so they behave as before.
+ */
+ if (inode->prop_compress)
+ comp_type = inode->prop_compress;
+ else if (fs_info->compress_type)
+ comp_type = fs_info->compress_type;
+ else
+ comp_type = BTRFS_COMPRESS_ZLIB;
+ comp = btrfs_compress_type2str(comp_type);
} else {
inode_flags &= ~(BTRFS_INODE_COMPRESS | BTRFS_INODE_NOCOMPRESS);
}
diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c
--- a/fs/btrfs/send.c
+++ b/fs/btrfs/send.c
@@ -6417,6 +6417,13 @@ static int process_extent(struct send_ctx *sctx,
if (S_ISLNK(sctx->cur_inode_mode))
return 0;
+ if (unlikely(!S_ISREG(sctx->cur_inode_mode))) {
+ btrfs_crit(sctx->send_root->fs_info,
+ "send: extent for non-regular inode %llu root %llu mode 0%llo",
+ key->objectid, btrfs_root_id(sctx->send_root),
+ sctx->cur_inode_mode & S_IFMT);
+ return -EUCLEAN;
+ }
if (sctx->parent_root && !sctx->cur_inode_new) {
ret = is_extent_unchanged(sctx, path, key);
diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c
--- a/fs/btrfs/zoned.c
+++ b/fs/btrfs/zoned.c
@@ -2710,6 +2710,7 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng
{
struct btrfs_block_group *block_group;
u64 min_alloc_bytes;
+ int ret = 0;
if (!btrfs_is_zoned(fs_info))
return 0;
@@ -2729,11 +2730,11 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng
block_group->start + block_group->zone_capacity)
goto out;
- do_zone_finish(block_group, true);
+ ret = do_zone_finish(block_group, true);
out:
btrfs_put_block_group(block_group);
- return 0;
+ return ret;
}
static void btrfs_zone_finish_endio_workfn(struct work_struct *work)
@@ -0,0 +1,109 @@
diff --git c/fs/btrfs/zstd.c i/fs/btrfs/zstd.c
--- c/fs/btrfs/zstd.c
+++ i/fs/btrfs/zstd.c
@@ -589,10 +589,48 @@ int zstd_compress_bio(struct list_head *ws, struct compressed_bio *cb)
return ret;
}
+/*
+ * Map the destination for the next chunk of output.
+ *
+ * @decompressed is the offset of the next output byte inside the fully
+ * decompressed extent. If that offset has reached the current destination
+ * segment, its page-bounded bio_vec is kmapped so that zstd can write into the
+ * page cache directly, and the number of bytes writable there is returned.
+ * Otherwise @kaddr_ret is set to NULL and the number of bytes to skip before
+ * that segment is returned. This covers both the initial prefix and gaps in
+ * the destination bio.
+ */
+static u32 zstd_map_dest(struct compressed_bio *cb, u32 decompressed,
+ void **kaddr_ret)
+{
+ struct bio *orig_bio = &cb->orig_bbio->bio;
+ struct bio_vec bvec;
+ u32 bvec_offset;
+ u32 off;
+
+ bvec = bio_iter_iovec(orig_bio, orig_bio->bi_iter);
+ /*
+ * cb->start may underflow, but subtracting that value can still give us
+ * the correct offset inside the full decompressed extent.
+ */
+ bvec_offset = page_offset(bvec.bv_page) + bvec.bv_offset - cb->start;
+
+ if (decompressed < bvec_offset) {
+ *kaddr_ret = NULL;
+ return bvec_offset - decompressed;
+ }
+
+ off = decompressed - bvec_offset;
+ ASSERT(off < bvec.bv_len);
+ *kaddr_ret = bvec_kmap_local(&bvec) + off;
+ return bvec.bv_len - off;
+}
+
int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
{
struct btrfs_fs_info *fs_info = cb_to_fs_info(cb);
struct workspace *workspace = list_entry(ws, struct workspace, list);
+ struct bio *orig_bio = &cb->orig_bbio->bio;
struct folio_iter fi;
size_t srclen = bio_get_size(&cb->bbio.bio);
zstd_dstream *stream;
@@ -600,7 +638,6 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
const unsigned int min_folio_size = btrfs_min_folio_size(fs_info);
unsigned long folio_in_index = 0;
unsigned long total_folios_in = DIV_ROUND_UP(srclen, min_folio_size);
- unsigned long buf_start;
unsigned long total_out = 0;
bio_first_folio(&fi, &cb->bbio.bio, 0);
@@ -624,15 +661,26 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
workspace->in_buf.pos = 0;
workspace->in_buf.size = min_t(size_t, srclen, min_folio_size);
- workspace->out_buf.dst = workspace->buf;
- workspace->out_buf.pos = 0;
- workspace->out_buf.size = fs_info->sectorsize;
-
- while (1) {
+ while (orig_bio->bi_iter.bi_size) {
size_t ret2;
+ void *kaddr;
+ u32 dstlen;
+
+ dstlen = zstd_map_dest(cb, total_out, &kaddr);
+ if (kaddr) {
+ workspace->out_buf.dst = kaddr;
+ workspace->out_buf.size = dstlen;
+ } else {
+ workspace->out_buf.dst = workspace->buf;
+ workspace->out_buf.size = min_t(u32, dstlen,
+ fs_info->sectorsize);
+ }
+ workspace->out_buf.pos = 0;
ret2 = zstd_decompress_stream(stream, &workspace->out_buf,
&workspace->in_buf);
+ if (kaddr)
+ kunmap_local(kaddr);
if (unlikely(zstd_is_error(ret2))) {
struct btrfs_inode *inode = cb->bbio.inode;
@@ -643,14 +691,9 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb)
ret = -EIO;
goto done;
}
- buf_start = total_out;
total_out += workspace->out_buf.pos;
- workspace->out_buf.pos = 0;
-
- ret = btrfs_decompress_buf2page(workspace->out_buf.dst,
- total_out - buf_start, cb, buf_start);
- if (ret == 0)
- break;
+ if (kaddr)
+ bio_advance(orig_bio, workspace->out_buf.pos);
if (workspace->in_buf.pos >= srclen)
break;
@@ -0,0 +1,97 @@
diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c
--- a/fs/btrfs/block-group.c
+++ b/fs/btrfs/block-group.c
@@ -3074,6 +3074,18 @@ struct btrfs_block_group *btrfs_make_block_group(struct btrfs_trans_handle *tran
return ERR_PTR(ret);
}
+ /*
+ * Ensure the corresponding space_info object is created and
+ * assigned to our block group. We want our bg to be added to the rbtree
+ * with its ->space_info set.
+ *
+ * On a zoned filesystem btrfs_add_new_free_space() ends up in
+ * __btrfs_add_free_space_zoned(), which dereferences
+ * block_group->space_info, so it has to be set beforehand.
+ */
+ cache->space_info = space_info;
+ ASSERT(cache->space_info);
+
ret = btrfs_add_new_free_space(cache, chunk_offset, chunk_offset + size, NULL);
btrfs_free_excluded_extents(cache);
if (ret) {
@@ -3081,14 +3093,6 @@ struct btrfs_block_group *btrfs_make_block_group(struct btrfs_trans_handle *tran
return ERR_PTR(ret);
}
- /*
- * Ensure the corresponding space_info object is created and
- * assigned to our block group. We want our bg to be added to the rbtree
- * with its ->space_info set.
- */
- cache->space_info = space_info;
- ASSERT(cache->space_info);
-
ret = btrfs_add_block_group_cache(cache);
if (ret) {
btrfs_remove_free_space_cache(cache);
diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c
--- a/fs/btrfs/tree-checker.c
+++ b/fs/btrfs/tree-checker.c
@@ -1909,6 +1909,16 @@ static int check_inode_ref(struct extent_buffer *leaf,
return -EUCLEAN;
}
+ if (unlikely(btrfs_is_fstree(btrfs_header_owner(leaf)) &&
+ (key->offset < BTRFS_FIRST_FREE_OBJECTID ||
+ key->offset > BTRFS_LAST_FREE_OBJECTID))) {
+ inode_ref_err(leaf, slot,
+ "invalid offset for ref key, have %llu expect [%llu, %lld]",
+ key->offset, BTRFS_FIRST_FREE_OBJECTID,
+ BTRFS_LAST_FREE_OBJECTID);
+ return -EUCLEAN;
+ }
+
ptr = btrfs_item_ptr_offset(leaf, slot);
end = ptr + btrfs_item_size(leaf, slot);
while (ptr < end) {
@@ -1952,12 +1962,14 @@ static int check_inode_extref(struct extent_buffer *leaf,
{
unsigned long ptr = btrfs_item_ptr_offset(leaf, slot);
unsigned long end = ptr + btrfs_item_size(leaf, slot);
+ const bool is_fstree = btrfs_is_fstree(btrfs_header_owner(leaf));
if (unlikely(!check_prev_ino(leaf, key, slot, prev_key)))
return -EUCLEAN;
while (ptr < end) {
struct btrfs_inode_extref *extref = (struct btrfs_inode_extref *)ptr;
+ u64 parent;
u16 namelen;
if (unlikely(ptr + sizeof(*extref) > end)) {
@@ -1967,7 +1979,24 @@ static int check_inode_extref(struct extent_buffer *leaf,
return -EUCLEAN;
}
+ parent = btrfs_inode_extref_parent(leaf, extref);
+ if (unlikely(is_fstree && (parent < BTRFS_FIRST_FREE_OBJECTID ||
+ parent > BTRFS_LAST_FREE_OBJECTID))) {
+ inode_ref_err(leaf, slot,
+ "invalid parent for extref key, have %llu expect [%llu, %lld]",
+ parent, BTRFS_FIRST_FREE_OBJECTID,
+ BTRFS_LAST_FREE_OBJECTID);
+ return -EUCLEAN;
+ }
+
namelen = btrfs_inode_extref_name_len(leaf, extref);
+ if (unlikely(namelen == 0 || namelen > BTRFS_NAME_LEN)) {
+ inode_ref_err(leaf, slot,
+ "invalid inode extref name length, has %u expect [1, %u]",
+ namelen, BTRFS_NAME_LEN);
+ return -EUCLEAN;
+ }
+
if (unlikely(ptr + sizeof(*extref) + namelen > end)) {
inode_ref_err(leaf, slot,
"inode extref overflow, ptr %lu end %lu namelen %u",
@@ -0,0 +1,77 @@
diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c
--- a/fs/btrfs/tree-checker.c
+++ b/fs/btrfs/tree-checker.c
@@ -2306,7 +2306,7 @@ static int check_free_space_extent(struct extent_buffer *leaf, struct btrfs_key
if (unlikely(btrfs_item_size(leaf, slot) != 0)) {
generic_err(leaf, slot,
- "invalid item size for free space info, has %u expect 0",
+ "invalid item size for free space extent, has %u expect 0",
btrfs_item_size(leaf, slot));
return -EUCLEAN;
}
diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c
--- a/fs/btrfs/volumes.c
+++ b/fs/btrfs/volumes.c
@@ -740,6 +740,36 @@ const u8 *btrfs_sb_fsid_ptr(const struct btrfs_super_block *sb)
return has_metadata_uuid ? sb->metadata_uuid : sb->fsid;
}
+static bool should_rename_device(const struct btrfs_device *dev)
+{
+ bool ret;
+ const char *old_name;
+
+ rcu_read_lock();
+ old_name = rcu_dereference(dev->name);
+ /*
+ * For systems booted without an initramfs, the rootfs has the device
+ * name "/dev/root".
+ *
+ * Although using btrfs without an initramfs is not recommended (if a
+ * new device is added to the rootfs, the system can no longer boot, as
+ * there is no way to register all devices), there is still a minority
+ * of users doing this.
+ *
+ * And after the system is up, a later device scan on the real block
+ * device file will never get this device's name updated, as the
+ * device->devt is still the same.
+ *
+ * Here we add one and only one exception for "/dev/root", to allow the
+ * device name to be updated even if the new path points to the same
+ * block device.
+ */
+ ret = (strcmp(old_name, "/dev/root") == 0);
+ rcu_read_unlock();
+
+ return ret;
+}
+
/*
* Add new device to list of registered devices
*
@@ -860,7 +890,8 @@ static noinline struct btrfs_device *device_list_add(const char *path,
MAJOR(path_devt), MINOR(path_devt),
current->comm, task_pid_nr(current));
- } else if (!device->name || device->devt != path_devt) {
+ } else if (!device->name || device->devt != path_devt ||
+ should_rename_device(device)) {
const char *old_name;
/*
diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c
--- a/fs/btrfs/zoned.c
+++ b/fs/btrfs/zoned.c
@@ -2688,6 +2688,11 @@ bool btrfs_can_activate_zone(struct btrfs_fs_devices *fs_devices, u64 flags)
switch (flags & BTRFS_BLOCK_GROUP_PROFILE_MASK) {
case 0: /* single */
+ case BTRFS_BLOCK_GROUP_RAID0:
+ case BTRFS_BLOCK_GROUP_RAID1:
+ case BTRFS_BLOCK_GROUP_RAID1C3:
+ case BTRFS_BLOCK_GROUP_RAID1C4:
+ case BTRFS_BLOCK_GROUP_RAID10:
ret = (atomic_read(&zinfo->active_zones_left) >= (1 + reserved));
break;
case BTRFS_BLOCK_GROUP_DUP:
@@ -0,0 +1,42 @@
diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c
--- a/fs/fuse/dir.c
+++ b/fs/fuse/dir.c
@@ -2289,6 +2289,9 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
*/
if ((is_truncate || !is_wb) &&
S_ISREG(inode->i_mode) && oldsize != outarg.attr.size) {
+ if (outarg.attr.size > oldsize)
+ truncate_pagecache_range(inode, oldsize,
+ outarg.attr.size - 1);
truncate_pagecache(inode, outarg.attr.size);
invalidate_inode_pages2(mapping);
}
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -1360,9 +1360,13 @@ static ssize_t fuse_perform_write(struct kiocb *iocb, struct iov_iter *ii)
struct fuse_conn *fc = get_fuse_conn(inode);
struct fuse_inode *fi = get_fuse_inode(inode);
loff_t pos = iocb->ki_pos;
+ loff_t old_size = i_size_read(inode);
int err = 0;
ssize_t res = 0;
+ if (pos > old_size)
+ truncate_pagecache_range(inode, old_size, pos - 1);
+
if (inode->i_size < pos + iov_iter_count(ii))
set_bit(FUSE_I_SIZE_UNSTABLE, &fi->state);
@@ -2909,6 +2913,11 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset,
/* we could have extended the file */
if (!(mode & FALLOC_FL_KEEP_SIZE)) {
+ loff_t oldsize = i_size_read(inode);
+
+ if (offset + length > oldsize)
+ truncate_pagecache_range(inode, oldsize,
+ offset + length - 1);
if (fuse_write_update_attr(inode, offset + length, length))
file_update_time(file);
}
@@ -0,0 +1,104 @@
diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -213,10 +213,13 @@ EXPORT_SYMBOL_GPL(fuse_req_hash);
/*
* A new request is available, wake fiq->waitq
*/
-static void fuse_dev_wake_and_unlock(struct fuse_iqueue *fiq)
+static void fuse_dev_wake_and_unlock(struct fuse_iqueue *fiq, bool sync)
__releases(fiq->lock)
{
- wake_up(&fiq->waitq);
+ if (sync)
+ wake_up_sync(&fiq->waitq);
+ else
+ wake_up(&fiq->waitq);
kill_fasync(&fiq->fasync, SIGIO, POLL_IN);
spin_unlock(&fiq->lock);
}
@@ -233,7 +236,7 @@ void fuse_dev_queue_forget(struct fuse_iqueue *fiq,
if (fiq->connected) {
fiq->forget_list_tail->next = forget;
fiq->forget_list_tail = forget;
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, false);
} else {
kfree(forget);
spin_unlock(&fiq->lock);
@@ -255,7 +258,7 @@ void fuse_dev_queue_interrupt(struct fuse_iqueue *fiq, struct fuse_req *req)
list_del_init(&req->intr_entry);
spin_unlock(&fiq->lock);
} else {
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, false);
}
} else {
spin_unlock(&fiq->lock);
@@ -285,11 +288,13 @@ EXPORT_SYMBOL_GPL(fuse_request_assign_unique);
static void fuse_dev_queue_req(struct fuse_iqueue *fiq, struct fuse_req *req)
{
+ bool sync = test_and_clear_bit(FR_SYNC_WAKEUP, &req->flags);
+
spin_lock(&fiq->lock);
if (fiq->connected) {
fuse_request_assign_unique_locked(fiq, req);
list_add_tail(&req->list, &fiq->pending);
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, sync);
} else {
spin_unlock(&fiq->lock);
req->out.h.error = -ENOTCONN;
@@ -752,6 +757,11 @@ static void __fuse_request_send(struct fuse_req *req)
/* acquire extra reference, since request is still needed after
fuse_request_end() */
__fuse_get_request(req);
+ /*
+ * This is a synchronous request: the caller will block waiting for
+ * the answer. Hint the scheduler via wake_up_sync().
+ */
+ set_bit(FR_SYNC_WAKEUP, &req->flags);
fuse_send_one(fiq, req);
request_wait_answer(req);
@@ -1806,7 +1816,7 @@ void fuse_chan_resend(struct fuse_chan *fch)
}
/* iq and pq requests are both oldest to newest */
list_splice(&to_queue, &fiq->pending);
- fuse_dev_wake_and_unlock(fiq);
+ fuse_dev_wake_and_unlock(fiq, false);
}
/* Look up request on processing list by unique ID */
diff --git a/fs/fuse/fuse_dev_i.h b/fs/fuse/fuse_dev_i.h
--- a/fs/fuse/fuse_dev_i.h
+++ b/fs/fuse/fuse_dev_i.h
@@ -38,6 +38,8 @@ struct fuse_iqueue;
* @FR_PRIVATE: request is on private list
* @FR_ASYNC: request is asynchronous
* @FR_URING: request is handled through fuse-io-uring
+ * @FR_SYNC_WAKEUP: use synchronous wakeup when queueing this request to
+ * give the scheduler a hint about the waker task
*/
enum fuse_req_flag {
FR_ISREPLY,
@@ -53,6 +55,7 @@ enum fuse_req_flag {
FR_PRIVATE,
FR_ASYNC,
FR_URING,
+ FR_SYNC_WAKEUP,
};
/**
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -1420,6 +1420,7 @@ static void process_init_reply(struct fuse_args *args, int error)
fm->sb->s_bdi->ra_pages =
min(fm->sb->s_bdi->ra_pages, ra_pages);
+ fm->sb->s_bdi->io_pages = fc->max_pages;
fc->minor = arg->minor;
fc->max_write = arg->minor < 5 ? 4096 : arg->max_write;
fc->max_write = max_t(unsigned, 4096, fc->max_write);
@@ -0,0 +1,38 @@
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -1219,8 +1219,7 @@ static ssize_t fuse_send_write_pages(struct fuse_io_args *ia,
struct file *file = iocb->ki_filp;
struct fuse_file *ff = file->private_data;
struct fuse_mount *fm = ff->fm;
- unsigned int offset, i;
- bool short_write;
+ unsigned int i;
int err;
for (i = 0; i < ap->num_folios; i++)
@@ -1235,24 +1234,9 @@ static ssize_t fuse_send_write_pages(struct fuse_io_args *ia,
if (!err && ia->write.out.size > count)
err = -EIO;
- short_write = ia->write.out.size < count;
- offset = ap->descs[0].offset;
- count = ia->write.out.size;
for (i = 0; i < ap->num_folios; i++) {
struct folio *folio = ap->folios[i];
- if (err) {
- folio_clear_uptodate(folio);
- } else {
- if (count >= folio_size(folio) - offset)
- count -= folio_size(folio) - offset;
- else {
- if (short_write)
- folio_clear_uptodate(folio);
- count = 0;
- }
- offset = 0;
- }
if (ia->write.folio_locked && (i == ap->num_folios - 1))
folio_unlock(folio);
folio_put(folio);
@@ -0,0 +1,13 @@
diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -406,7 +406,8 @@ void fuse_chan_max_background_set(struct fuse_chan *fch, unsigned int val)
fch->max_background = val;
fch->blocked = fch->num_background >= fch->max_background;
if (!fch->blocked)
- wake_up(&fch->blocked_waitq);
+ wake_up_nr(&fch->blocked_waitq,
+ fch->max_background - fch->num_background);
spin_unlock(&fch->bg_lock);
}
@@ -0,0 +1,119 @@
diff --git a/fs/internal.h b/fs/internal.h
--- a/fs/internal.h
+++ b/fs/internal.h
@@ -204,6 +204,7 @@ int do_fchownat(int dfd, const char __user *filename, uid_t user, gid_t group,
int flag);
int chown_common(const struct path *path, uid_t user, gid_t group);
extern int vfs_open(const struct path *, struct file *);
+int vfs_open_consume(struct path *, struct file *);
/*
* inode.c
diff --git a/fs/namei.c b/fs/namei.c
--- a/fs/namei.c
+++ b/fs/namei.c
@@ -4652,6 +4652,7 @@ static const char *open_last_lookups(struct nameidata *nd,
static int do_open(struct nameidata *nd,
struct file *file, const struct open_flags *op)
{
+ struct vfsmount *mnt;
struct mnt_idmap *idmap;
int open_flag = op->open_flag;
bool do_truncate;
@@ -4693,11 +4694,17 @@ static int do_open(struct nameidata *nd,
error = mnt_want_write(nd->path.mnt);
if (error)
return error;
+ /*
+ * A dedicated reference is needed because after the call to
+ * vfs_open_consume() we no longer own the reference in nd->path.mnt
+ * while we need to undo write acess below.
+ */
+ mnt = mntget(nd->path.mnt);
do_truncate = true;
}
error = may_open(idmap, &nd->path, acc_mode, open_flag);
if (!error && !(file->f_mode & FMODE_OPENED))
- error = vfs_open(&nd->path, file);
+ error = vfs_open_consume(&nd->path, file);
if (!error)
error = security_file_post_open(file, op->acc_mode);
if (!error && do_truncate)
@@ -4706,8 +4713,10 @@ static int do_open(struct nameidata *nd,
WARN_ON(1);
error = -EINVAL;
}
- if (do_truncate)
- mnt_drop_write(nd->path.mnt);
+ if (do_truncate) {
+ mnt_drop_write(mnt);
+ mntput(mnt);
+ }
return error;
}
diff --git a/fs/open.c b/fs/open.c
--- a/fs/open.c
+++ b/fs/open.c
@@ -882,6 +882,11 @@ static inline int file_get_write_access(struct file *f)
return error;
}
+/*
+ * Populate struct file
+ *
+ * NOTE: it assumes f_path is populated and consumes the caller's reference.
+ */
static int do_dentry_open(struct file *f,
int (*open)(struct inode *, struct file *))
{
@@ -889,7 +894,6 @@ static int do_dentry_open(struct file *f,
struct inode *inode = f->f_path.dentry->d_inode;
int error;
- path_get(&f->f_path);
f->f_inode = inode;
f->f_mapping = inode->i_mapping;
f->f_wb_err = filemap_sample_wb_err(f->f_mapping);
@@ -1006,6 +1010,7 @@ int finish_open(struct file *file, struct dentry *dentry,
BUG_ON(file->f_mode & FMODE_OPENED); /* once it's opened, it's opened */
file->__f_path.dentry = dentry;
+ path_get(&file->f_path);
return do_dentry_open(file, open);
}
EXPORT_SYMBOL(finish_open);
@@ -1049,6 +1054,7 @@ int vfs_open(const struct path *path, struct file *file)
int ret;
file->__f_path = *path;
+ path_get(&file->f_path);
ret = do_dentry_open(file, NULL);
if (!ret) {
/*
@@ -1061,6 +1067,25 @@ int vfs_open(const struct path *path, struct file *file)
return ret;
}
+/**
+ * vfs_open_consume - open the file at the given path and consume the reference
+ * @path: path to open
+ * @file: newly allocated file with f_flag initialized
+ */
+int vfs_open_consume(struct path *path, struct file *file)
+{
+ int ret;
+
+ file->__f_path = *path;
+ path->mnt = NULL;
+ path->dentry = NULL;
+ ret = do_dentry_open(file, NULL);
+ if (!ret) {
+ fsnotify_open(file);
+ }
+ return ret;
+}
+
struct file *dentry_open(const struct path *path, int flags,
const struct cred *cred)
{
@@ -0,0 +1,123 @@
diff --git a/drivers/gpu/drm/drm_displayid_internal.h b/drivers/gpu/drm/drm_displayid_internal.h
index 4590d6a3d821..dbdc2053332d 100644
--- a/drivers/gpu/drm/drm_displayid_internal.h
+++ b/drivers/gpu/drm/drm_displayid_internal.h
@@ -67,6 +67,7 @@ struct drm_edid;
#define DATA_BLOCK_2_TILED_DISPLAY_TOPOLOGY 0x28
#define DATA_BLOCK_2_CONTAINER_ID 0x29
#define DATA_BLOCK_2_TYPE_10_FORMULA_TIMING 0x2a
+#define DATA_BLOCK_2_ADAPTIVE_SYNC 0x2b
#define DATA_BLOCK_2_VENDOR_SPECIFIC 0x7e
#define DATA_BLOCK_2_CTA_DISPLAY_ID 0x81
diff --git a/drivers/gpu/drm/drm_edid.c b/drivers/gpu/drm/drm_edid.c
index df3c25bac761..354d4ff1aff7 100644
--- a/drivers/gpu/drm/drm_edid.c
+++ b/drivers/gpu/drm/drm_edid.c
@@ -6598,6 +6598,49 @@ static void drm_get_monitor_range(struct drm_connector *connector,
info->monitor_range.min_vfreq, info->monitor_range.max_vfreq);
}
+#define DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE 6
+
+static bool drm_update_displayid_adaptive_sync_range(struct drm_connector *connector,
+ const struct displayid_block *block)
+{
+ struct drm_monitor_range_info *range = &connector->display_info.monitor_range;
+ const u8 *data = (const u8 *)(block + 1);
+ u16 best_min = 0, best_max = 0;
+ unsigned int best_span = 0;
+ int i;
+
+ if (block->rev != 0 || block->num_bytes < DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE ||
+ block->num_bytes % DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE)
+ return false;
+
+ for (i = 0; i < block->num_bytes; i += DISPLAYID_ADAPTIVE_SYNC_DESC_SIZE) {
+ const u8 *desc = &data[i];
+ u16 min_vfreq = desc[2];
+ u16 max_vfreq = ((((u16)desc[4]) & 0x3) << 8) | desc[3];
+ unsigned int span;
+
+ max_vfreq += 1;
+ if (!min_vfreq || max_vfreq <= min_vfreq)
+ continue;
+
+ span = max_vfreq - min_vfreq;
+ if (span <= best_span)
+ continue;
+
+ best_min = min_vfreq;
+ best_max = max_vfreq;
+ best_span = span;
+ }
+
+ if (best_span) {
+ range->min_vfreq = best_min;
+ range->max_vfreq = best_max;
+ return true;
+ }
+
+ return false;
+}
+
static void drm_parse_vesa_mso_data(struct drm_connector *connector,
const struct displayid_block *block)
{
@@ -6720,26 +6763,43 @@ static void update_displayid_info(struct drm_connector *connector,
{
struct drm_display_info *info = &connector->display_info;
const struct displayid_block *block;
+ bool adaptive_sync_range = false;
+ bool displayid_base_logged = false;
struct displayid_iter iter;
displayid_iter_edid_begin(drm_edid, &iter);
displayid_iter_for_each(block, &iter) {
- drm_dbg_kms(connector->dev,
- "[CONNECTOR:%d:%s] DisplayID extension version 0x%02x, primary use 0x%02x\n",
- connector->base.id, connector->name,
- displayid_version(&iter),
- displayid_primary_use(&iter));
- if (displayid_version(&iter) == DISPLAY_ID_STRUCTURE_VER_20 &&
- (displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_VR ||
- displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_AR))
- info->non_desktop = true;
-
/*
- * We're only interested in the base section here, no need to
- * iterate further.
+ * Primary use is a DisplayID base section property, but later
+ * blocks may still carry useful metadata like adaptive sync ranges.
*/
- break;
+ if (!displayid_base_logged) {
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] DisplayID extension version 0x%02x, primary use 0x%02x\n",
+ connector->base.id, connector->name,
+ displayid_version(&iter),
+ displayid_primary_use(&iter));
+ if (displayid_version(&iter) == DISPLAY_ID_STRUCTURE_VER_20 &&
+ (displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_VR ||
+ displayid_primary_use(&iter) == PRIMARY_USE_HEAD_MOUNTED_AR))
+ info->non_desktop = true;
+
+ displayid_base_logged = true;
+ }
+
+ if (!info->monitor_range.min_vfreq && !info->monitor_range.max_vfreq &&
+ block->tag == DATA_BLOCK_2_ADAPTIVE_SYNC)
+ adaptive_sync_range =
+ drm_update_displayid_adaptive_sync_range(connector, block);
}
+
+ if (adaptive_sync_range)
+ drm_dbg_kms(connector->dev,
+ "[CONNECTOR:%d:%s] DisplayID adaptive sync refresh rate range is %d Hz - %d Hz\n",
+ connector->base.id, connector->name,
+ info->monitor_range.min_vfreq,
+ info->monitor_range.max_vfreq);
+
displayid_iter_end(&iter);
}
@@ -0,0 +1,601 @@
diff --git a/drivers/gpu/drm/nouveau/nouveau_ttm.c b/drivers/gpu/drm/nouveau/nouveau_ttm.c
--- a/drivers/gpu/drm/nouveau/nouveau_ttm.c
+++ b/drivers/gpu/drm/nouveau/nouveau_ttm.c
@@ -188,6 +188,11 @@ nouveau_ttm_init_vram(struct nouveau_drm *drm)
man->func = &nouveau_vram_manager;
+ man->cg = drmm_cgroup_register_region(drm->dev, "vram",
+ drm->gem.vram_available);
+ if (IS_ERR(man->cg))
+ return PTR_ERR(man->cg);
+
ttm_resource_manager_init(man, &drm->ttm.bdev,
drm->gem.vram_available >> PAGE_SHIFT);
ttm_set_driver_manager(&drm->ttm.bdev, TTM_PL_VRAM, man);
diff --git a/drivers/gpu/drm/ttm/ttm_bo.c b/drivers/gpu/drm/ttm/ttm_bo.c
--- a/drivers/gpu/drm/ttm/ttm_bo.c
+++ b/drivers/gpu/drm/ttm/ttm_bo.c
@@ -488,6 +488,118 @@ int ttm_bo_evict_first(struct ttm_device *bdev, struct ttm_resource_manager *man
return ret;
}
+struct ttm_bo_alloc_state {
+ /** @charge_pool: The memory pool the resource is charged to */
+ struct dmem_cgroup_pool_state *charge_pool;
+ /** @limit_pool: Which pool limit we should test against */
+ struct dmem_cgroup_pool_state *limit_pool;
+ /** @in_evict: Whether we are currently evicting buffers */
+ bool in_evict;
+ /** @may_try_low: If only unprotected BOs, i.e. BOs whose cgroup
+ * is exceeding its dmem low/min protection, should be considered for eviction
+ */
+ bool may_try_low;
+};
+
+/**
+ * ttm_bo_alloc_at_place - Attempt allocating a BO's backing store in a place
+ *
+ * @bo: The buffer to allocate the backing store of
+ * @place: The place to attempt allocation in
+ * @ctx: ttm_operation_ctx associated with this allocation
+ * @force_space: If we should evict buffers to force space
+ * @res: On allocation success, the resulting struct ttm_resource.
+ * @alloc_state: Object holding allocation state such as charged cgroups.
+ *
+ * Returns:
+ * -EBUSY: No space available, but allocation should be retried with ttm_bo_evict_alloc.
+ * -ENOSPC: No space available, allocation should not be retried.
+ * -ERESTARTSYS: An interruptible sleep was interrupted by a signal.
+ *
+ */
+static int ttm_bo_alloc_at_place(struct ttm_buffer_object *bo,
+ const struct ttm_place *place,
+ bool force_space,
+ struct ttm_resource **res,
+ struct ttm_bo_alloc_state *alloc_state)
+{
+ bool may_evict;
+ int ret;
+
+ may_evict = !alloc_state->in_evict && force_space &&
+ place->mem_type != TTM_PL_SYSTEM;
+ if (!alloc_state->charge_pool) {
+ ret = ttm_resource_try_charge(bo, place, &alloc_state->charge_pool,
+ force_space ? &alloc_state->limit_pool
+ : NULL);
+ if (ret) {
+ /*
+ * -EAGAIN means the charge failed, which we treat
+ * like an allocation failure. Therefore, return an
+ * error code indicating the allocation failed -
+ * either -EBUSY if the allocation should be
+ * retried with eviction, or -ENOSPC if there should
+ * be no second attempt.
+ */
+ if (!alloc_state->in_evict)
+ alloc_state->may_try_low = may_evict;
+ if (ret == -EAGAIN)
+ ret = may_evict ? -EBUSY : -ENOSPC;
+ return ret;
+ }
+ }
+
+ /*
+ * cgroup protection plays a special role in eviction.
+ * Conceptually, protection of memory via the dmem cgroup controller
+ * entitles the protected cgroup to use a certain amount of memory.
+ * There are two types of protection - the 'low' limit is a
+ * "best-effort" protection, whereas the 'min' limit provides a hard
+ * guarantee that memory within the cgroup's allowance will not be
+ * evicted under any circumstance.
+ *
+ * To faithfully model this concept in TTM, we also need to take cgroup
+ * protection into account when allocating. When allocation in one
+ * place fails, TTM will default to trying other places first before
+ * evicting.
+ * If the allocation is covered by dmem cgroup protection, however,
+ * this prevents the allocation from using the memory it is "entitled"
+ * to. To make sure unprotected allocations cannot push new protected
+ * allocations out of places they are "entitled" to use, we should
+ * evict buffers not covered by any cgroup protection, if this
+ * allocation is covered by cgroup protection.
+ *
+ * Buffers covered by 'min' protection are a special case - the 'min'
+ * limit is a stronger guarantee than 'low', and thus buffers protected
+ * by 'low' but not 'min' should also be considered for eviction.
+ * Buffers protected by 'min' will never be considered for eviction
+ * anyway, so the regular eviction path should be triggered here.
+ * Buffers protected by 'low' but not 'min' will take a special
+ * eviction path that only evicts buffers covered by neither 'low' or
+ * 'min' protections.
+ */
+ if (!alloc_state->in_evict) {
+ may_evict |= dmem_cgroup_below_min(NULL, alloc_state->charge_pool);
+ alloc_state->may_try_low = may_evict;
+
+ may_evict |= dmem_cgroup_below_low(NULL, alloc_state->charge_pool);
+ }
+
+ ret = ttm_resource_alloc(bo, place, res, alloc_state->charge_pool);
+ if (ret) {
+ if (ret == -ENOSPC && may_evict)
+ ret = -EBUSY;
+ return ret;
+ }
+
+ /*
+ * Ownership of charge_pool has been transferred to the TTM resource,
+ * don't make the caller think we still hold a reference to it.
+ */
+ alloc_state->charge_pool = NULL;
+ return 0;
+}
+
/**
* struct ttm_bo_evict_walk - Parameters for the evict walk.
*/
@@ -503,22 +615,61 @@ struct ttm_bo_evict_walk {
/** @evicted: Number of successful evictions. */
unsigned long evicted;
- /** @limit_pool: Which pool limit we should test against */
- struct dmem_cgroup_pool_state *limit_pool;
/** @try_low: Whether we should attempt to evict BO's with low watermark threshold */
bool try_low;
/** @hit_low: If we cannot evict a bo when @try_low is false (first pass) */
bool hit_low;
+
+ /** @alloc_state: State associated with the allocation attempt. */
+ struct ttm_bo_alloc_state *alloc_state;
};
static s64 ttm_bo_evict_cb(struct ttm_lru_walk *walk, struct ttm_buffer_object *bo)
{
struct ttm_bo_evict_walk *evict_walk =
container_of(walk, typeof(*evict_walk), walk);
+ struct dmem_cgroup_pool_state *limit_pool, *ancestor = NULL;
+ bool evict_valuable;
s64 lret;
- if (!dmem_cgroup_state_evict_valuable(evict_walk->limit_pool, bo->resource->css,
- evict_walk->try_low, &evict_walk->hit_low))
+ /*
+ * If may_try_low is not set, then we're trying to evict unprotected
+ * buffers in favor of a protected allocation for charge_pool. Explicitly skip
+ * buffers belonging to the same cgroup here - that cgroup is definitely protected,
+ * even though dmem_cgroup_state_evict_valuable would allow the eviction because a
+ * cgroup is always allowed to evict from itself even if it is protected.
+ */
+ if (!evict_walk->alloc_state->may_try_low &&
+ bo->resource->css == evict_walk->alloc_state->charge_pool)
+ return 0;
+
+ limit_pool = evict_walk->alloc_state->limit_pool;
+ /*
+ * If there is no explicit limit pool, find the root of the shared subtree between
+ * evictor and evictee. This is important so that recursive protection rules can
+ * apply properly: Recursive protection distributes cgroup protection afforded
+ * to a parent cgroup but not used explicitly by a child cgroup between all child
+ * cgroups (see docs of effective_protection in mm/page_counter.c). However, when
+ * direct siblings compete for memory, siblings that were explicitly protected
+ * should get prioritized over siblings that weren't. This only happens correctly
+ * when the root of the shared subtree is passed to
+ * dmem_cgroup_state_evict_valuable. Otherwise, the effective-protection
+ * calculation cannot distinguish direct siblings from unrelated subtrees and the
+ * calculated protection ends up wrong.
+ */
+ if (!limit_pool) {
+ ancestor = dmem_cgroup_get_common_ancestor(bo->resource->css,
+ evict_walk->alloc_state->charge_pool);
+ limit_pool = ancestor;
+ }
+
+ evict_valuable = dmem_cgroup_state_evict_valuable(limit_pool, bo->resource->css,
+ evict_walk->try_low,
+ &evict_walk->hit_low);
+ if (ancestor)
+ dmem_cgroup_pool_state_put(ancestor);
+
+ if (!evict_valuable)
return 0;
if (bo->pin_count || !bo->bdev->funcs->eviction_valuable(bo, evict_walk->place))
@@ -537,8 +688,10 @@ static s64 ttm_bo_evict_cb(struct ttm_lru_walk *walk, struct ttm_buffer_object *
evict_walk->evicted++;
if (evict_walk->res)
- lret = ttm_resource_alloc(evict_walk->evictor, evict_walk->place,
- evict_walk->res, NULL);
+ lret = ttm_bo_alloc_at_place(evict_walk->evictor,
+ evict_walk->place, false,
+ evict_walk->res,
+ evict_walk->alloc_state);
if (lret == 0)
return 1;
out:
@@ -560,7 +713,7 @@ static int ttm_bo_evict_alloc(struct ttm_device *bdev,
struct ttm_operation_ctx *ctx,
struct ww_acquire_ctx *ticket,
struct ttm_resource **res,
- struct dmem_cgroup_pool_state *limit_pool)
+ struct ttm_bo_alloc_state *state)
{
struct ttm_bo_evict_walk evict_walk = {
.walk = {
@@ -573,15 +726,21 @@ static int ttm_bo_evict_alloc(struct ttm_device *bdev,
.place = place,
.evictor = evictor,
.res = res,
- .limit_pool = limit_pool,
+ .alloc_state = state,
};
s64 lret;
+ state->in_evict = true;
+
evict_walk.walk.arg.trylock_only = true;
lret = ttm_lru_walk_for_evict(&evict_walk.walk, bdev, man, 1);
- /* One more attempt if we hit low limit? */
- if (!lret && evict_walk.hit_low) {
+ /* If we failed to find enough BOs to evict, but we skipped over
+ * some BOs because they were covered by dmem low protection, retry
+ * evicting these protected BOs too, except if we're told not to
+ * consider protected BOs at all.
+ */
+ if (!lret && evict_walk.hit_low && state->may_try_low) {
evict_walk.try_low = true;
lret = ttm_lru_walk_for_evict(&evict_walk.walk, bdev, man, 1);
}
@@ -602,11 +761,13 @@ static int ttm_bo_evict_alloc(struct ttm_device *bdev,
} while (!lret && evict_walk.evicted);
/* We hit the low limit? Try once more */
- if (!lret && evict_walk.hit_low && !evict_walk.try_low) {
+ if (!lret && evict_walk.hit_low && !evict_walk.try_low &&
+ state->may_try_low) {
evict_walk.try_low = true;
goto retry;
}
out:
+ state->in_evict = false;
if (lret < 0)
return lret;
if (lret == 0)
@@ -724,9 +885,8 @@ static int ttm_bo_alloc_resource(struct ttm_buffer_object *bo,
for (i = 0; i < placement->num_placement; ++i) {
const struct ttm_place *place = &placement->placement[i];
- struct dmem_cgroup_pool_state *limit_pool = NULL;
+ struct ttm_bo_alloc_state alloc_state = {};
struct ttm_resource_manager *man;
- bool may_evict;
man = ttm_manager_type(bdev, place->mem_type);
if (!man || !ttm_resource_manager_used(man))
@@ -736,25 +896,30 @@ static int ttm_bo_alloc_resource(struct ttm_buffer_object *bo,
TTM_PL_FLAG_FALLBACK))
continue;
- may_evict = (force_space && place->mem_type != TTM_PL_SYSTEM);
- ret = ttm_resource_alloc(bo, place, res, force_space ? &limit_pool : NULL);
- if (ret) {
- if (ret != -ENOSPC) {
- dmem_cgroup_pool_state_put(limit_pool);
- return ret;
- }
- if (!may_evict) {
- dmem_cgroup_pool_state_put(limit_pool);
- continue;
- }
+ ret = ttm_bo_alloc_at_place(bo, place, force_space, res,
+ &alloc_state);
+ if (ret == -ENOSPC) {
+ dmem_cgroup_uncharge(alloc_state.charge_pool, bo->base.size);
+ dmem_cgroup_pool_state_put(alloc_state.limit_pool);
+ continue;
+ } else if (ret == -EBUSY) {
ret = ttm_bo_evict_alloc(bdev, man, place, bo, ctx,
- ticket, res, limit_pool);
- dmem_cgroup_pool_state_put(limit_pool);
- if (ret == -EBUSY)
- continue;
- if (ret)
+ ticket, res, &alloc_state);
+
+ dmem_cgroup_pool_state_put(alloc_state.limit_pool);
+
+ if (ret) {
+ dmem_cgroup_uncharge(alloc_state.charge_pool,
+ bo->base.size);
+ if (ret == -EBUSY)
+ continue;
return ret;
+ }
+ } else if (ret) {
+ dmem_cgroup_uncharge(alloc_state.charge_pool, bo->base.size);
+ dmem_cgroup_pool_state_put(alloc_state.limit_pool);
+ return ret;
}
ret = ttm_bo_add_pipelined_eviction_fences(bo, man, ctx->no_wait_gpu);
diff --git a/drivers/gpu/drm/ttm/ttm_resource.c b/drivers/gpu/drm/ttm/ttm_resource.c
--- a/drivers/gpu/drm/ttm/ttm_resource.c
+++ b/drivers/gpu/drm/ttm/ttm_resource.c
@@ -386,33 +386,52 @@ void ttm_resource_fini(struct ttm_resource_manager *man,
}
EXPORT_SYMBOL(ttm_resource_fini);
+/**
+ * ttm_resource_try_charge - charge a resource manager's cgroup pool
+ * @bo: buffer for which an allocation should be charged
+ * @place: where the allocation is attempted to be placed
+ * @ret_pool: on charge success, the pool that was charged
+ * @ret_limit_pool: on charge failure, the pool responsible for the failure
+ *
+ * Should be used to charge cgroups before attempting resource allocation.
+ * When charging succeeds, the value of ret_pool should be passed to
+ * ttm_resource_alloc.
+ *
+ * Returns: 0 on charge success, negative errno on failure.
+ */
+int ttm_resource_try_charge(struct ttm_buffer_object *bo,
+ const struct ttm_place *place,
+ struct dmem_cgroup_pool_state **ret_pool,
+ struct dmem_cgroup_pool_state **ret_limit_pool)
+{
+ struct ttm_resource_manager *man =
+ ttm_manager_type(bo->bdev, place->mem_type);
+
+ if (!man->cg) {
+ *ret_pool = NULL;
+ if (ret_limit_pool)
+ *ret_limit_pool = NULL;
+ return 0;
+ }
+
+ return dmem_cgroup_try_charge(man->cg, bo->base.size, ret_pool,
+ ret_limit_pool);
+}
+
int ttm_resource_alloc(struct ttm_buffer_object *bo,
const struct ttm_place *place,
struct ttm_resource **res_ptr,
- struct dmem_cgroup_pool_state **ret_limit_pool)
+ struct dmem_cgroup_pool_state *charge_pool)
{
struct ttm_resource_manager *man =
ttm_manager_type(bo->bdev, place->mem_type);
- struct dmem_cgroup_pool_state *pool = NULL;
int ret;
- if (man->cg) {
- ret = dmem_cgroup_try_charge(man->cg, bo->base.size, &pool, ret_limit_pool);
- if (ret) {
- if (ret == -EAGAIN)
- ret = -ENOSPC;
- return ret;
- }
- }
-
ret = man->func->alloc(man, bo, place, res_ptr);
- if (ret) {
- if (pool)
- dmem_cgroup_uncharge(pool, bo->base.size);
+ if (ret)
return ret;
- }
- (*res_ptr)->css = pool;
+ (*res_ptr)->css = charge_pool;
spin_lock(&bo->bdev->lru_lock);
ttm_resource_add_bulk_move(*res_ptr, bo);
diff --git a/include/drm/ttm/ttm_resource.h b/include/drm/ttm/ttm_resource.h
--- a/include/drm/ttm/ttm_resource.h
+++ b/include/drm/ttm/ttm_resource.h
@@ -458,10 +458,14 @@ void ttm_resource_init(struct ttm_buffer_object *bo,
void ttm_resource_fini(struct ttm_resource_manager *man,
struct ttm_resource *res);
+int ttm_resource_try_charge(struct ttm_buffer_object *bo,
+ const struct ttm_place *place,
+ struct dmem_cgroup_pool_state **ret_pool,
+ struct dmem_cgroup_pool_state **ret_limit_pool);
int ttm_resource_alloc(struct ttm_buffer_object *bo,
const struct ttm_place *place,
struct ttm_resource **res,
- struct dmem_cgroup_pool_state **ret_limit_pool);
+ struct dmem_cgroup_pool_state *charge_pool);
void ttm_resource_free(struct ttm_buffer_object *bo, struct ttm_resource **res);
bool ttm_resource_intersects(struct ttm_device *bdev,
struct ttm_resource *res,
diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h
--- a/include/linux/cgroup.h
+++ b/include/linux/cgroup.h
@@ -623,6 +623,27 @@ static inline struct cgroup *cgroup_ancestor(struct cgroup *cgrp,
return cgrp->ancestors[ancestor_level];
}
+/**
+ * cgroup_common_ancestor - find common ancestor of two cgroups
+ * @a: first cgroup to find common ancestor of
+ * @b: second cgroup to find common ancestor of
+ *
+ * Find the first cgroup that is an ancestor of both @a and @b, if it exists
+ * and return a pointer to it. If such a cgroup doesn't exist, return NULL.
+ *
+ * This function is safe to call as long as both @a and @b are accessible.
+ */
+static inline struct cgroup *cgroup_common_ancestor(struct cgroup *a,
+ struct cgroup *b)
+{
+ int level;
+
+ for (level = min(a->level, b->level); level >= 0; level--)
+ if (a->ancestors[level] == b->ancestors[level])
+ return a->ancestors[level];
+ return NULL;
+}
+
/**
* task_under_cgroup_hierarchy - test task's membership of cgroup ancestry
* @task: the task to be tested
diff --git a/include/linux/cgroup_dmem.h b/include/linux/cgroup_dmem.h
--- a/include/linux/cgroup_dmem.h
+++ b/include/linux/cgroup_dmem.h
@@ -24,6 +24,12 @@ void dmem_cgroup_uncharge(struct dmem_cgroup_pool_state *pool, u64 size);
bool dmem_cgroup_state_evict_valuable(struct dmem_cgroup_pool_state *limit_pool,
struct dmem_cgroup_pool_state *test_pool,
bool ignore_low, bool *ret_hit_low);
+bool dmem_cgroup_below_min(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test);
+bool dmem_cgroup_below_low(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test);
+struct dmem_cgroup_pool_state *dmem_cgroup_get_common_ancestor(struct dmem_cgroup_pool_state *a,
+ struct dmem_cgroup_pool_state *b);
void dmem_cgroup_pool_state_put(struct dmem_cgroup_pool_state *pool);
#else
@@ -59,6 +65,25 @@ bool dmem_cgroup_state_evict_valuable(struct dmem_cgroup_pool_state *limit_pool,
return true;
}
+static inline bool dmem_cgroup_below_min(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ return false;
+}
+
+static inline bool dmem_cgroup_below_low(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ return false;
+}
+
+static inline
+struct dmem_cgroup_pool_state *dmem_cgroup_get_common_ancestor(struct dmem_cgroup_pool_state *a,
+ struct dmem_cgroup_pool_state *b)
+{
+ return NULL;
+}
+
static inline void dmem_cgroup_pool_state_put(struct dmem_cgroup_pool_state *pool)
{ }
diff --git a/kernel/cgroup/dmem.c b/kernel/cgroup/dmem.c
--- a/kernel/cgroup/dmem.c
+++ b/kernel/cgroup/dmem.c
@@ -695,6 +695,110 @@ int dmem_cgroup_try_charge(struct dmem_cgroup_region *region, u64 size,
}
EXPORT_SYMBOL_GPL(dmem_cgroup_try_charge);
+/**
+ * dmem_cgroup_below_min() - Tests whether current usage is within min limit.
+ *
+ * @root: Root of the subtree to calculate protection for, or NULL to calculate global protection.
+ * @test: The pool to test the usage/min limit of.
+ *
+ * Return: true if usage is below min and the cgroup is protected, false otherwise.
+ */
+bool dmem_cgroup_below_min(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ if (root == test || !pool_parent(test))
+ return false;
+
+ if (!root) {
+ for (root = test; pool_parent(root); root = pool_parent(root))
+ {}
+ }
+
+ /*
+ * In mem_cgroup_below_min(), the memcg pendant, this call is missing.
+ * mem_cgroup_below_min() gets called during traversal of the cgroup tree, where
+ * protection is already calculated as part of the traversal. dmem cgroup eviction
+ * does not traverse the cgroup tree, so we need to recalculate effective protection
+ * here.
+ */
+ dmem_cgroup_calculate_protection(root, test);
+ return page_counter_read(&test->cnt) <= READ_ONCE(test->cnt.emin);
+}
+EXPORT_SYMBOL_GPL(dmem_cgroup_below_min);
+
+/**
+ * dmem_cgroup_below_low() - Tests whether current usage is within low limit.
+ *
+ * @root: Root of the subtree to calculate protection for, or NULL to calculate global protection.
+ * @test: The pool to test the usage/low limit of.
+ *
+ * Return: true if usage is below low and the cgroup is protected, false otherwise.
+ */
+bool dmem_cgroup_below_low(struct dmem_cgroup_pool_state *root,
+ struct dmem_cgroup_pool_state *test)
+{
+ if (root == test || !pool_parent(test))
+ return false;
+
+ if (!root) {
+ for (root = test; pool_parent(root); root = pool_parent(root))
+ {}
+ }
+
+ /*
+ * In mem_cgroup_below_low(), the memcg pendant, this call is missing.
+ * mem_cgroup_below_low() gets called during traversal of the cgroup tree, where
+ * protection is already calculated as part of the traversal. dmem cgroup eviction
+ * does not traverse the cgroup tree, so we need to recalculate effective protection
+ * here.
+ */
+ dmem_cgroup_calculate_protection(root, test);
+ return page_counter_read(&test->cnt) <= READ_ONCE(test->cnt.elow);
+}
+EXPORT_SYMBOL_GPL(dmem_cgroup_below_low);
+
+/**
+ * dmem_cgroup_get_common_ancestor(): Find the first common ancestor of two pools.
+ * @a: First pool to find the common ancestor of.
+ * @b: First pool to find the common ancestor of.
+ *
+ * Return: The first pool that is a parent of both @a and @b, or NULL if either @a or @b are NULL,
+ * or if such a pool does not exist. A reference to the returned pool is grabbed and must be
+ * released by the caller when it is done using the pool.
+ */
+struct dmem_cgroup_pool_state *dmem_cgroup_get_common_ancestor(struct dmem_cgroup_pool_state *a,
+ struct dmem_cgroup_pool_state *b)
+{
+ struct cgroup *ancestor_cgroup;
+ struct cgroup_subsys_state *ancestor_css;
+ struct dmemcg_state *ancestor_dmemcs = NULL;
+ struct dmem_cgroup_pool_state *pool = NULL;
+
+ if (!a || !b)
+ return NULL;
+
+ ancestor_cgroup = cgroup_common_ancestor(a->cs->css.cgroup, b->cs->css.cgroup);
+ if (!ancestor_cgroup)
+ return NULL;
+
+ rcu_read_lock();
+ ancestor_css = cgroup_e_css(ancestor_cgroup, &dmem_cgrp_subsys);
+ if (css_tryget(ancestor_css))
+ ancestor_dmemcs = css_to_dmemcs(ancestor_css);
+ rcu_read_unlock();
+
+ if (ancestor_dmemcs) {
+ pool = get_cg_pool_unlocked(css_to_dmemcs(ancestor_css),
+ a->region);
+ if (WARN_ON(IS_ERR(pool))) {
+ pool = NULL;
+ css_put(ancestor_css);
+ }
+ }
+ return pool;
+}
+EXPORT_SYMBOL_GPL(dmem_cgroup_get_common_ancestor);
+
static int dmem_cgroup_region_capacity_show(struct seq_file *sf, void *v)
{
struct dmem_cgroup_region *region;
@@ -0,0 +1,13 @@
diff --git a/drivers/gpu/drm/i915/display/intel_alpm.c b/drivers/gpu/drm/i915/display/intel_alpm.c
index c6963ea420cc..c38d613f15d6 100644
--- a/drivers/gpu/drm/i915/display/intel_alpm.c
+++ b/drivers/gpu/drm/i915/display/intel_alpm.c
@@ -399,7 +399,7 @@ static void lnl_alpm_configure(struct intel_dp *intel_dp,
ALPM_CTL_AUX_LESS_SLEEP_HOLD_TIME_50_SYMBOLS |
ALPM_CTL_AUX_LESS_WAKE_TIME(crtc_state->alpm_state.aux_less_wake_lines);
- if (intel_dp->as_sdp_supported) {
+ if (intel_dp->as_sdp_supported && crtc_state->has_panel_replay) {
u32 pr_alpm_ctl = get_pr_alpm_as_sdp_transmission_time(crtc_state);
if (crtc_state->link_off_after_as_sdp_when_pr_active)
@@ -0,0 +1,20 @@
diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c
index beaa1d626..5fdf4594c 100644
--- a/drivers/gpu/drm/i915/display/intel_psr.c
+++ b/drivers/gpu/drm/i915/display/intel_psr.c
@@ -714,8 +714,14 @@ static void _psr_init_dpcd(struct intel_dp *intel_dp, struct intel_connector *co
* To support PSR version 02h and PSR version 03h without
* Y-coordinate requirement panels we would need to enable
* GTC first.
+ *
+ * Early Transport (version 04h) implies Y-coordinate
+ * support. Accept it without explicit Y-coordinate
+ * requirement bit.
*/
- connector->dp.psr_caps.su_support = y_req &&
+ connector->dp.psr_caps.su_support =
+ (y_req || connector->dp.psr_caps.dpcd[0] ==
+ DP_PSR2_WITH_Y_COORD_ET_SUPPORTED) &&
intel_alpm_aux_wake_supported(intel_dp);
drm_dbg_kms(display->drm, "PSR2 %ssupported\n",
connector->dp.psr_caps.su_support ? "" : "not ");
@@ -0,0 +1,172 @@
diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h
--- a/drivers/gpu/drm/i915/display/intel_display_types.h
+++ b/drivers/gpu/drm/i915/display/intel_display_types.h
@@ -1780,6 +1780,7 @@ struct intel_psr {
u32 dc3co_exitline;
u32 dc3co_exit_delay;
struct delayed_work dc3co_work;
+ struct delayed_work panel_replay_reenable_work;
u8 entry_setup_frames;
u8 io_wake_lines;
diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c
--- a/drivers/gpu/drm/i915/display/intel_psr.c
+++ b/drivers/gpu/drm/i915/display/intel_psr.c
@@ -2481,6 +2481,7 @@ void intel_psr_disable(struct intel_dp *intel_dp,
mutex_unlock(&intel_dp->psr.lock);
cancel_work_sync(&intel_dp->psr.work);
cancel_delayed_work_sync(&intel_dp->psr.dc3co_work);
+ cancel_delayed_work_sync(&intel_dp->psr.panel_replay_reenable_work);
}
/**
@@ -2512,6 +2513,7 @@ void intel_psr_pause(struct intel_dp *intel_dp)
cancel_work_sync(&psr->work);
cancel_delayed_work_sync(&psr->dc3co_work);
+ cancel_delayed_work_sync(&psr->panel_replay_reenable_work);
}
/**
@@ -3545,6 +3547,17 @@ static void intel_psr_handle_irq(struct intel_dp *intel_dp)
drm_dp_dpcd_writeb(&intel_dp->aux, DP_SET_POWER, DP_SET_POWER_D0);
}
+#define PANEL_REPLAY_REENABLE_DELAY_MS 50
+
+static bool panel_replay_exit_on_flush_enabled(struct intel_dp *intel_dp)
+{
+ return intel_dp_is_edp(intel_dp) &&
+ intel_dp->psr.panel_replay_enabled &&
+ intel_dp->psr.sel_update_enabled &&
+ intel_has_dpcd_quirk(intel_dp,
+ QUIRK_PANEL_REPLAY_EXIT_ON_FLUSH);
+}
+
static void intel_psr_work(struct work_struct *work)
{
struct intel_dp *intel_dp =
@@ -3563,6 +3576,11 @@ static void intel_psr_work(struct work_struct *work)
if (intel_dp->psr.pause_counter)
goto unlock;
+ /* The dedicated delayed work owns re-entry while the quirk is armed. */
+ if (panel_replay_exit_on_flush_enabled(intel_dp) &&
+ delayed_work_pending(&intel_dp->psr.panel_replay_reenable_work))
+ goto unlock;
+
/*
* We have to make sure PSR is ready for re-enable
* otherwise it keeps disabled until next full enable/disable cycle.
@@ -3572,6 +3590,10 @@ static void intel_psr_work(struct work_struct *work)
if (!__psr_wait_for_idle_locked(intel_dp))
goto unlock;
+ if (panel_replay_exit_on_flush_enabled(intel_dp) &&
+ delayed_work_pending(&intel_dp->psr.panel_replay_reenable_work))
+ goto unlock;
+
/*
* The delayed work can race with an invalidate hence we need to
* recheck. Since psr_flush first clears this and then reschedules we
@@ -3585,6 +3607,33 @@ static void intel_psr_work(struct work_struct *work)
mutex_unlock(&intel_dp->psr.lock);
}
+static void panel_replay_reenable_work(struct work_struct *work)
+{
+ struct intel_dp *intel_dp =
+ container_of(work, typeof(*intel_dp),
+ psr.panel_replay_reenable_work.work);
+
+ mutex_lock(&intel_dp->psr.lock);
+
+ if (!intel_dp->psr.enabled ||
+ !panel_replay_exit_on_flush_enabled(intel_dp) ||
+ intel_dp->psr.pause_counter ||
+ READ_ONCE(intel_dp->psr.irq_aux_error))
+ goto unlock;
+
+ if (!__psr_wait_for_idle_locked(intel_dp))
+ goto unlock;
+
+ /* Recheck for activity or a newer deadline after the unlocked wait. */
+ if (delayed_work_pending(&intel_dp->psr.panel_replay_reenable_work) ||
+ intel_dp->psr.busy_frontbuffer_bits || intel_dp->psr.active)
+ goto unlock;
+
+ intel_psr_activate(intel_dp);
+unlock:
+ mutex_unlock(&intel_dp->psr.lock);
+}
+
static void intel_psr_configure_full_frame_update(struct intel_dp *intel_dp)
{
struct intel_display *display = to_intel_display(intel_dp);
@@ -3656,8 +3705,11 @@ void intel_psr_invalidate(struct intel_display *display,
INTEL_FRONTBUFFER_ALL_MASK(intel_dp->psr.pipe);
intel_dp->psr.busy_frontbuffer_bits |= pipe_frontbuffer_bits;
- if (pipe_frontbuffer_bits)
+ if (pipe_frontbuffer_bits) {
+ if (panel_replay_exit_on_flush_enabled(intel_dp))
+ cancel_delayed_work(&intel_dp->psr.panel_replay_reenable_work);
_psr_invalidate_handle(intel_dp);
+ }
mutex_unlock(&intel_dp->psr.lock);
}
@@ -3773,6 +3825,15 @@ void intel_psr_flush(struct intel_display *display,
if (intel_dp->psr.pause_counter)
goto unlock;
+ if (pipe_frontbuffer_bits &&
+ panel_replay_exit_on_flush_enabled(intel_dp)) {
+ intel_psr_exit(intel_dp);
+ mod_delayed_work(display->wq.unordered,
+ &intel_dp->psr.panel_replay_reenable_work,
+ msecs_to_jiffies(PANEL_REPLAY_REENABLE_DELAY_MS));
+ goto unlock;
+ }
+
if (origin == ORIGIN_FLIP ||
(origin == ORIGIN_CURSOR_UPDATE &&
!intel_dp->psr.psr2_sel_fetch_enabled)) {
@@ -3836,6 +3897,8 @@ void intel_psr_init(struct intel_dp *intel_dp)
INIT_WORK(&intel_dp->psr.work, intel_psr_work);
INIT_DELAYED_WORK(&intel_dp->psr.dc3co_work, tgl_dc3co_disable_work);
+ INIT_DELAYED_WORK(&intel_dp->psr.panel_replay_reenable_work,
+ panel_replay_reenable_work);
mutex_init(&intel_dp->psr.lock);
}
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c
--- a/drivers/gpu/drm/i915/display/intel_quirks.c
+++ b/drivers/gpu/drm/i915/display/intel_quirks.c
@@ -94,6 +94,14 @@ static void quirk_disable_edp_panel_replay(struct intel_dp *intel_dp)
drm_info(display->drm, "Applying disable Panel Replay quirk\n");
}
+static void quirk_panel_replay_exit_on_flush(struct intel_dp *intel_dp)
+{
+ struct intel_display *display = to_intel_display(intel_dp);
+
+ intel_set_dpcd_quirk(intel_dp, QUIRK_PANEL_REPLAY_EXIT_ON_FLUSH);
+ drm_info(display->drm, "Applying Panel Replay exit on flush quirk\n");
+}
+
static void quirk_disable_psr2(struct intel_display *display)
{
intel_set_quirk(display, QUIRK_DISABLE_PSR2);
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.h b/drivers/gpu/drm/i915/display/intel_quirks.h
--- a/drivers/gpu/drm/i915/display/intel_quirks.h
+++ b/drivers/gpu/drm/i915/display/intel_quirks.h
@@ -23,6 +23,7 @@ enum intel_quirk_id {
QUIRK_EDP_LIMIT_RATE_HBR2,
QUIRK_DISABLE_EDP_PANEL_REPLAY,
QUIRK_DISABLE_PSR2,
+ QUIRK_PANEL_REPLAY_EXIT_ON_FLUSH,
};
void intel_init_quirks(struct intel_display *display);
@@ -0,0 +1,12 @@
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c
--- a/drivers/gpu/drm/i915/display/intel_quirks.c
+++ b/drivers/gpu/drm/i915/display/intel_quirks.c
@@ -284,7 +284,7 @@ static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
.subsystem_vendor = 0x1028,
.subsystem_device = 0x0db9,
.sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
- .hook = quirk_disable_edp_panel_replay,
+ .hook = quirk_panel_replay_exit_on_flush,
},
/* Dell XPS 16 DA16260 */
{
@@ -0,0 +1,12 @@
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c
--- a/drivers/gpu/drm/i915/display/intel_quirks.c
+++ b/drivers/gpu/drm/i915/display/intel_quirks.c
@@ -292,7 +292,7 @@ static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
.subsystem_vendor = 0x1028,
.subsystem_device = 0x0dba,
.sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
- .hook = quirk_disable_edp_panel_replay,
+ .hook = quirk_panel_replay_exit_on_flush,
},
};
@@ -0,0 +1,28 @@
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c
--- a/drivers/gpu/drm/i915/display/intel_quirks.c
+++ b/drivers/gpu/drm/i915/display/intel_quirks.c
@@ -267,6 +267,9 @@ static struct intel_quirk intel_quirks[] = {
/* Xiaomi Book Pro 14 2026 */
{ 0xb081, 0x1d72, 0x2424, quirk_disable_psr2 },
+
+ /* Dell XPS 13 DX13260 */
+ { DEVICE_ID_ANY, 0x1028, 0x0e54, quirk_disable_psr2 },
};
static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
@@ -294,6 +297,14 @@ static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
.sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
.hook = quirk_panel_replay_exit_on_flush,
},
+ /* Dell XPS 13 DX13260 */
+ {
+ .device = DEVICE_ID_ANY,
+ .subsystem_vendor = 0x1028,
+ .subsystem_device = 0x0e54,
+ .sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
+ .hook = quirk_disable_edp_panel_replay,
+ },
};
void intel_init_quirks(struct intel_display *display)
@@ -0,0 +1,19 @@
diff --git a/drivers/gpu/drm/i915/display/intel_quirks.c b/drivers/gpu/drm/i915/display/intel_quirks.c
--- a/drivers/gpu/drm/i915/display/intel_quirks.c
+++ b/drivers/gpu/drm/i915/display/intel_quirks.c
@@ -305,6 +305,15 @@ static const struct intel_dpcd_quirk intel_dpcd_quirks[] = {
.sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
.hook = quirk_disable_edp_panel_replay,
},
+ /* Dell XPS 13 DX13260 (Wildcat Lake), sink "Bamboo" */
+ {
+ .device = DEVICE_ID_ANY,
+ .subsystem_vendor = 0x1028,
+ .subsystem_device = 0x0e53,
+ .sink_oui = SINK_OUI(0x00, 0x22, 0xb9),
+ .sink_device_id = SINK_DEVICE_ID(0x42, 0x61, 0x6d, 0x62, 0x6f, 0x6f),
+ .hook = quirk_panel_replay_exit_on_flush,
+ },
};
void intel_init_quirks(struct intel_display *display)
@@ -0,0 +1,43 @@
diff --git a/drivers/gpu/drm/xe/display/xe_display_bo.c b/drivers/gpu/drm/xe/display/xe_display_bo.c
--- a/drivers/gpu/drm/xe/display/xe_display_bo.c
+++ b/drivers/gpu/drm/xe/display/xe_display_bo.c
@@ -147,33 +147,12 @@ static struct drm_gem_object *xe_display_bo_fbdev_create(struct drm_device *drm,
struct xe_device *xe = to_xe_device(drm);
struct xe_bo *obj;
- obj = ERR_PTR(-ENODEV);
-
- if (xe_display_bo_fbdev_prefer_stolen(xe, size)) {
- obj = xe_bo_create_pin_map_novm(xe, xe_device_get_root_tile(xe),
- size,
- ttm_bo_type_kernel,
- XE_BO_FLAG_FORCE_WC |
- XE_BO_FLAG_STOLEN |
- XE_BO_FLAG_GGTT,
- false);
- if (!IS_ERR(obj))
- drm_info(&xe->drm, "Allocated fbdev into stolen\n");
- else
- drm_info(&xe->drm, "Allocated fbdev into stolen failed: %li\n", PTR_ERR(obj));
- } else {
- drm_info(&xe->drm, "Allocating fbdev: Stolen memory not preferred.\n");
- }
-
- if (IS_ERR(obj)) {
- obj = xe_bo_create_pin_map_novm(xe, xe_device_get_root_tile(xe), size,
- ttm_bo_type_kernel,
- XE_BO_FLAG_FORCE_WC |
- XE_BO_FLAG_VRAM_IF_DGFX(xe_device_get_root_tile(xe)) |
- XE_BO_FLAG_GGTT,
- false);
- }
-
+ obj = xe_bo_create_pin_map_novm(xe, xe_device_get_root_tile(xe), size,
+ ttm_bo_type_kernel,
+ XE_BO_FLAG_FORCE_WC |
+ XE_BO_FLAG_VRAM_IF_DGFX(xe_device_get_root_tile(xe)) |
+ XE_BO_FLAG_GGTT,
+ false);
if (IS_ERR(obj)) {
drm_err(&xe->drm, "failed to allocate framebuffer (%pe)\n", obj);
return ERR_PTR(-ENOMEM);
@@ -0,0 +1,23 @@
diff --git a/drivers/gpu/drm/i915/display/intel_display.c b/drivers/gpu/drm/i915/display/intel_display.c
--- a/drivers/gpu/drm/i915/display/intel_display.c
+++ b/drivers/gpu/drm/i915/display/intel_display.c
@@ -2452,6 +2452,19 @@ static int intel_crtc_set_context_latency(struct intel_crtc_state *crtc_state)
set_context_latency = max(set_context_latency,
intel_psr_min_set_context_latency(crtc_state));
+ /*
+ * From PTL onwards, the set context latency can be in the vactive
+ * region, letting the safe window start some lines before the vblank
+ * start. With modes that have a smaller vblank region, the computed
+ * guardband is clamped to the vblank length, making the undelayed and
+ * delayed vblank coincide. If the SCL is also 0, the 'safe window'
+ * becomes effectively 0, and the DSB configured to wait for it gets
+ * stalled, since the hardware never signals the safe window. Keep the
+ * set context latency at a minimum of 1 to avoid this.
+ */
+ if (DISPLAY_VER(display) >= 30)
+ set_context_latency = max(1, set_context_latency);
+
return set_context_latency;
}
@@ -0,0 +1,112 @@
diff --git a/drivers/gpu/drm/i915/display/intel_fbc.c b/drivers/gpu/drm/i915/display/intel_fbc.c
--- a/drivers/gpu/drm/i915/display/intel_fbc.c
+++ b/drivers/gpu/drm/i915/display/intel_fbc.c
@@ -59,6 +59,7 @@
#include "intel_fbc_regs.h"
#include "intel_frontbuffer.h"
#include "intel_parent.h"
+#include "skl_universal_plane.h"
#define for_each_fbc_id(__display, __fbc_id) \
for ((__fbc_id) = INTEL_FBC_A; (__fbc_id) < I915_MAX_FBCS; (__fbc_id)++) \
@@ -1305,6 +1306,9 @@ static bool intel_fbc_surface_size_ok(const struct intel_plane_state *plane_stat
struct intel_display *display = to_intel_display(plane_state);
unsigned int effective_w, effective_h, max_w, max_h;
+ if (DISPLAY_VER(display) >= 20)
+ return true;
+
intel_fbc_max_surface_size(display, &max_w, &max_h);
effective_w = plane_state->view.color_plane[0].x +
@@ -1315,10 +1319,18 @@ static bool intel_fbc_surface_size_ok(const struct intel_plane_state *plane_stat
return effective_w <= max_w && effective_h <= max_h;
}
-static void intel_fbc_max_plane_size(struct intel_display *display,
+static void intel_fbc_max_plane_size(const struct intel_plane_state *plane_state,
unsigned int *w, unsigned int *h)
{
- if (DISPLAY_VER(display) >= 10) {
+ struct intel_display *display = to_intel_display(plane_state);
+ struct intel_plane *plane = to_intel_plane(plane_state->uapi.plane);
+ const struct drm_framebuffer *fb = plane_state->hw.fb;
+ unsigned int rotation = plane_state->hw.rotation;
+
+ if (DISPLAY_VER(display) >= 20) {
+ *w = intel_plane_max_width(plane, fb, 0, rotation);
+ *h = 4096;
+ } else if (DISPLAY_VER(display) >= 10) {
*w = 5120;
*h = 4096;
} else if (DISPLAY_VER(display) >= 8 || display->platform.haswell) {
@@ -1335,10 +1347,9 @@ static void intel_fbc_max_plane_size(struct intel_display *display,
static bool intel_fbc_plane_size_valid(const struct intel_plane_state *plane_state)
{
- struct intel_display *display = to_intel_display(plane_state);
unsigned int w, h, max_w, max_h;
- intel_fbc_max_plane_size(display, &max_w, &max_h);
+ intel_fbc_max_plane_size(plane_state, &max_w, &max_h);
w = drm_rect_width(&plane_state->uapi.src) >> 16;
h = drm_rect_height(&plane_state->uapi.src) >> 16;
diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c
--- a/drivers/gpu/drm/i915/display/skl_universal_plane.c
+++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c
@@ -1932,10 +1932,10 @@ static int intel_plane_min_height(struct intel_plane *plane,
return 1;
}
-static int intel_plane_max_width(struct intel_plane *plane,
- const struct drm_framebuffer *fb,
- int color_plane,
- unsigned int rotation)
+int intel_plane_max_width(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation)
{
if (plane->max_width)
return plane->max_width(fb, color_plane, rotation);
@@ -1943,10 +1943,10 @@ static int intel_plane_max_width(struct intel_plane *plane,
return INT_MAX;
}
-static int intel_plane_max_height(struct intel_plane *plane,
- const struct drm_framebuffer *fb,
- int color_plane,
- unsigned int rotation)
+int intel_plane_max_height(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation)
{
if (plane->max_height)
return plane->max_height(fb, color_plane, rotation);
diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.h b/drivers/gpu/drm/i915/display/skl_universal_plane.h
--- a/drivers/gpu/drm/i915/display/skl_universal_plane.h
+++ b/drivers/gpu/drm/i915/display/skl_universal_plane.h
@@ -8,6 +8,7 @@
#include <linux/types.h>
+struct drm_framebuffer;
struct intel_crtc;
struct intel_display;
struct intel_initial_plane_config;
@@ -42,5 +43,13 @@ bool icl_is_hdr_plane(struct intel_display *display, enum plane_id plane_id);
u32 skl_plane_aux_dist(const struct intel_plane_state *plane_state,
int color_plane);
+int intel_plane_max_width(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation);
+int intel_plane_max_height(struct intel_plane *plane,
+ const struct drm_framebuffer *fb,
+ int color_plane,
+ unsigned int rotation);
#endif
Binary file not shown.
@@ -0,0 +1,67 @@
diff --git a/drivers/gpu/drm/i915/display/intel_bw.c b/drivers/gpu/drm/i915/display/intel_bw.c
--- a/drivers/gpu/drm/i915/display/intel_bw.c
+++ b/drivers/gpu/drm/i915/display/intel_bw.c
@@ -52,6 +52,8 @@
#define DEPROGBWPCLIMIT 60
+#define PEAK_BW_THRESHOLD 20000
+
struct intel_psf_gv_point {
u8 clk; /* clock in multiples of 16.6666 MHz */
};
@@ -587,6 +589,33 @@
return num_channels * (channel_width / 8) * dclk;
}
+static void xe3_add_peakbw_threshold(struct intel_display *display)
+{
+ u8 qgv_points = display->bw.max[0].num_qgv_points;
+
+ if (!HAS_PEAK_BW_THRESHOLD(display))
+ return;
+
+ if (qgv_points >= I915_NUM_QGV_POINTS) {
+ drm_dbg_kms(display->drm, "QGV points maxed out; skipping peak bandwidth threshold.\n");
+ return;
+ }
+
+ if (qgv_points <= 1)
+ return;
+
+ for (int i = 0; i < ARRAY_SIZE(display->bw.max); i++) {
+ struct intel_bw_info *bi = &display->bw.max[i];
+
+ bi->num_qgv_points++;
+ bi->peakbw[qgv_points] = PEAK_BW_THRESHOLD;
+ bi->deratedbw[qgv_points] = PEAK_BW_THRESHOLD;
+ }
+
+ drm_dbg_kms(display->drm, "An extra QGV point %d added for Peak bw threshod of %d\n",
+ qgv_points, PEAK_BW_THRESHOLD);
+}
+
static int tgl_get_bw_info(struct intel_display *display,
const struct dram_info *dram_info,
const struct intel_soc_bw_params *soc_bw_params,
@@ -684,6 +713,9 @@
}
}
+ /* For xe3 cases add an extra qgv point for Peak bw threshold */
+ xe3_add_peakbw_threshold(display);
+
/*
* In case if SAGV is disabled in BIOS, we always get 1
* SAGV point, but we can't send PCode commands to restrict it
diff --git a/drivers/gpu/drm/i915/display/intel_display_device.h b/drivers/gpu/drm/i915/display/intel_display_device.h
--- a/drivers/gpu/drm/i915/display/intel_display_device.h
+++ b/drivers/gpu/drm/i915/display/intel_display_device.h
@@ -191,6 +191,7 @@
#define HAS_MBUS_JOINING(__display) ((__display)->platform.alderlake_p || DISPLAY_VER(__display) >= 14)
#define HAS_MSO(__display) (DISPLAY_VER(__display) >= 12)
#define HAS_OVERLAY(__display) (DISPLAY_INFO(__display)->has_overlay)
+#define HAS_PEAK_BW_THRESHOLD(__display) (DISPLAY_VER(__display) >= 30)
#define HAS_PIPEDMC(__display) (DISPLAY_VER(__display) >= 12)
#define HAS_PIXEL_NORMALIZER(__display) (DISPLAY_VER(__display) >= 35)
#define HAS_PSR(__display) (DISPLAY_INFO(__display)->has_psr)
@@ -0,0 +1,95 @@
diff --git a/drivers/gpu/drm/i915/display/intel_cursor.c b/drivers/gpu/drm/i915/display/intel_cursor.c
--- a/drivers/gpu/drm/i915/display/intel_cursor.c
+++ b/drivers/gpu/drm/i915/display/intel_cursor.c
@@ -536,7 +536,8 @@ static void i9xx_cursor_disable_sel_fetch_arm(struct intel_dsb *dsb,
struct intel_display *display = to_intel_display(plane);
enum pipe pipe = plane->pipe;
- if (!crtc_state->enable_psr2_sel_fetch)
+ if (!crtc_state->enable_psr2_sel_fetch &&
+ !crtc_state->clear_psr2_sel_fetch)
return;
intel_de_write_dsb(display, dsb, SEL_FETCH_CUR_CTL(pipe), 0);
@@ -569,8 +570,10 @@ static void i9xx_cursor_update_sel_fetch_arm(struct intel_dsb *dsb,
struct intel_display *display = to_intel_display(plane);
enum pipe pipe = plane->pipe;
- if (!crtc_state->enable_psr2_sel_fetch)
+ if (!crtc_state->enable_psr2_sel_fetch) {
+ i9xx_cursor_disable_sel_fetch_arm(dsb, plane, crtc_state);
return;
+ }
if (drm_rect_height(&plane_state->psr2_sel_fetch_area) > 0) {
if (crtc_state->enable_psr2_su_region_et) {
diff --git a/drivers/gpu/drm/i915/display/intel_display_types.h b/drivers/gpu/drm/i915/display/intel_display_types.h
--- a/drivers/gpu/drm/i915/display/intel_display_types.h
+++ b/drivers/gpu/drm/i915/display/intel_display_types.h
@@ -1177,6 +1177,8 @@ struct intel_crtc_state {
bool has_sel_update;
bool enable_psr2_sel_fetch;
bool enable_psr2_su_region_et;
+ /* Drop the stale selective fetch enable bits as selective fetch is turned off */
+ bool clear_psr2_sel_fetch;
bool req_psr2_sdp_prior_scanline;
bool has_panel_replay;
bool link_off_after_as_sdp_when_pr_active;
diff --git a/drivers/gpu/drm/i915/display/intel_psr.c b/drivers/gpu/drm/i915/display/intel_psr.c
--- a/drivers/gpu/drm/i915/display/intel_psr.c
+++ b/drivers/gpu/drm/i915/display/intel_psr.c
@@ -2941,6 +2941,8 @@ int intel_psr2_sel_fetch_update(struct intel_atomic_state *state,
struct intel_crtc *crtc)
{
struct intel_display *display = to_intel_display(state);
+ const struct intel_crtc_state *old_crtc_state =
+ intel_atomic_get_old_crtc_state(state, crtc);
struct intel_crtc_state *crtc_state = intel_atomic_get_new_crtc_state(state, crtc);
struct intel_plane_state *new_plane_state, *old_plane_state;
struct intel_plane *plane;
@@ -2953,6 +2955,19 @@ int intel_psr2_sel_fetch_update(struct intel_atomic_state *state,
bool full_update = false, su_area_changed;
int i, ret;
+ /*
+ * Selective fetch is not always usable, for instance it is dropped
+ * while pipe CRC is active. The planes keep their selective fetch
+ * enable bit set in hardware over that, and a plane disabled while
+ * selective fetch is off never gets the bit cleared. Once selective
+ * fetch comes back the hardware would resume fetching for a plane that
+ * is no longer enabled and keep its DDB range reserved, so have the
+ * plane update drop the bit for every plane of the pipe as selective
+ * fetch is turned off.
+ */
+ crtc_state->clear_psr2_sel_fetch = old_crtc_state->enable_psr2_sel_fetch &&
+ !crtc_state->enable_psr2_sel_fetch;
+
if (!crtc_state->enable_psr2_sel_fetch)
return 0;
diff --git a/drivers/gpu/drm/i915/display/skl_universal_plane.c b/drivers/gpu/drm/i915/display/skl_universal_plane.c
--- a/drivers/gpu/drm/i915/display/skl_universal_plane.c
+++ b/drivers/gpu/drm/i915/display/skl_universal_plane.c
@@ -885,7 +885,8 @@ static void icl_plane_disable_sel_fetch_arm(struct intel_dsb *dsb,
struct intel_display *display = to_intel_display(plane);
enum pipe pipe = plane->pipe;
- if (!crtc_state->enable_psr2_sel_fetch)
+ if (!crtc_state->enable_psr2_sel_fetch &&
+ !crtc_state->clear_psr2_sel_fetch)
return;
intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id), 0);
@@ -1634,10 +1635,8 @@ static void icl_plane_update_sel_fetch_arm(struct intel_dsb *dsb,
struct intel_display *display = to_intel_display(plane);
enum pipe pipe = plane->pipe;
- if (!crtc_state->enable_psr2_sel_fetch)
- return;
-
- if (drm_rect_height(&plane_state->psr2_sel_fetch_area) > 0)
+ if (crtc_state->enable_psr2_sel_fetch &&
+ drm_rect_height(&plane_state->psr2_sel_fetch_area) > 0)
intel_de_write_dsb(display, dsb, SEL_FETCH_PLANE_CTL(pipe, plane->id),
SEL_FETCH_PLANE_CTL_ENABLE);
else
@@ -0,0 +1,22 @@
diff --git a/drivers/gpu/drm/i915/display/intel_dsb.c b/drivers/gpu/drm/i915/display/intel_dsb.c
--- a/drivers/gpu/drm/i915/display/intel_dsb.c
+++ b/drivers/gpu/drm/i915/display/intel_dsb.c
@@ -902,9 +902,17 @@ void intel_dsb_wait_for_delayed_vblank(struct intel_atomic_state *state,
* the hardware itself guarantees that we're SCL lines
* away from the delayed vblank, and we won't be inside
* the vmin safe window so this extra wait does nothing.
+ *
+ * Experimentally, DSB may observe a slightly stale
+ * PIPEDSL value. When the actual scanline has just reached
+ * safe_window_start, WAIT_DSL_OUT may complete immediately
+ * due to the stale value.
+ *
+ * Shift the start back by one scanline to ensure the wait
+ * window is entered reliably.
*/
intel_dsb_wait_scanline_out(state, dsb,
- intel_vrr_safe_window_start(crtc_state),
+ intel_vrr_safe_window_start(crtc_state) - 1,
intel_vrr_vmin_safe_window_end(crtc_state));
/*
* When the push is sent during vblank it will trigger
Loaded 100 of 304 files, more files were not shown because too many files have changed in this diff. Show more