125 lines
4.3 KiB
Diff
125 lines
4.3 KiB
Diff
diff --git a/include/linux/sched/sd_flags.h b/include/linux/sched/sd_flags.h
|
|
--- a/include/linux/sched/sd_flags.h
|
|
+++ b/include/linux/sched/sd_flags.h
|
|
@@ -146,8 +146,7 @@ SD_FLAG(SD_ASYM_PACKING, SDF_NEEDS_GROUPS)
|
|
/*
|
|
* Prefer to place tasks in a sibling domain
|
|
*
|
|
- * Set up until domains start spanning NUMA nodes. Close to being a SHARED_CHILD
|
|
- * flag, but cleared below domains with SD_ASYM_CPUCAPACITY.
|
|
+ * Set up until domains start spanning NUMA nodes.
|
|
*
|
|
* NEEDS_GROUPS: Load balancing flag.
|
|
*/
|
|
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
|
|
--- a/kernel/sched/fair.c
|
|
+++ b/kernel/sched/fair.c
|
|
@@ -11902,10 +11902,25 @@ static inline void update_sg_lb_stats(struct lb_env *env,
|
|
continue;
|
|
|
|
if (sd_flags & SD_ASYM_CPUCAPACITY) {
|
|
- /* Check for a misfit task on the cpu */
|
|
- if (sgs->group_misfit_task_load < rq->misfit_task_load) {
|
|
- sgs->group_misfit_task_load = rq->misfit_task_load;
|
|
- *sg_overloaded = 1;
|
|
+ if (rq->misfit_task_load) {
|
|
+ /*
|
|
+ * Always mark the root domain overloaded so big
|
|
+ * CPUs can pick up misfit tasks via newly idle
|
|
+ * balance.
|
|
+ */
|
|
+ if (balancing_at_rd)
|
|
+ *sg_overloaded = 1;
|
|
+
|
|
+ /*
|
|
+ * Only account misfit load if @dst_cpu can
|
|
+ * help; otherwise, the group may be classified
|
|
+ * as misfit_task and update_sd_pick_busiest()
|
|
+ * will skip it.
|
|
+ */
|
|
+ if (capacity_greater(capacity_of(env->dst_cpu),
|
|
+ group->sgc->max_capacity) &&
|
|
+ (sgs->group_misfit_task_load < rq->misfit_task_load))
|
|
+ sgs->group_misfit_task_load = rq->misfit_task_load;
|
|
}
|
|
} else if (env->idle && sched_reduced_capacity(rq, env->sd)) {
|
|
/* Check for a task running on a CPU with reduced capacity */
|
|
@@ -11984,6 +11999,17 @@ static bool update_sd_pick_busiest(struct lb_env *env,
|
|
sds->local_stat.group_type != group_has_spare))
|
|
return false;
|
|
|
|
+ /*
|
|
+ * Candidate sg has no more than one task per CPU and has higher
|
|
+ * per-CPU capacity. Migrating tasks to less capable CPUs may harm
|
|
+ * throughput. Maximize throughput, power/energy consequences are not
|
|
+ * considered.
|
|
+ */
|
|
+ if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
|
|
+ (sgs->group_type <= group_fully_busy) &&
|
|
+ (capacity_greater(sg->sgc->min_capacity, capacity_of(env->dst_cpu))))
|
|
+ return false;
|
|
+
|
|
if (sgs->group_type > busiest->group_type)
|
|
return true;
|
|
|
|
@@ -12090,17 +12116,6 @@ static bool update_sd_pick_busiest(struct lb_env *env,
|
|
break;
|
|
}
|
|
|
|
- /*
|
|
- * Candidate sg has no more than one task per CPU and has higher
|
|
- * per-CPU capacity. Migrating tasks to less capable CPUs may harm
|
|
- * throughput. Maximize throughput, power/energy consequences are not
|
|
- * considered.
|
|
- */
|
|
- if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
|
|
- (sgs->group_type <= group_fully_busy) &&
|
|
- (capacity_greater(sg->sgc->min_capacity, capacity_of(env->dst_cpu))))
|
|
- return false;
|
|
-
|
|
return true;
|
|
}
|
|
|
|
@@ -13018,9 +13033,24 @@ static struct rq *sched_balance_find_src_rq(struct lb_env *env,
|
|
* average load.
|
|
*/
|
|
if (env->sd->flags & SD_ASYM_CPUCAPACITY &&
|
|
- !capacity_greater(capacity_of(env->dst_cpu), capacity) &&
|
|
- nr_running == 1)
|
|
- continue;
|
|
+ nr_running == 1) {
|
|
+ bool cluster_equal_cap = static_branch_unlikely(&sched_cluster_active) &&
|
|
+ (get_actual_cpu_capacity(env->dst_cpu) ==
|
|
+ get_actual_cpu_capacity(i));
|
|
+ bool smt_degraded_cap = sched_smt_active() && !is_core_idle(i);
|
|
+
|
|
+ /*
|
|
+ * Busy SMT siblings reduce the capacity of CPU @i. Do
|
|
+ * not skip it in this case.
|
|
+ *
|
|
+ * CONFIG_SCHED_CLUSTER requires balancing load across
|
|
+ * clusters of identical capacity, accounting for
|
|
+ * hardware and cpufreq pressure.
|
|
+ */
|
|
+ if (!smt_degraded_cap && !cluster_equal_cap &&
|
|
+ !capacity_greater(capacity_of(env->dst_cpu), capacity))
|
|
+ continue;
|
|
+ }
|
|
|
|
/*
|
|
* Make sure we only pull tasks from a CPU of lower priority
|
|
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
|
|
--- a/kernel/sched/topology.c
|
|
+++ b/kernel/sched/topology.c
|
|
@@ -1995,10 +1995,6 @@ sd_init(struct sched_domain_topology_level *tl,
|
|
/*
|
|
* Convert topological properties into behaviour.
|
|
*/
|
|
- /* Don't attempt to spread across CPUs of different capacities. */
|
|
- if ((sd->flags & SD_ASYM_CPUCAPACITY) && sd->child)
|
|
- sd->child->flags &= ~SD_PREFER_SIBLING;
|
|
-
|
|
if (sd->flags & SD_SHARE_CPUCAPACITY) {
|
|
sd->imbalance_pct = 110;
|
|
|