mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-09-18 22:19:30 +02:00
Merge tag 'sched_ext-for-7.3-rc3-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext
Pull sched_ext fixes from Tejun Heo: - An error raised by a BPF program before the scheduler finished enabling was consumed by the disable path's pre-enable shortcut, leaving a running scheduler that couldn't be disabled and was later freed while in use. - Two compat kfuncs dereferenced a NULL scheduler when handed an exited or idle task, oopsing the kernel. - Keep-running decisions in the dispatch path used the root scheduler's flags for tasks belonging to a sub-scheduler, causing warnings and stalls. - Schedulers with their own CPU ID mapping had no way to learn which IDs are online. Add a kernel-maintained online mask to plug the hole. - Cgroup idle state: the initial cpu.idle state wasn't passed on cgroup init and same-value rewrites delivered spurious callbacks. - Example scheduler fixes for a reenqueue loop on attach, placements on CPUs without effective grants, stalled partition work and stale idle tracking. * tag 'sched_ext-for-7.3-rc3-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext: sched_ext: Maintain an online cid mask in the scheduler arena sched_ext: scx_qmap: Restore unused idle claims from ops.dispatch() sched_ext: Close the pre-enable ops error claim window sched_ext: scx_qmap: Fix pending partition work handoff sched_ext: scx_qmap: Place only on cids whose caps are in effect sched_ext: scx_qmap: Do not add IMMED to rescue inserts sched_ext: Use @prev's scheduler for the keep decisions in dispatch_one() sched_ext: Rename sch to root_sch in dispatch_one() sched_ext: Fix NULL sched deref in kfunc sub-sched error paths sched_ext: Don't deliver duplicate ops.cgroup_set_idle() for same value sched_ext: Pass the initial cpu.idle state in scx_cgroup_init_args
This commit is contained in:
+111
-31
@@ -2919,7 +2919,7 @@ static inline void maybe_queue_balance_callback(struct rq *rq)
|
|||||||
|
|
||||||
static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev)
|
static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev)
|
||||||
{
|
{
|
||||||
struct scx_sched *sch = scx_root_protected_live();
|
struct scx_sched *root_sch = scx_root_protected_live();
|
||||||
enum scx_dsp_verdict verdict;
|
enum scx_dsp_verdict verdict;
|
||||||
s32 cpu = cpu_of(rq);
|
s32 cpu = cpu_of(rq);
|
||||||
|
|
||||||
@@ -2928,7 +2928,7 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev
|
|||||||
|
|
||||||
scx_process_sync_ecaps(rq, prev);
|
scx_process_sync_ecaps(rq, prev);
|
||||||
|
|
||||||
if ((sch->ops.flags & SCX_OPS_HAS_CPU_PREEMPT) &&
|
if ((root_sch->ops.flags & SCX_OPS_HAS_CPU_PREEMPT) &&
|
||||||
unlikely(rq->scx.cpu_released)) {
|
unlikely(rq->scx.cpu_released)) {
|
||||||
/*
|
/*
|
||||||
* If the previous sched_class for the current CPU was not SCX,
|
* If the previous sched_class for the current CPU was not SCX,
|
||||||
@@ -2936,8 +2936,8 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev
|
|||||||
* core. This callback complements ->cpu_release(), which is
|
* core. This callback complements ->cpu_release(), which is
|
||||||
* emitted in switch_class().
|
* emitted in switch_class().
|
||||||
*/
|
*/
|
||||||
if (sch->ops.cpu_acquire)
|
if (root_sch->ops.cpu_acquire)
|
||||||
SCX_CALL_OP(sch, cpu_acquire, rq, cpu, NULL);
|
SCX_CALL_OP(root_sch, cpu_acquire, rq, cpu, NULL);
|
||||||
rq->scx.cpu_released = false;
|
rq->scx.cpu_released = false;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2955,7 +2955,7 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev
|
|||||||
* test.
|
* test.
|
||||||
*/
|
*/
|
||||||
if ((prev->scx.flags & SCX_TASK_QUEUED) && prev->scx.slice &&
|
if ((prev->scx.flags & SCX_TASK_QUEUED) && prev->scx.slice &&
|
||||||
!scx_bypassing(sch, cpu)) {
|
!scx_bypassing(scx_task_sched(prev), cpu)) {
|
||||||
verdict = SCX_DSP_PREV;
|
verdict = SCX_DSP_PREV;
|
||||||
goto has_tasks;
|
goto has_tasks;
|
||||||
}
|
}
|
||||||
@@ -2967,20 +2967,25 @@ static enum scx_dsp_verdict dispatch_one(struct rq *rq, struct task_struct *prev
|
|||||||
goto has_tasks;
|
goto has_tasks;
|
||||||
}
|
}
|
||||||
|
|
||||||
verdict = scx_dispatch_sched(sch, rq, prev, false);
|
verdict = scx_dispatch_sched(root_sch, rq, prev, false);
|
||||||
if (verdict != SCX_DSP_NONE)
|
if (verdict != SCX_DSP_NONE)
|
||||||
goto has_tasks;
|
goto has_tasks;
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Didn't find another task to run. Keep running @prev unless
|
* Didn't find another task to run. Keep running @prev unless its own
|
||||||
* %SCX_OPS_ENQ_LAST is in effect.
|
* scheduler set %SCX_OPS_ENQ_LAST and takes the enqueue instead, see
|
||||||
|
* put_prev_task_scx(). Read the scheduler here as the dispatch above
|
||||||
|
* may have dropped the rq lock while @prev changed class or scheduler.
|
||||||
*/
|
*/
|
||||||
if ((prev->scx.flags & SCX_TASK_QUEUED) &&
|
if (prev->scx.flags & SCX_TASK_QUEUED) {
|
||||||
(!(sch->ops.flags & SCX_OPS_ENQ_LAST) || scx_bypassing(sch, cpu)) &&
|
struct scx_sched *prev_sch = scx_task_sched(prev);
|
||||||
scx_task_can_stay_on_cpu(rq, prev)) {
|
|
||||||
__scx_add_event(sch, SCX_EV_DISPATCH_KEEP_LAST, 1);
|
if ((!(prev_sch->ops.flags & SCX_OPS_ENQ_LAST) ||
|
||||||
verdict = SCX_DSP_PREV;
|
scx_bypassing(prev_sch, cpu)) && scx_task_can_stay_on_cpu(rq, prev)) {
|
||||||
goto has_tasks;
|
__scx_add_event(prev_sch, SCX_EV_DISPATCH_KEEP_LAST, 1);
|
||||||
|
verdict = SCX_DSP_PREV;
|
||||||
|
goto has_tasks;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
rq->scx.flags &= ~SCX_RQ_IN_DISPATCH;
|
rq->scx.flags &= ~SCX_RQ_IN_DISPATCH;
|
||||||
return SCX_DSP_NONE;
|
return SCX_DSP_NONE;
|
||||||
@@ -3665,8 +3670,20 @@ static void handle_hotplug(struct rq *rq, bool online)
|
|||||||
s16 *tbl = rcu_dereference_check(scx_cpu_to_cid_tbl,
|
s16 *tbl = rcu_dereference_check(scx_cpu_to_cid_tbl,
|
||||||
lockdep_is_cpus_held());
|
lockdep_is_cpus_held());
|
||||||
|
|
||||||
if (tbl)
|
if (tbl) {
|
||||||
|
struct scx_sched *pos;
|
||||||
|
|
||||||
cpu_or_cid = tbl[cpu];
|
cpu_or_cid = tbl[cpu];
|
||||||
|
|
||||||
|
guard(raw_spinlock_irqsave)(&scx_sched_lock);
|
||||||
|
list_for_each_entry(pos, &scx_sched_all, all) {
|
||||||
|
struct scx_cmask *mask = pos->online_cmask;
|
||||||
|
|
||||||
|
if (mask)
|
||||||
|
__assign_bit(cpu_or_cid, (unsigned long *)mask->bits,
|
||||||
|
online);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (online && SCX_HAS_OP(sch, cpu_online))
|
if (online && SCX_HAS_OP(sch, cpu_online))
|
||||||
@@ -4766,7 +4783,8 @@ int scx_tg_online(struct task_group *tg)
|
|||||||
{ .weight = tg->scx.weight,
|
{ .weight = tg->scx.weight,
|
||||||
.bw_period_us = tg->scx.bw_period_us,
|
.bw_period_us = tg->scx.bw_period_us,
|
||||||
.bw_quota_us = tg->scx.bw_quota_us,
|
.bw_quota_us = tg->scx.bw_quota_us,
|
||||||
.bw_burst_us = tg->scx.bw_burst_us };
|
.bw_burst_us = tg->scx.bw_burst_us,
|
||||||
|
.sched_idle = tg->scx.idle };
|
||||||
|
|
||||||
ret = SCX_CALL_OP_RET(sch, cgroup_init,
|
ret = SCX_CALL_OP_RET(sch, cgroup_init,
|
||||||
NULL, tg->css.cgroup, &args);
|
NULL, tg->css.cgroup, &args);
|
||||||
@@ -4932,7 +4950,8 @@ void scx_group_set_idle(struct task_group *tg, bool idle)
|
|||||||
percpu_down_read(&scx_cgroup_ops_rwsem);
|
percpu_down_read(&scx_cgroup_ops_rwsem);
|
||||||
sch = scx_tg_knob_sched(tg);
|
sch = scx_tg_knob_sched(tg);
|
||||||
|
|
||||||
if (scx_cgroup_enabled && sch && SCX_HAS_OP(sch, cgroup_set_idle))
|
if (scx_cgroup_enabled && sch && SCX_HAS_OP(sch, cgroup_set_idle) &&
|
||||||
|
tg->scx.idle != idle)
|
||||||
SCX_CALL_OP(sch, cgroup_set_idle, NULL, tg_cgrp(tg), idle);
|
SCX_CALL_OP(sch, cgroup_set_idle, NULL, tg_cgrp(tg), idle);
|
||||||
|
|
||||||
/* Update the task group's idle state */
|
/* Update the task group's idle state */
|
||||||
@@ -5187,6 +5206,7 @@ static int scx_cgroup_init(struct scx_sched *sch)
|
|||||||
.bw_period_us = tg->scx.bw_period_us,
|
.bw_period_us = tg->scx.bw_period_us,
|
||||||
.bw_quota_us = tg->scx.bw_quota_us,
|
.bw_quota_us = tg->scx.bw_quota_us,
|
||||||
.bw_burst_us = tg->scx.bw_burst_us,
|
.bw_burst_us = tg->scx.bw_burst_us,
|
||||||
|
.sched_idle = tg->scx.idle,
|
||||||
};
|
};
|
||||||
|
|
||||||
ret = SCX_CALL_OP_RET(sch, cgroup_init, NULL, css->cgroup, &args);
|
ret = SCX_CALL_OP_RET(sch, cgroup_init, NULL, css->cgroup, &args);
|
||||||
@@ -5272,12 +5292,17 @@ static void free_exit_info(struct scx_exit_info *ei);
|
|||||||
static const char *scx_exit_reason(enum scx_exit_kind kind);
|
static const char *scx_exit_reason(enum scx_exit_kind kind);
|
||||||
static bool scx_claim_exit(struct scx_sched *sch, enum scx_exit_kind kind);
|
static bool scx_claim_exit(struct scx_sched *sch, enum scx_exit_kind kind);
|
||||||
|
|
||||||
s32 scx_set_cmask_scratch_alloc(struct scx_sched *sch)
|
s32 scx_alloc_kern_arena_objs(struct scx_sched *sch)
|
||||||
{
|
{
|
||||||
size_t size = struct_size_t(struct scx_cmask, bits,
|
size_t size = struct_size_t(struct scx_cmask, bits,
|
||||||
SCX_CMASK_NR_WORDS(num_possible_cpus()));
|
SCX_CMASK_NR_WORDS(num_possible_cpus()));
|
||||||
|
struct scx_cmask *online;
|
||||||
|
struct scx_cmask_ref ref;
|
||||||
int cpu;
|
int cpu;
|
||||||
|
|
||||||
|
/* hotplug stays excluded until the online mask is published */
|
||||||
|
lockdep_assert_cpus_held();
|
||||||
|
|
||||||
if (!sch->is_cid_type || !sch->arena_pool)
|
if (!sch->is_cid_type || !sch->arena_pool)
|
||||||
return 0;
|
return 0;
|
||||||
|
|
||||||
@@ -5293,15 +5318,28 @@ s32 scx_set_cmask_scratch_alloc(struct scx_sched *sch)
|
|||||||
return -ENOMEM;
|
return -ENOMEM;
|
||||||
scx_cmask_init(*slot, 0, num_possible_cpus());
|
scx_cmask_init(*slot, 0, num_possible_cpus());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* pack the online mask alongside the scratch masks */
|
||||||
|
online = scx_arena_alloc(sch, size);
|
||||||
|
if (!online)
|
||||||
|
return -ENOMEM;
|
||||||
|
|
||||||
|
scoped_guard(rcu) {
|
||||||
|
scx_cmask_ref_init_kern(sch, online, 0, num_possible_cpus(), &ref);
|
||||||
|
scx_cmask_ref_from_cpumask(&ref, cpu_active_mask);
|
||||||
|
}
|
||||||
|
sch->online_cmask = online;
|
||||||
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void scx_set_cmask_scratch_free(struct scx_sched *sch)
|
static void scx_free_kern_arena_objs(struct scx_sched *sch)
|
||||||
{
|
{
|
||||||
size_t size = struct_size_t(struct scx_cmask, bits,
|
size_t size = struct_size_t(struct scx_cmask, bits,
|
||||||
SCX_CMASK_NR_WORDS(num_possible_cpus()));
|
SCX_CMASK_NR_WORDS(num_possible_cpus()));
|
||||||
int cpu;
|
int cpu;
|
||||||
|
|
||||||
|
scx_arena_free(sch, sch->online_cmask, size);
|
||||||
if (!sch->set_cmask_scratch)
|
if (!sch->set_cmask_scratch)
|
||||||
return;
|
return;
|
||||||
|
|
||||||
@@ -5388,7 +5426,7 @@ static void scx_sched_free_rcu_work(struct work_struct *work)
|
|||||||
|
|
||||||
rhashtable_free_and_destroy(&sch->dsq_hash, NULL, NULL);
|
rhashtable_free_and_destroy(&sch->dsq_hash, NULL, NULL);
|
||||||
free_exit_info(sch->exit_info);
|
free_exit_info(sch->exit_info);
|
||||||
scx_set_cmask_scratch_free(sch);
|
scx_free_kern_arena_objs(sch);
|
||||||
scx_arena_pool_destroy(sch);
|
scx_arena_pool_destroy(sch);
|
||||||
if (sch->arena_map)
|
if (sch->arena_map)
|
||||||
bpf_map_put(sch->arena_map);
|
bpf_map_put(sch->arena_map);
|
||||||
@@ -7508,22 +7546,24 @@ static void scx_root_enable_workfn(struct kthread_work *work)
|
|||||||
#ifdef CONFIG_EXT_SUB_SCHED
|
#ifdef CONFIG_EXT_SUB_SCHED
|
||||||
cgroup_get(cgrp);
|
cgroup_get(cgrp);
|
||||||
#endif
|
#endif
|
||||||
|
/*
|
||||||
|
* Transition to ENABLING to arm the disable path. Allocation failure
|
||||||
|
* still unwinds locally. Full disabling on failure applies only after
|
||||||
|
* scx_alloc_and_add_sched() succeeds.
|
||||||
|
*/
|
||||||
|
WARN_ON_ONCE(scx_set_enable_state(SCX_ENABLING) != SCX_DISABLED);
|
||||||
|
WARN_ON_ONCE(scx_root);
|
||||||
|
|
||||||
sch = scx_alloc_and_add_sched(cmd, cgrp, NULL);
|
sch = scx_alloc_and_add_sched(cmd, cgrp, NULL);
|
||||||
if (IS_ERR(sch)) {
|
if (IS_ERR(sch)) {
|
||||||
ret = PTR_ERR(sch);
|
ret = PTR_ERR(sch);
|
||||||
|
WARN_ON_ONCE(scx_set_enable_state(SCX_DISABLED) != SCX_ENABLING);
|
||||||
goto err_free_tid_hash;
|
goto err_free_tid_hash;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (sch->is_cid_type)
|
if (sch->is_cid_type)
|
||||||
static_branch_enable(&__scx_is_cid_type);
|
static_branch_enable(&__scx_is_cid_type);
|
||||||
|
|
||||||
/*
|
|
||||||
* Transition to ENABLING and clear exit info to arm the disable path.
|
|
||||||
* Failure triggers full disabling from here on.
|
|
||||||
*/
|
|
||||||
WARN_ON_ONCE(scx_set_enable_state(SCX_ENABLING) != SCX_DISABLED);
|
|
||||||
WARN_ON_ONCE(scx_root);
|
|
||||||
|
|
||||||
atomic_long_set(&scx_nr_rejected, 0);
|
atomic_long_set(&scx_nr_rejected, 0);
|
||||||
|
|
||||||
for_each_possible_cpu(cpu) {
|
for_each_possible_cpu(cpu) {
|
||||||
@@ -7591,7 +7631,7 @@ static void scx_root_enable_workfn(struct kthread_work *work)
|
|||||||
goto err_disable;
|
goto err_disable;
|
||||||
}
|
}
|
||||||
|
|
||||||
ret = scx_set_cmask_scratch_alloc(sch);
|
ret = scx_alloc_kern_arena_objs(sch);
|
||||||
if (ret) {
|
if (ret) {
|
||||||
cpus_read_unlock();
|
cpus_read_unlock();
|
||||||
goto err_disable;
|
goto err_disable;
|
||||||
@@ -8946,10 +8986,17 @@ __bpf_kfunc void scx_bpf_dsq_insert_vtime(struct task_struct *p, u64 dsq_id,
|
|||||||
#ifdef CONFIG_EXT_SUB_SCHED
|
#ifdef CONFIG_EXT_SUB_SCHED
|
||||||
/*
|
/*
|
||||||
* Disallow if any sub-scheds are attached. There is no way to tell
|
* Disallow if any sub-scheds are attached. There is no way to tell
|
||||||
* which scheduler called us, just error out @p's scheduler.
|
* which scheduler called us, so error out @p's scheduler -- read it
|
||||||
|
* under RCU as @p's locks aren't necessarily held here. @p may be a
|
||||||
|
* task past sched_ext_dead() or an idle task, in which case its
|
||||||
|
* scheduler can't be determined and there is nothing obviously wrong
|
||||||
|
* to report; just refuse the call.
|
||||||
*/
|
*/
|
||||||
if (unlikely(!list_empty(&sch->children))) {
|
if (unlikely(!list_empty(&sch->children))) {
|
||||||
scx_error(scx_task_sched(p), "__scx_bpf_dsq_insert_vtime() must be used");
|
struct scx_sched *tsch = scx_task_sched_rcu(p);
|
||||||
|
|
||||||
|
if (tsch)
|
||||||
|
scx_error(tsch, "__scx_bpf_dsq_insert_vtime() must be used");
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
@@ -10321,13 +10368,45 @@ __bpf_kfunc u32 scx_bpf_nr_cids(void)
|
|||||||
* hotplug, which lets schedulers treat [0, nr_online_cids) as the online
|
* hotplug, which lets schedulers treat [0, nr_online_cids) as the online
|
||||||
* range. Schedulers that prefer to handle hotplug without a restart should
|
* range. Schedulers that prefer to handle hotplug without a restart should
|
||||||
* install a custom mapping via scx_bpf_cid_override() and track onlining
|
* install a custom mapping via scx_bpf_cid_override() and track onlining
|
||||||
* through the ops.cid_online / ops.cid_offline callbacks.
|
* through the ops.cid_online / ops.cid_offline callbacks, starting from the
|
||||||
|
* mask scx_bpf_online_cmask() returns.
|
||||||
*/
|
*/
|
||||||
__bpf_kfunc u32 scx_bpf_nr_online_cids(void)
|
__bpf_kfunc u32 scx_bpf_nr_online_cids(void)
|
||||||
{
|
{
|
||||||
return num_online_cpus();
|
return num_online_cpus();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* scx_bpf_online_cmask - Return the online cid mask in the scheduler arena
|
||||||
|
* @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs
|
||||||
|
*
|
||||||
|
* Return a kernel-maintained cmask covering [0, scx_bpf_nr_cids()), or NULL if
|
||||||
|
* the calling program is not associated with a live cid-form scheduler or the
|
||||||
|
* mask is not allocated yet, as in ops.init_cids(). Treat the mask as read-only
|
||||||
|
* even though arena memory stays writable by the BPF scheduler. The mask
|
||||||
|
* follows the SCX hotplug notifications: a cid's bit is updated before
|
||||||
|
* ops.cid_online/offline() runs for it. The pointer is valid from ops.init()
|
||||||
|
* through ops.exit(). Root ops.init() runs with hotplug excluded. Other
|
||||||
|
* contexts can observe concurrent updates.
|
||||||
|
*/
|
||||||
|
__bpf_kfunc const void *scx_bpf_online_cmask(const struct bpf_prog_aux *aux)
|
||||||
|
{
|
||||||
|
struct scx_sched *sch;
|
||||||
|
struct scx_cmask *online;
|
||||||
|
|
||||||
|
guard(rcu)();
|
||||||
|
|
||||||
|
sch = scx_prog_sched(aux);
|
||||||
|
if (unlikely(!sch))
|
||||||
|
return NULL;
|
||||||
|
online = sch->online_cmask;
|
||||||
|
if (unlikely(!online))
|
||||||
|
return NULL;
|
||||||
|
|
||||||
|
/* BPF rebases by the low 32 bits, like __arena callback args */
|
||||||
|
return (void *)((unsigned long)online - sch->arena_kern_base);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* scx_bpf_this_cid - Return the cid of the CPU this program is running on
|
* scx_bpf_this_cid - Return the cid of the CPU this program is running on
|
||||||
*
|
*
|
||||||
@@ -10691,6 +10770,7 @@ BTF_ID_FLAGS(func, scx_bpf_nr_node_ids)
|
|||||||
BTF_ID_FLAGS(func, scx_bpf_nr_cpu_ids)
|
BTF_ID_FLAGS(func, scx_bpf_nr_cpu_ids)
|
||||||
BTF_ID_FLAGS(func, scx_bpf_nr_cids)
|
BTF_ID_FLAGS(func, scx_bpf_nr_cids)
|
||||||
BTF_ID_FLAGS(func, scx_bpf_nr_online_cids)
|
BTF_ID_FLAGS(func, scx_bpf_nr_online_cids)
|
||||||
|
BTF_ID_FLAGS(func, scx_bpf_online_cmask, KF_IMPLICIT_ARGS | KF_ARENA_RET)
|
||||||
BTF_ID_FLAGS(func, scx_bpf_this_cid)
|
BTF_ID_FLAGS(func, scx_bpf_this_cid)
|
||||||
BTF_ID_FLAGS(func, scx_bpf_get_possible_cpumask, KF_ACQUIRE)
|
BTF_ID_FLAGS(func, scx_bpf_get_possible_cpumask, KF_ACQUIRE)
|
||||||
BTF_ID_FLAGS(func, scx_bpf_get_online_cpumask, KF_ACQUIRE)
|
BTF_ID_FLAGS(func, scx_bpf_get_online_cpumask, KF_ACQUIRE)
|
||||||
|
|||||||
@@ -1142,10 +1142,17 @@ __bpf_kfunc s32 scx_bpf_select_cpu_and(struct task_struct *p, s32 prev_cpu, u64
|
|||||||
#ifdef CONFIG_EXT_SUB_SCHED
|
#ifdef CONFIG_EXT_SUB_SCHED
|
||||||
/*
|
/*
|
||||||
* Disallow if any sub-scheds are attached. There is no way to tell
|
* Disallow if any sub-scheds are attached. There is no way to tell
|
||||||
* which scheduler called us, just error out @p's scheduler.
|
* which scheduler called us, so error out @p's scheduler -- read it
|
||||||
|
* under RCU as @p's locks aren't necessarily held here. @p may be a
|
||||||
|
* task past sched_ext_dead() or an idle task, in which case its
|
||||||
|
* scheduler can't be determined and there is nothing obviously wrong
|
||||||
|
* to report; just refuse the call.
|
||||||
*/
|
*/
|
||||||
if (unlikely(!list_empty(&sch->children))) {
|
if (unlikely(!list_empty(&sch->children))) {
|
||||||
scx_error(scx_task_sched(p), "__scx_bpf_select_cpu_and() must be used");
|
struct scx_sched *tsch = scx_task_sched_rcu(p);
|
||||||
|
|
||||||
|
if (tsch)
|
||||||
|
scx_error(tsch, "__scx_bpf_select_cpu_and() must be used");
|
||||||
return -EINVAL;
|
return -EINVAL;
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -259,6 +259,9 @@ struct scx_cgroup_init_args {
|
|||||||
u64 bw_period_us;
|
u64 bw_period_us;
|
||||||
u64 bw_quota_us;
|
u64 bw_quota_us;
|
||||||
u64 bw_burst_us;
|
u64 bw_burst_us;
|
||||||
|
|
||||||
|
/* whether the cgroup is configured SCHED_IDLE via cpu.idle */
|
||||||
|
bool sched_idle;
|
||||||
};
|
};
|
||||||
|
|
||||||
enum scx_cpu_preempt_reason {
|
enum scx_cpu_preempt_reason {
|
||||||
@@ -569,6 +572,12 @@ struct sched_ext_ops {
|
|||||||
*
|
*
|
||||||
* Specify the %SCX_OPS_KEEP_BUILTIN_IDLE flag to keep the built-in idle
|
* Specify the %SCX_OPS_KEEP_BUILTIN_IDLE flag to keep the built-in idle
|
||||||
* tracking.
|
* tracking.
|
||||||
|
*
|
||||||
|
* Only actual transitions are reported. A CPU that is claimed with an
|
||||||
|
* idle pick and kicked but dispatches no task returns to idle without a
|
||||||
|
* transition. A scheduler tracking idle CPUs itself must restore the
|
||||||
|
* idle state from ops.dispatch() when it returns without the next task
|
||||||
|
* to run.
|
||||||
*/
|
*/
|
||||||
void (*update_idle)(s32 cpu, bool idle);
|
void (*update_idle)(s32 cpu, bool idle);
|
||||||
|
|
||||||
@@ -1552,6 +1561,7 @@ struct scx_sched {
|
|||||||
* and passes it to the callback's __arena argument.
|
* and passes it to the callback's __arena argument.
|
||||||
*/
|
*/
|
||||||
struct scx_cmask * __percpu *set_cmask_scratch;
|
struct scx_cmask * __percpu *set_cmask_scratch;
|
||||||
|
struct scx_cmask *online_cmask;
|
||||||
|
|
||||||
DECLARE_BITMAP(has_op, SCX_OPI_END);
|
DECLARE_BITMAP(has_op, SCX_OPI_END);
|
||||||
|
|
||||||
@@ -2078,7 +2088,7 @@ void scx_disable_and_exit_task(struct scx_sched *sch, struct task_struct *p);
|
|||||||
void scx_cgroup_lock(void);
|
void scx_cgroup_lock(void);
|
||||||
void scx_cgroup_unlock(void);
|
void scx_cgroup_unlock(void);
|
||||||
#endif
|
#endif
|
||||||
s32 scx_set_cmask_scratch_alloc(struct scx_sched *sch);
|
s32 scx_alloc_kern_arena_objs(struct scx_sched *sch);
|
||||||
void scx_disable_bypass_dsp(struct scx_sched *sch);
|
void scx_disable_bypass_dsp(struct scx_sched *sch);
|
||||||
void scx_bypass(struct scx_sched *sch, bool bypass);
|
void scx_bypass(struct scx_sched *sch, bool bypass);
|
||||||
s32 scx_link_sched(struct scx_sched *sch);
|
s32 scx_link_sched(struct scx_sched *sch);
|
||||||
|
|||||||
@@ -1361,6 +1361,7 @@ static s32 scx_cgroup_claim_subtree(struct scx_sched *sch)
|
|||||||
.bw_period_us = tg->scx.bw_period_us,
|
.bw_period_us = tg->scx.bw_period_us,
|
||||||
.bw_quota_us = tg->scx.bw_quota_us,
|
.bw_quota_us = tg->scx.bw_quota_us,
|
||||||
.bw_burst_us = tg->scx.bw_burst_us,
|
.bw_burst_us = tg->scx.bw_burst_us,
|
||||||
|
.sched_idle = tg->scx.idle,
|
||||||
};
|
};
|
||||||
|
|
||||||
if (tg->scx.sched != parent ||
|
if (tg->scx.sched != parent ||
|
||||||
@@ -1464,6 +1465,7 @@ static void scx_cgroup_return_subtree(struct scx_sched *sch)
|
|||||||
.bw_period_us = tg->scx.bw_period_us,
|
.bw_period_us = tg->scx.bw_period_us,
|
||||||
.bw_quota_us = tg->scx.bw_quota_us,
|
.bw_quota_us = tg->scx.bw_quota_us,
|
||||||
.bw_burst_us = tg->scx.bw_burst_us,
|
.bw_burst_us = tg->scx.bw_burst_us,
|
||||||
|
.sched_idle = tg->scx.idle,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* the first pass must have transferred everything */
|
/* the first pass must have transferred everything */
|
||||||
@@ -1803,6 +1805,12 @@ void scx_sub_enable_workfn(struct kthread_work *work)
|
|||||||
goto err_disable;
|
goto err_disable;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
scoped_guard(cpus_read_lock) {
|
||||||
|
ret = scx_alloc_kern_arena_objs(sch);
|
||||||
|
if (ret)
|
||||||
|
goto err_disable;
|
||||||
|
}
|
||||||
|
|
||||||
if (sch->ops.init) {
|
if (sch->ops.init) {
|
||||||
ret = SCX_CALL_OP_RET(sch, init, NULL);
|
ret = SCX_CALL_OP_RET(sch, init, NULL);
|
||||||
if (ret) {
|
if (ret) {
|
||||||
@@ -1813,10 +1821,6 @@ void scx_sub_enable_workfn(struct kthread_work *work)
|
|||||||
sch->exit_info->flags |= SCX_EFLAG_INITIALIZED;
|
sch->exit_info->flags |= SCX_EFLAG_INITIALIZED;
|
||||||
}
|
}
|
||||||
|
|
||||||
ret = scx_set_cmask_scratch_alloc(sch);
|
|
||||||
if (ret)
|
|
||||||
goto err_disable;
|
|
||||||
|
|
||||||
struct scx_sub_attach_args sub_attach_args = {
|
struct scx_sub_attach_args sub_attach_args = {
|
||||||
.ops = &sch->ops,
|
.ops = &sch->ops,
|
||||||
.cgroup_path = sch->cgrp_path,
|
.cgroup_path = sch->cgrp_path,
|
||||||
|
|||||||
@@ -113,6 +113,7 @@ s32 scx_bpf_this_cid(void) __ksym __weak;
|
|||||||
struct task_struct *scx_bpf_cid_curr(s32 cid) __ksym __weak;
|
struct task_struct *scx_bpf_cid_curr(s32 cid) __ksym __weak;
|
||||||
u32 scx_bpf_nr_cids(void) __ksym __weak;
|
u32 scx_bpf_nr_cids(void) __ksym __weak;
|
||||||
u32 scx_bpf_nr_online_cids(void) __ksym __weak;
|
u32 scx_bpf_nr_online_cids(void) __ksym __weak;
|
||||||
|
const void __arena *scx_bpf_online_cmask(void) __ksym __weak;
|
||||||
u32 scx_bpf_cidperf_cap(s32 cid) __ksym __weak;
|
u32 scx_bpf_cidperf_cap(s32 cid) __ksym __weak;
|
||||||
u32 scx_bpf_cidperf_cur(s32 cid) __ksym __weak;
|
u32 scx_bpf_cidperf_cur(s32 cid) __ksym __weak;
|
||||||
s32 scx_bpf_cidperf_set(s32 cid, u32 perf) __ksym __weak;
|
s32 scx_bpf_cidperf_set(s32 cid, u32 perf) __ksym __weak;
|
||||||
|
|||||||
+112
-52
@@ -24,6 +24,9 @@
|
|||||||
* time-share that stays self-local.
|
* time-share that stays self-local.
|
||||||
* self - The excl cpus the node kept for itself, plus all of held_shared.
|
* self - The excl cpus the node kept for itself, plus all of held_shared.
|
||||||
* owner - Who holds a cid - a child slot, CID_SELF, or CID_NONE.
|
* owner - Who holds a cid - a child slot, CID_SELF, or CID_NONE.
|
||||||
|
* avail - Cpus whose caps are in effect, per ops.sub_ecaps_updated().
|
||||||
|
* usable - self AND avail. Placement decisions use this: self is the
|
||||||
|
* delegation split and can run ahead of what the cpus honor.
|
||||||
*
|
*
|
||||||
* The scheduler splits its held-excl cpus among self and the children in
|
* The scheduler splits its held-excl cpus among self and the children in
|
||||||
* proportion to each node's cpu.weight, handing each the floor of its share as
|
* proportion to each node's cpu.weight, handing each the floor of its share as
|
||||||
@@ -208,8 +211,8 @@ static int qmap_spin_lock(struct bpf_res_spin_lock *lock)
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Try prev_cid, then scan cpus_allowed AND idle_cids AND self_cids round-robin
|
* Try prev_cid, then scan cpus_allowed AND idle_cids AND usable_cids
|
||||||
* from prev_cid + 1. Atomic claim retries on race; bounded by
|
* round-robin from prev_cid + 1. Atomic claim retries on race; bounded by
|
||||||
* IDLE_PICK_RETRIES to keep the verifier's insn budget in check.
|
* IDLE_PICK_RETRIES to keep the verifier's insn budget in check.
|
||||||
*/
|
*/
|
||||||
#define IDLE_PICK_RETRIES 16
|
#define IDLE_PICK_RETRIES 16
|
||||||
@@ -221,7 +224,7 @@ static s32 pick_direct_dispatch_cid(struct task_struct *p, s32 prev_cid,
|
|||||||
s32 cid;
|
s32 cid;
|
||||||
u32 i;
|
u32 i;
|
||||||
|
|
||||||
if (cmask_test(prev_cid, &qa.self_cids.mask) &&
|
if (cmask_test(prev_cid, &qa.usable_cids.mask) &&
|
||||||
cmask_test_and_clear(prev_cid, &qa.idle_cids.mask))
|
cmask_test_and_clear(prev_cid, &qa.idle_cids.mask))
|
||||||
return prev_cid;
|
return prev_cid;
|
||||||
|
|
||||||
@@ -229,7 +232,7 @@ static s32 pick_direct_dispatch_cid(struct task_struct *p, s32 prev_cid,
|
|||||||
bpf_for(i, 0, IDLE_PICK_RETRIES) {
|
bpf_for(i, 0, IDLE_PICK_RETRIES) {
|
||||||
cid = cmask_next_and2_set_wrap(&taskc->cpus_allowed,
|
cid = cmask_next_and2_set_wrap(&taskc->cpus_allowed,
|
||||||
&qa.idle_cids.mask,
|
&qa.idle_cids.mask,
|
||||||
&qa.self_cids.mask, cid + 1);
|
&qa.usable_cids.mask, cid + 1);
|
||||||
barrier_var(cid);
|
barrier_var(cid);
|
||||||
if (cid >= nr_cids)
|
if (cid >= nr_cids)
|
||||||
return -1;
|
return -1;
|
||||||
@@ -358,8 +361,8 @@ s32 BPF_STRUCT_OPS(qmap_select_cid, struct task_struct *p,
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* A received time-shared cid is held ENQ_IMMED-only, so inserts must set
|
* A received time-shared cid is held ENQ_IMMED-only, so inserts meant to run
|
||||||
* SCX_ENQ_IMMED.
|
* there must set SCX_ENQ_IMMED.
|
||||||
*/
|
*/
|
||||||
static u64 needs_immed(s32 cid)
|
static u64 needs_immed(s32 cid)
|
||||||
{
|
{
|
||||||
@@ -444,9 +447,11 @@ void BPF_STRUCT_OPS(qmap_enqueue, struct task_struct *p, u64 enq_flags)
|
|||||||
* didn't grant them or we delegated them to children - would starve in
|
* didn't grant them or we delegated them to children - would starve in
|
||||||
* SHARED/FIFO since we only pull from those on self cids.
|
* SHARED/FIFO since we only pull from those on self cids.
|
||||||
*
|
*
|
||||||
* Force it onto its first allowed cid's local DSQ. If we hold that cid
|
* Force it onto its first allowed cid's local DSQ with SCX_ENQ_RESCUE.
|
||||||
* it runs. Otherwise the insert carries SCX_ENQ_RESCUE and the kernel
|
* If we hold ENQ on that cid it runs. Otherwise the kernel diverts the
|
||||||
* diverts the task to its rescue path.
|
* task to its rescue path. IMMED would turn the insert into a legal
|
||||||
|
* placement on a time-shared cid and the kernel would bounce it back
|
||||||
|
* here instead of rescuing it.
|
||||||
*/
|
*/
|
||||||
if (!cmask_intersects(&taskc->cpus_allowed, &qa.self_cids.mask)) {
|
if (!cmask_intersects(&taskc->cpus_allowed, &qa.self_cids.mask)) {
|
||||||
s32 c = cmask_next_set_wrap(&taskc->cpus_allowed, 0);
|
s32 c = cmask_next_set_wrap(&taskc->cpus_allowed, 0);
|
||||||
@@ -455,7 +460,7 @@ void BPF_STRUCT_OPS(qmap_enqueue, struct task_struct *p, u64 enq_flags)
|
|||||||
taskc->force_local = false;
|
taskc->force_local = false;
|
||||||
__sync_fetch_and_add(&qa.nr_rescue_dsp, 1);
|
__sync_fetch_and_add(&qa.nr_rescue_dsp, 1);
|
||||||
scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL_ON | c, slice_ns,
|
scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL_ON | c, slice_ns,
|
||||||
enq_flags | needs_immed(c) | SCX_ENQ_RESCUE);
|
enq_flags | SCX_ENQ_RESCUE);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -540,7 +545,7 @@ void BPF_STRUCT_OPS(qmap_enqueue, struct task_struct *p, u64 enq_flags)
|
|||||||
scx_bpf_dsq_insert(p, SHARED_DSQ, 0, enq_flags);
|
scx_bpf_dsq_insert(p, SHARED_DSQ, 0, enq_flags);
|
||||||
cid = cmask_next_and2_set_wrap(&taskc->cpus_allowed,
|
cid = cmask_next_and2_set_wrap(&taskc->cpus_allowed,
|
||||||
&qa.idle_cids.mask,
|
&qa.idle_cids.mask,
|
||||||
&qa.self_cids.mask, 0);
|
&qa.usable_cids.mask, 0);
|
||||||
if (cid < scx_bpf_nr_cids())
|
if (cid < scx_bpf_nr_cids())
|
||||||
scx_bpf_kick_cid(cid, SCX_KICK_IDLE);
|
scx_bpf_kick_cid(cid, SCX_KICK_IDLE);
|
||||||
return;
|
return;
|
||||||
@@ -618,7 +623,7 @@ static bool scan_shared_dsq(bool from_timer)
|
|||||||
if (c >= 0 && c < scx_bpf_nr_cids()) {
|
if (c >= 0 && c < scx_bpf_nr_cids()) {
|
||||||
__sync_fetch_and_add(&qa.nr_rescue_dsp, 1);
|
__sync_fetch_and_add(&qa.nr_rescue_dsp, 1);
|
||||||
scx_bpf_dsq_move(BPF_FOR_EACH_ITER, p, SCX_DSQ_LOCAL_ON | c,
|
scx_bpf_dsq_move(BPF_FOR_EACH_ITER, p, SCX_DSQ_LOCAL_ON | c,
|
||||||
needs_immed(c) | SCX_ENQ_RESCUE);
|
SCX_ENQ_RESCUE);
|
||||||
}
|
}
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -644,22 +649,27 @@ static bool scan_shared_dsq(bool from_timer)
|
|||||||
if (!(taskc = lookup_task_ctx(p)))
|
if (!(taskc = lookup_task_ctx(p)))
|
||||||
return false;
|
return false;
|
||||||
|
|
||||||
/* only run highpri tasks on cids this node holds, not delegated ones */
|
/* only run highpri tasks on cids this node can use right now */
|
||||||
if (cmask_test(this_cid, &taskc->cpus_allowed) &&
|
if (cmask_test(this_cid, &taskc->cpus_allowed) &&
|
||||||
cmask_test(this_cid, &qa.self_cids.mask))
|
cmask_test(this_cid, &qa.usable_cids.mask))
|
||||||
cid = this_cid;
|
cid = this_cid;
|
||||||
else
|
else
|
||||||
cid = cmask_next_and_set_wrap(&taskc->cpus_allowed,
|
cid = cmask_next_and_set_wrap(&taskc->cpus_allowed,
|
||||||
&qa.self_cids.mask,
|
&qa.usable_cids.mask,
|
||||||
this_cid + 1);
|
this_cid + 1);
|
||||||
if (cid >= nr_cids) {
|
if (cid >= nr_cids) {
|
||||||
/* stranded after the cull - rescue it from here */
|
s32 c;
|
||||||
s32 c = cmask_next_set_wrap(&taskc->cpus_allowed, 0);
|
|
||||||
|
|
||||||
|
/* self cids lack caps in effect yet, leave it queued */
|
||||||
|
if (cmask_intersects(&taskc->cpus_allowed, &qa.self_cids.mask))
|
||||||
|
continue;
|
||||||
|
|
||||||
|
/* stranded after the cull - rescue it from here */
|
||||||
|
c = cmask_next_set_wrap(&taskc->cpus_allowed, 0);
|
||||||
if (c >= 0 && c < nr_cids) {
|
if (c >= 0 && c < nr_cids) {
|
||||||
__sync_fetch_and_add(&qa.nr_rescue_dsp, 1);
|
__sync_fetch_and_add(&qa.nr_rescue_dsp, 1);
|
||||||
scx_bpf_dsq_move(BPF_FOR_EACH_ITER, p, SCX_DSQ_LOCAL_ON | c,
|
scx_bpf_dsq_move(BPF_FOR_EACH_ITER, p, SCX_DSQ_LOCAL_ON | c,
|
||||||
needs_immed(c) | SCX_ENQ_RESCUE);
|
SCX_ENQ_RESCUE);
|
||||||
}
|
}
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -808,10 +818,10 @@ void BPF_STRUCT_OPS(qmap_dispatch, s32 cid, struct task_struct *prev)
|
|||||||
batch--;
|
batch--;
|
||||||
cpuc->dsp_cnt--;
|
cpuc->dsp_cnt--;
|
||||||
if (!batch || !scx_bpf_dispatch_nr_slots()) {
|
if (!batch || !scx_bpf_dispatch_nr_slots()) {
|
||||||
if (scan_shared_dsq(false))
|
if (scan_shared_dsq(false) ||
|
||||||
|
scx_bpf_dsq_move_to_local(SHARED_DSQ, needs_immed(cid)))
|
||||||
return;
|
return;
|
||||||
scx_bpf_dsq_move_to_local(SHARED_DSQ, needs_immed(cid));
|
goto prev;
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
if (!cpuc->dsp_cnt)
|
if (!cpuc->dsp_cnt)
|
||||||
break;
|
break;
|
||||||
@@ -822,10 +832,14 @@ void BPF_STRUCT_OPS(qmap_dispatch, s32 cid, struct task_struct *prev)
|
|||||||
|
|
||||||
if (scan_shared_dsq(false))
|
if (scan_shared_dsq(false))
|
||||||
return;
|
return;
|
||||||
|
prev:
|
||||||
/*
|
/*
|
||||||
* No other tasks. @prev will keep running. Update its core_sched_seq as
|
* No other tasks. @prev will keep running. Update its core_sched_seq as
|
||||||
* if the task were enqueued and dispatched immediately.
|
* if the task were enqueued and dispatched immediately.
|
||||||
|
*
|
||||||
|
* No @prev to keep running means the CPU goes idle. If its claim was
|
||||||
|
* never used, that is not a transition and ops.update_idle() stays
|
||||||
|
* silent. Restore the claim here.
|
||||||
*/
|
*/
|
||||||
if (prev) {
|
if (prev) {
|
||||||
taskc = lookup_task_ctx(prev);
|
taskc = lookup_task_ctx(prev);
|
||||||
@@ -834,6 +848,8 @@ void BPF_STRUCT_OPS(qmap_dispatch, s32 cid, struct task_struct *prev)
|
|||||||
|
|
||||||
taskc->core_sched_seq =
|
taskc->core_sched_seq =
|
||||||
qa.core_sched_tail_seqs[weight_to_idx(prev->scx.weight)]++;
|
qa.core_sched_tail_seqs[weight_to_idx(prev->scx.weight)]++;
|
||||||
|
} else {
|
||||||
|
cmask_set(cid, &qa.idle_cids.mask);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1113,7 +1129,7 @@ void BPF_STRUCT_OPS(qmap_update_idle, s32 cid, bool idle)
|
|||||||
/*
|
/*
|
||||||
* The kernel delivers update_idle() for every cid this node holds
|
* The kernel delivers update_idle() for every cid this node holds
|
||||||
* SCX_CAP_BASE on. Track every cid's idle state regardless of
|
* SCX_CAP_BASE on. Track every cid's idle state regardless of
|
||||||
* delegation: the direct-dispatch pick masks idle_cids with self_cids
|
* delegation: the direct-dispatch pick masks idle_cids with usable_cids
|
||||||
* at selection, so a cid already idle when it returns to self needs no
|
* at selection, so a cid already idle when it returns to self needs no
|
||||||
* reseed here.
|
* reseed here.
|
||||||
*/
|
*/
|
||||||
@@ -1285,11 +1301,16 @@ struct {
|
|||||||
__type(value, struct round_robin_timer);
|
__type(value, struct round_robin_timer);
|
||||||
} round_robin_timer SEC(".maps");
|
} round_robin_timer SEC(".maps");
|
||||||
|
|
||||||
|
enum part_pending_flags {
|
||||||
|
PART_REFRESH = BIT_U64(0),
|
||||||
|
PART_REDISTRIBUTE = BIT_U64(1),
|
||||||
|
};
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Partition update synchronization. qa.part can be written from concurrent
|
* Partition update synchronization. qa.part can be written from concurrent
|
||||||
* contexts. This single-runner guard admits one writer at a time without
|
* contexts. This single-runner guard admits one writer at a time without
|
||||||
* holding a lock across the grant/revoke kfuncs. part_pending coalesces
|
* holding a lock across the grant/revoke kfuncs. part_pending coalesces
|
||||||
* repartition requests that arrive while it is held.
|
* refresh and repartition requests that arrive while it is held.
|
||||||
*
|
*
|
||||||
* They live in .bss, not the arena: rr_advance() runs from a bpf_timer
|
* They live in .bss, not the arena: rr_advance() runs from a bpf_timer
|
||||||
* callback, where the verifier rejects atomic ops on arena memory.
|
* callback, where the verifier rejects atomic ops on arena memory.
|
||||||
@@ -1537,6 +1558,19 @@ static __noinline void account_alloc(void)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* usable_cids = self_cids & avail_cids. The inputs have separate writers,
|
||||||
|
* apply_partition() and qmap_sub_ecaps_updated(), so the result is rebuilt in
|
||||||
|
* full under the partition guard, in scratch first so that readers never see
|
||||||
|
* self_cids alone.
|
||||||
|
*/
|
||||||
|
static void refresh_usable(void)
|
||||||
|
{
|
||||||
|
cmask_copy(&qa.usable_scratch.mask, &qa.self_cids.mask);
|
||||||
|
cmask_and(&qa.usable_scratch.mask, &qa.avail_cids.mask);
|
||||||
|
cmask_copy(&qa.usable_cids.mask, &qa.usable_scratch.mask);
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* apply_partition - execute the plan compute_partition() built
|
* apply_partition - execute the plan compute_partition() built
|
||||||
*
|
*
|
||||||
@@ -1559,6 +1593,7 @@ __noinline void apply_partition(void)
|
|||||||
/* no excl cpu: run own tasks on the held shares, evict children */
|
/* no excl cpu: run own tasks on the held shares, evict children */
|
||||||
if (!qa.part.nr_excl) {
|
if (!qa.part.nr_excl) {
|
||||||
cmask_copy(&qa.self_cids.mask, &qa.held_shared.mask);
|
cmask_copy(&qa.self_cids.mask, &qa.held_shared.mask);
|
||||||
|
refresh_usable();
|
||||||
bpf_for(i, 0, MAX_SUB_SCHEDS)
|
bpf_for(i, 0, MAX_SUB_SCHEDS)
|
||||||
if (qa.sub_sched_ctxs[i].cgroup_id)
|
if (qa.sub_sched_ctxs[i].cgroup_id)
|
||||||
scx_bpf_sub_kill(qa.sub_sched_ctxs[i].cgroup_id,
|
scx_bpf_sub_kill(qa.sub_sched_ctxs[i].cgroup_id,
|
||||||
@@ -1596,6 +1631,7 @@ __noinline void apply_partition(void)
|
|||||||
else if (o == CID_SELF)
|
else if (o == CID_SELF)
|
||||||
cmask_set(cid, &qa.self_cids.mask);
|
cmask_set(cid, &qa.self_cids.mask);
|
||||||
}
|
}
|
||||||
|
refresh_usable();
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Apply each child's exclusive cids as a delta against its previous
|
* Apply each child's exclusive cids as a delta against its previous
|
||||||
@@ -1643,33 +1679,46 @@ __noinline void apply_partition(void)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/**
|
||||||
* Recompute the split off the node's held caps and apply it. The contexts this
|
* execute_partition - Run pending partition updates
|
||||||
* runs from (the sub-sched and cgroup callbacks, the rr timer) are not
|
*
|
||||||
* serialized by the kernel, so a single runner does the work. A caller that
|
* The rr timer is the backstop if the loop reaches its iteration limit.
|
||||||
* finds the guard held leaves part_pending set; the holder drains it before
|
|
||||||
* releasing, with the rr timer as a backstop.
|
|
||||||
*/
|
*/
|
||||||
static void redistribute(void)
|
static void execute_partition(void)
|
||||||
{
|
{
|
||||||
|
u64 pending;
|
||||||
s32 i;
|
s32 i;
|
||||||
|
|
||||||
__sync_fetch_and_or(&part_pending, 1);
|
|
||||||
|
|
||||||
if (!part_try_start())
|
|
||||||
return;
|
|
||||||
|
|
||||||
bpf_for(i, 0, 1024) {
|
bpf_for(i, 0, 1024) {
|
||||||
__sync_fetch_and_and(&part_pending, 0);
|
if (!part_try_start())
|
||||||
/* charge elapsed time to the current partition before rebuilding it */
|
break;
|
||||||
account_alloc();
|
|
||||||
compute_partition();
|
pending = __sync_fetch_and_and(&part_pending, 0);
|
||||||
apply_partition();
|
if (pending & PART_REDISTRIBUTE) {
|
||||||
|
/* charge elapsed time before repartitioning */
|
||||||
|
account_alloc();
|
||||||
|
compute_partition();
|
||||||
|
apply_partition();
|
||||||
|
} else if (pending & PART_REFRESH) {
|
||||||
|
refresh_usable();
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Requests are published before trying the guard. Releasing it
|
||||||
|
* before checking pending work ensures a racing request is
|
||||||
|
* either observed here or handled by a caller that acquires the
|
||||||
|
* guard.
|
||||||
|
*/
|
||||||
|
part_end();
|
||||||
if (!__sync_fetch_and_or(&part_pending, 0))
|
if (!__sync_fetch_and_or(&part_pending, 0))
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
part_end();
|
static void redistribute(void)
|
||||||
|
{
|
||||||
|
__sync_fetch_and_or(&part_pending, PART_REDISTRIBUTE);
|
||||||
|
execute_partition();
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
@@ -1683,6 +1732,7 @@ int flush_alloc(void *ctx)
|
|||||||
if (part_try_start()) {
|
if (part_try_start()) {
|
||||||
account_alloc();
|
account_alloc();
|
||||||
part_end();
|
part_end();
|
||||||
|
execute_partition();
|
||||||
}
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
@@ -1740,9 +1790,7 @@ static void rr_advance(void)
|
|||||||
|
|
||||||
part_end();
|
part_end();
|
||||||
|
|
||||||
/* a resplit queued while we held the guard supersedes this rotation */
|
execute_partition();
|
||||||
if (__sync_fetch_and_or(&part_pending, 0))
|
|
||||||
redistribute();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/* advance the time-shared cid pool every round_robin_ns */
|
/* advance the time-shared cid pool every round_robin_ns */
|
||||||
@@ -1837,8 +1885,11 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(qmap_init)
|
|||||||
cmask_init(&qa.rr_cids.mask, 0, nr_cids);
|
cmask_init(&qa.rr_cids.mask, 0, nr_cids);
|
||||||
cmask_init(&qa.prev_rr_cids.mask, 0, nr_cids);
|
cmask_init(&qa.prev_rr_cids.mask, 0, nr_cids);
|
||||||
cmask_init(&qa.self_cids.mask, 0, nr_cids);
|
cmask_init(&qa.self_cids.mask, 0, nr_cids);
|
||||||
|
cmask_init(&qa.avail_cids.mask, 0, nr_cids);
|
||||||
|
cmask_init(&qa.usable_cids.mask, 0, nr_cids);
|
||||||
cmask_init(&qa.to_revoke_cids.mask, 0, nr_cids);
|
cmask_init(&qa.to_revoke_cids.mask, 0, nr_cids);
|
||||||
cmask_init(&qa.to_grant_cids.mask, 0, nr_cids);
|
cmask_init(&qa.to_grant_cids.mask, 0, nr_cids);
|
||||||
|
cmask_init(&qa.usable_scratch.mask, 0, nr_cids);
|
||||||
cmask_init(&qa.held_excl.mask, 0, nr_cids);
|
cmask_init(&qa.held_excl.mask, 0, nr_cids);
|
||||||
cmask_init(&qa.held_shared.mask, 0, nr_cids);
|
cmask_init(&qa.held_shared.mask, 0, nr_cids);
|
||||||
|
|
||||||
@@ -1852,14 +1903,16 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(qmap_init)
|
|||||||
}
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* The root starts holding every cid. qmap_sub_ecaps_updated() maintains
|
* The root starts holding every cid and gets no ecaps notifications, so
|
||||||
* per-cid shared state as effective caps settle, and redistribute()
|
* its avail set is fixed here. qmap_sub_ecaps_updated() maintains the
|
||||||
* rebuilds owner and self from held caps. A non-root node starts with
|
* per-cid state as effective caps settle, and redistribute() rebuilds
|
||||||
* nothing.
|
* owner and self from held caps. A non-root node starts with nothing.
|
||||||
*/
|
*/
|
||||||
bpf_for(i, 0, nr_cids) {
|
bpf_for(i, 0, nr_cids) {
|
||||||
if (!sub_cgroup_id) {
|
if (!sub_cgroup_id) {
|
||||||
cmask_set(i, &qa.self_cids.mask);
|
cmask_set(i, &qa.self_cids.mask);
|
||||||
|
cmask_set(i, &qa.avail_cids.mask);
|
||||||
|
cmask_set(i, &qa.usable_cids.mask);
|
||||||
qa.part.cid_owner[i] = CID_SELF;
|
qa.part.cid_owner[i] = CID_SELF;
|
||||||
} else {
|
} else {
|
||||||
qa.part.cid_owner[i] = CID_NONE;
|
qa.part.cid_owner[i] = CID_NONE;
|
||||||
@@ -2000,12 +2053,19 @@ void BPF_STRUCT_OPS(qmap_sub_ecaps_updated, s32 cid, u64 before, u64 after)
|
|||||||
{
|
{
|
||||||
/*
|
/*
|
||||||
* Effective caps updated. Track which cids hold shared caps so a self
|
* Effective caps updated. Track which cids hold shared caps so a self
|
||||||
* task placed there enqueues IMMED.
|
* task placed there enqueues IMMED, and which cids have ENQ_IMMED in
|
||||||
|
* effect at all (avail, see the header comment).
|
||||||
*/
|
*/
|
||||||
if (after & SCX_CAP_ENQ_IMMED)
|
if (after & SCX_CAP_ENQ_IMMED) {
|
||||||
qa.cid_shared[cid] = (after & SCX_CAP_ENQ) ? 0 : 1;
|
qa.cid_shared[cid] = (after & SCX_CAP_ENQ) ? 0 : 1;
|
||||||
else
|
cmask_set(cid, &qa.avail_cids.mask);
|
||||||
|
} else {
|
||||||
qa.cid_shared[cid] = 0;
|
qa.cid_shared[cid] = 0;
|
||||||
|
cmask_clear(cid, &qa.avail_cids.mask);
|
||||||
|
}
|
||||||
|
|
||||||
|
__sync_fetch_and_or(&part_pending, PART_REFRESH);
|
||||||
|
execute_partition();
|
||||||
}
|
}
|
||||||
|
|
||||||
SCX_OPS_CID_DEFINE(qmap_ops,
|
SCX_OPS_CID_DEFINE(qmap_ops,
|
||||||
|
|||||||
@@ -165,12 +165,15 @@ struct qmap_arena {
|
|||||||
|
|
||||||
/* bpf-internal cmasks (embedded, see struct qmap_cmask) */
|
/* bpf-internal cmasks (embedded, see struct qmap_cmask) */
|
||||||
struct qmap_cmask self_cids; /* cids this node runs its own tasks on */
|
struct qmap_cmask self_cids; /* cids this node runs its own tasks on */
|
||||||
|
struct qmap_cmask avail_cids; /* cids with caps in effect on the cpu */
|
||||||
|
struct qmap_cmask usable_cids; /* self_cids & avail_cids, placeable right now */
|
||||||
struct qmap_cmask idle_cids; /* idle state of all cids regardless of delegation */
|
struct qmap_cmask idle_cids; /* idle state of all cids regardless of delegation */
|
||||||
struct qmap_cmask rr_cids; /* the shared pool, as a mask for grant/revoke */
|
struct qmap_cmask rr_cids; /* the shared pool, as a mask for grant/revoke */
|
||||||
|
|
||||||
/* scratch cmasks */
|
/* scratch cmasks */
|
||||||
struct qmap_cmask to_revoke_cids; /* delta cids to revoke */
|
struct qmap_cmask to_revoke_cids; /* delta cids to revoke */
|
||||||
struct qmap_cmask to_grant_cids; /* delta cids to grant */
|
struct qmap_cmask to_grant_cids; /* delta cids to grant */
|
||||||
|
struct qmap_cmask usable_scratch; /* refresh_usable() build area */
|
||||||
struct qmap_cmask prev_rr_cids; /* previous shared pool, to clear stale grants */
|
struct qmap_cmask prev_rr_cids; /* previous shared pool, to clear stale grants */
|
||||||
struct qmap_cmask held_excl; /* cids held excl (ENQ): delegatable */
|
struct qmap_cmask held_excl; /* cids held excl (ENQ): delegatable */
|
||||||
struct qmap_cmask held_shared; /* cids held shared (ENQ_IMMED only): self-local */
|
struct qmap_cmask held_shared; /* cids held shared (ENQ_IMMED only): self-local */
|
||||||
|
|||||||
Reference in New Issue
Block a user