Merge branch 'for-7.3' into for-7.3-arena-args

This commit is contained in:
Tejun Heo
2026-08-12 09:55:20 -10:00
9 changed files with 271 additions and 101 deletions

View File

@@ -443,6 +443,17 @@ static void __sched_core_flip(bool enabled)
sched_core_lock(cpu, &flags);
/*
* A core-wide selection may have the shared rq lock temporarily
* released by a lock-dropping ->pick_task(). Flipping would
* rebind rq_lockp() under it. Wait it out.
*/
while (cpu_rq(cpu)->core->core_pick_in_flight) {
sched_core_unlock(cpu, &flags);
cpu_relax();
sched_core_lock(cpu, &flags);
}
for_each_cpu(t, smt_mask)
cpu_rq(t)->core_enabled = enabled;
@@ -6223,7 +6234,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
unsigned long cookie;
int i, cpu, occ = 0;
struct rq *rq_i;
bool need_sync;
bool need_sync = false;
if (!sched_core_enabled(rq))
return __pick_next_task(rq, rf);
@@ -6242,6 +6253,8 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
return __pick_next_task(rq, rf);
}
rq->core->core_pick_in_flight++;
/*
* If there were no {en,de}queues since we picked (IOW, the task
* pointers are all still valid), and we haven't scheduled the last
@@ -6266,7 +6279,9 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
prev_balance(rq, rf);
smt_mask = cpu_smt_mask(cpu);
need_sync = !!rq->core->core_cookie;
restart:
need_sync |= !!rq->core->core_cookie;
/* reset state */
rq->core->core_cookie = 0UL;
@@ -6301,10 +6316,15 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
* and there are no cookied tasks running on siblings.
*/
if (!need_sync) {
restart_single:
next = pick_task(rq, rf);
if (unlikely(next == RETRY_TASK))
goto restart_single;
if (unlikely(next == RETRY_TASK)) {
/* rq lock may have been dropped, clocks invalidated */
core_clock_updated = false;
if (!(rq->clock_update_flags & RQCF_UPDATED))
update_rq_clock(rq);
goto restart;
}
if (!next->core_cookie) {
rq->core_pick = NULL;
rq->core_dl_server = NULL;
@@ -6324,7 +6344,6 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
*
* Tie-break prio towards the current CPU
*/
restart_multi:
max = NULL;
for_each_cpu_wrap(i, smt_mask, cpu) {
rq_i = cpu_rq(i);
@@ -6338,8 +6357,13 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
update_rq_clock(rq_i);
p = pick_task(rq_i, rf);
if (unlikely(p == RETRY_TASK))
goto restart_multi;
if (unlikely(p == RETRY_TASK)) {
/* rq lock may have been dropped, clocks invalidated */
core_clock_updated = false;
if (!(rq->clock_update_flags & RQCF_UPDATED))
update_rq_clock(rq);
goto restart;
}
rq_i->core_pick = p;
rq_i->core_dl_server = rq_i->dl_server;
@@ -6445,6 +6469,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
}
out_set_next:
rq->core->core_pick_in_flight--;
put_prev_set_next_task(rq, rq->donor, next);
if (rq->core->core_forceidle_count && next == rq->idle)
queue_core_balance(rq);
@@ -6639,6 +6664,13 @@ static void sched_core_cpu_deactivate(unsigned int cpu)
core_rq->core_forceidle_seq = rq->core_forceidle_seq;
core_rq->core_forceidle_occupation = rq->core_forceidle_occupation;
/*
* A stale leftover would bias the count forever if this CPU later
* returns as its own leader. Move, don't copy.
*/
core_rq->core_pick_in_flight = rq->core_pick_in_flight;
rq->core_pick_in_flight = 0;
/*
* Accounting edge for forced idle is handled in pick_next_task().
* Don't need another one here, since the hotplug thread shouldn't
@@ -9053,6 +9085,7 @@ void __init sched_init(void)
rq->core_forceidle_count = 0;
rq->core_forceidle_occupation = 0;
rq->core_forceidle_start = 0;
rq->core_pick_in_flight = 0;
rq->core_cookie = 0UL;
#endif

View File

@@ -67,6 +67,7 @@ extern s32 __rcu *scx_shard_node;
extern struct scx_cid_shard __rcu *scx_cid_shard_ranges;
extern struct scx_cid_topo __rcu *scx_cid_topo;
extern struct btf_id_set8 scx_kfunc_ids_init_cids;
extern struct btf_id_set8 scx_kfunc_ids_cid;
void scx_cmask_clear(struct scx_cmask *m);
void scx_cmask_fill(struct scx_cmask *m);

View File

@@ -413,12 +413,27 @@ static bool rq_is_open(struct rq *rq, u64 enq_flags)
*/
DEFINE_PER_CPU(struct rq *, scx_locked_rq_state);
/*
* Under core scheduling, a pick that releases the rq lock invalidates the
* core-wide selection it is part of. Count the releases so that the core-sched
* pick can tell whether one happened across dispatch.
*/
static void scx_rq_lock_drop(struct rq *rq)
{
lockdep_assert_rq_held(rq);
#ifdef CONFIG_SCHED_CORE
if (sched_core_enabled(rq))
rq->scx.lock_drop_seq++;
#endif
}
static void switch_rq_lock(struct rq *from, struct rq *to)
{
bool tracked = scx_locked_rq() == from;
if (tracked)
update_locked_rq(NULL);
scx_rq_lock_drop(from);
raw_spin_rq_unlock(from);
raw_spin_rq_lock(to);
if (tracked)
@@ -987,6 +1002,7 @@ static void deferred_irq_workfn(struct irq_work *irq_work)
raw_spin_rq_lock(rq);
run_deferred(rq);
scx_rq_lock_drop(rq);
raw_spin_rq_unlock(rq);
}
@@ -2922,7 +2938,7 @@ static void finish_dispatch(struct scx_sched *sch, struct rq *rq, struct task_st
BUG_ON(!(p->scx.flags & SCX_TASK_QUEUED));
dsq = find_dsq_for_dispatch(sch, this_rq(), dsq_id, task_cpu(p));
dsq = find_dsq_for_dispatch(sch, rq, dsq_id, task_cpu(p));
if (dsq->id == SCX_DSQ_LOCAL)
dispatch_to_local_dsq(sch, rq, dsq, p, slice, vtime, enq_flags);
@@ -2960,14 +2976,14 @@ static inline void maybe_queue_balance_callback(struct rq *rq)
rq->scx.flags &= ~SCX_RQ_BAL_CB_PENDING;
}
static int balance_one(struct rq *rq, struct task_struct *prev)
static enum scx_dsp_verdict balance_one(struct rq *rq, struct task_struct *prev)
{
struct scx_sched *sch = scx_root_protected_live();
enum scx_dsp_verdict verdict;
s32 cpu = cpu_of(rq);
lockdep_assert_rq_held(rq);
rq->scx.flags |= SCX_RQ_IN_BALANCE;
rq->scx.flags &= ~SCX_RQ_BAL_KEEP;
scx_process_sync_ecaps(rq, prev);
@@ -2999,16 +3015,19 @@ static int balance_one(struct rq *rq, struct task_struct *prev)
*/
if ((prev->scx.flags & SCX_TASK_QUEUED) && prev->scx.slice &&
!scx_bypassing(sch, cpu)) {
rq->scx.flags |= SCX_RQ_BAL_KEEP;
verdict = SCX_DSP_PREV;
goto has_tasks;
}
}
/* if there already are tasks to run, nothing to do */
if (rq->scx.local_dsq.nr)
if (rq->scx.local_dsq.nr) {
verdict = SCX_DSP_LOCAL;
goto has_tasks;
}
if (scx_dispatch_sched(sch, rq, prev, false))
verdict = scx_dispatch_sched(sch, rq, prev, false);
if (verdict != SCX_DSP_NONE)
goto has_tasks;
/*
@@ -3018,12 +3037,12 @@ static int balance_one(struct rq *rq, struct task_struct *prev)
if ((prev->scx.flags & SCX_TASK_QUEUED) &&
(!(sch->ops.flags & SCX_OPS_ENQ_LAST) || scx_bypassing(sch, cpu)) &&
scx_task_can_stay_on_cpu(rq, prev)) {
rq->scx.flags |= SCX_RQ_BAL_KEEP;
__scx_add_event(sch, SCX_EV_DISPATCH_KEEP_LAST, 1);
verdict = SCX_DSP_PREV;
goto has_tasks;
}
rq->scx.flags &= ~SCX_RQ_IN_BALANCE;
return false;
return SCX_DSP_NONE;
has_tasks:
/*
@@ -3040,7 +3059,7 @@ static int balance_one(struct rq *rq, struct task_struct *prev)
scx_schedule_reenq_local(rq, 0);
rq->scx.flags &= ~SCX_RQ_IN_BALANCE;
return true;
return verdict;
}
static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
@@ -3229,13 +3248,13 @@ static void put_prev_task_scx(struct rq *rq, struct task_struct *p,
* which should trigger an explicit follow-up scheduling event.
* This doesn't apply if the baseline access on the CPU is lost.
*
* Core scheduling can force this CPU idle while @p stays
* runnable. @p's cookie then won't match the core's, so skip
* the warning in that case.
* Under core scheduling, a pick dispatches only when nothing is
* locally runnable and can legitimately go idle with @p still
* runnable (see do_pick_task_scx()).
*/
if (next && sched_class_above(&ext_sched_class, next->sched_class) &&
scx_task_can_stay_on_cpu(rq, p)) {
WARN_ON_ONCE(sched_cpu_cookie_match(rq, p) &&
WARN_ON_ONCE(!sched_core_enabled(rq) &&
!(sch->ops.flags & SCX_OPS_ENQ_LAST));
scx_do_enqueue_task(rq, p, SCX_ENQ_LAST, -1);
} else {
@@ -3277,6 +3296,7 @@ static void kick_sync_wait_bal_cb(struct rq *rq)
continue;
}
scx_rq_lock_drop(rq);
raw_spin_rq_unlock_irq(rq);
while (READ_ONCE(cpu_rq(cpu)->scx.kick_sync) == ksyncs[cpu]) {
smp_store_release(&rq->scx.kick_sync, rq->scx.kick_sync + 1);
@@ -3296,27 +3316,23 @@ static struct task_struct *first_local_task(struct rq *rq)
struct task_struct, scx.dsq_list.node);
}
static struct task_struct *
do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
/*
* Run dispatch and queue the follow-up work for a pick.
*/
static enum scx_dsp_verdict dispatch_pick(struct rq *rq, struct rq_flags *rf,
struct task_struct *prev)
{
struct task_struct *prev = rq->curr;
bool keep_prev;
struct task_struct *p;
/* see kick_sync_wait_bal_cb() */
smp_store_release(&rq->scx.kick_sync, rq->scx.kick_sync + 1);
rq_modified_begin(rq, &ext_sched_class);
enum scx_dsp_verdict verdict;
rq_unpin_lock(rq, rf);
balance_one(rq, prev);
verdict = balance_one(rq, prev);
rq_repin_lock(rq, rf);
maybe_queue_balance_callback(rq);
/*
* Defer to a balance callback which can drop rq lock and enable
* IRQs. Waiting directly in the pick path would deadlock against
* CPUs sending us IPIs (e.g. TLB flushes) while we wait for them.
* Defer to a balance callback which can drop rq lock and enable IRQs.
* Waiting directly in the pick path would deadlock against CPUs sending
* us IPIs (e.g. TLB flushes) while we wait for them.
*/
if (unlikely(rq->scx.kick_sync_pending)) {
rq->scx.kick_sync_pending = false;
@@ -3324,10 +3340,102 @@ do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
kick_sync_wait_bal_cb);
}
if (unlikely(verdict == SCX_DSP_PREV && prev->sched_class != &ext_sched_class)) {
WARN_ON_ONCE(scx_enable_state() == SCX_ENABLED);
verdict = SCX_DSP_LOCAL;
}
return verdict;
}
#ifdef CONFIG_SCHED_CORE
/*
* Dispatch for a pick when core scheduling is enabled. The selection picks for
* all SMT siblings and the rq_i->core_pick state it builds must stay atomic
* throughout. If the dispatch released the rq lock, anything can have happened
* in between - return %SCX_DSP_RETRY to restart the selection against current
* state.
*/
static enum scx_dsp_verdict dispatch_core_pick(struct rq *rq, struct rq_flags *rf,
struct task_struct *prev)
{
enum scx_dsp_verdict verdict;
u32 seq = rq->scx.lock_drop_seq;
/* another dispatch is in flight on @rq, let that handle it */
if (rq->scx.flags & SCX_RQ_IN_BALANCE)
return SCX_DSP_NONE;
rq_unpin_lock(rq, rf);
verdict = balance_one(rq, prev);
if (cpu_of(rq) == smp_processor_id()) {
maybe_queue_balance_callback(rq);
/* see dispatch_pick() */
if (unlikely(rq->scx.kick_sync_pending)) {
rq->scx.kick_sync_pending = false;
queue_balance_callback(rq, &rq->scx.kick_sync_bal_cb,
kick_sync_wait_bal_cb);
}
} else if (unlikely(rq->scx.flags & SCX_RQ_BAL_CB_PENDING)) {
/*
* Balance callbacks must run in the context that queued them,
* so they can't be queued on another CPU's rq. Run the deferred
* work directly instead.
*/
rq->scx.flags &= ~SCX_RQ_BAL_CB_PENDING;
run_deferred(rq);
}
rq_repin_lock(rq, rf);
/* if balance_one() released the rq lock, restart the selection */
if (rq->scx.lock_drop_seq != seq)
return SCX_DSP_RETRY;
/* see dispatch_pick() */
if (unlikely(verdict == SCX_DSP_PREV &&
prev->sched_class != &ext_sched_class)) {
WARN_ON_ONCE(scx_enable_state() == SCX_ENABLED);
verdict = SCX_DSP_LOCAL;
}
return verdict;
}
#else /* CONFIG_SCHED_CORE */
static enum scx_dsp_verdict dispatch_core_pick(struct rq *rq, struct rq_flags *rf,
struct task_struct *prev)
{
return SCX_DSP_NONE;
}
#endif /* CONFIG_SCHED_CORE */
static struct task_struct *
do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
{
struct task_struct *prev = rq->curr;
enum scx_dsp_verdict verdict;
struct task_struct *p;
/* see kick_sync_wait_bal_cb() */
smp_store_release(&rq->scx.kick_sync, rq->scx.kick_sync + 1);
rq_modified_begin(rq, &ext_sched_class);
if (sched_core_enabled(rq))
verdict = dispatch_core_pick(rq, rf, prev);
else
verdict = dispatch_pick(rq, rf, prev);
if (verdict == SCX_DSP_RETRY)
return RETRY_TASK;
/*
* If any higher-priority sched class enqueued a runnable task on
* this rq during balance_one(), abort and return RETRY_TASK, so
* that the scheduler loop can restart.
* If any higher-priority sched class enqueued a runnable task on this
* rq during balance_one(), abort and return RETRY_TASK, so that the
* scheduler loop can restart.
*
* If @force_scx is true, always try to pick a SCHED_EXT task,
* regardless of any higher-priority sched classes activity.
@@ -3335,19 +3443,11 @@ do_pick_task_scx(struct rq *rq, struct rq_flags *rf, bool force_scx)
if (!force_scx && rq_modified_above(rq, &ext_sched_class))
return RETRY_TASK;
keep_prev = rq->scx.flags & SCX_RQ_BAL_KEEP;
if (unlikely(keep_prev &&
prev->sched_class != &ext_sched_class)) {
WARN_ON_ONCE(scx_enable_state() == SCX_ENABLED);
keep_prev = false;
}
/*
* If balance_one() is telling us to keep running @prev, replenish slice
* if necessary and keep running @prev. Otherwise, pop the first one
* from the local DSQ.
* If we're keeping @prev, replenish slice if necessary and keep running
* @prev. Otherwise, pop the first one from the local DSQ.
*/
if (keep_prev) {
if (verdict == SCX_DSP_PREV) {
p = prev;
if (!p->scx.slice) {
/* the slice is consumed, protection ends */
@@ -4457,8 +4557,10 @@ static void reenq_user(struct rq *rq, struct scx_dispatch_q *dsq, u64 reenq_flag
task_rq = task_rq(p);
if (locked_rq != task_rq) {
if (locked_rq)
if (locked_rq) {
scx_rq_lock_drop(locked_rq);
raw_spin_rq_unlock(locked_rq);
}
if (unlikely(!raw_spin_rq_trylock(task_rq))) {
raw_spin_unlock(&dsq->lock);
raw_spin_rq_lock(task_rq);
@@ -4484,6 +4586,7 @@ static void reenq_user(struct rq *rq, struct scx_dispatch_q *dsq, u64 reenq_flag
p->scx.flags &= ~SCX_TASK_REENQ_REASON_MASK;
if (!(++nr_enqueued % SCX_TASK_ITER_BATCH)) {
scx_rq_lock_drop(locked_rq);
raw_spin_rq_unlock(locked_rq);
locked_rq = NULL;
cpu_relax();
@@ -4496,8 +4599,10 @@ static void reenq_user(struct rq *rq, struct scx_dispatch_q *dsq, u64 reenq_flag
raw_spin_unlock(&dsq->lock);
if (locked_rq != rq) {
if (locked_rq)
if (locked_rq) {
scx_rq_lock_drop(locked_rq);
raw_spin_rq_unlock(locked_rq);
}
raw_spin_rq_lock(rq);
}
}
@@ -5707,6 +5812,7 @@ static u32 bypass_lb_cpu(struct scx_sched *sch, s32 donor,
if (!(nr_balanced % SCX_BYPASS_LB_BATCH) && n) {
list_move_tail(&cursor.node, &n->scx.dsq_list.node);
raw_spin_unlock(&donor_dsq->lock);
scx_rq_lock_drop(donor_rq);
raw_spin_rq_unlock_irq(donor_rq);
cpu_relax();
raw_spin_rq_lock_irq(donor_rq);
@@ -5717,6 +5823,7 @@ static u32 bypass_lb_cpu(struct scx_sched *sch, s32 donor,
list_del_init(&cursor.node);
raw_spin_unlock(&donor_dsq->lock);
scx_rq_lock_drop(donor_rq);
raw_spin_rq_unlock_irq(donor_rq);
return nr_balanced;
@@ -5962,7 +6069,7 @@ static void unbypass_renotify_idle(struct rq *rq, struct scx_sched *pos,
*
* - ops.dispatch() is ignored.
*
* - balance_one() does not set %SCX_RQ_BAL_KEEP on non-zero slice as slice
* - balance_one() does not report %SCX_DSP_PREV on non-zero slice as slice
* can't be trusted. Whenever a tick triggers, the running task is rotated to
* the tail of the queue with core_sched_at touched.
*
@@ -6045,6 +6152,7 @@ void scx_bypass(struct scx_sched *sch, bool bypass)
* sees scx_bypassing() before moving tasks to SCX.
*/
if (!scx_enabled()) {
scx_rq_lock_drop(rq);
raw_spin_rq_unlock(rq);
continue;
}
@@ -6078,6 +6186,7 @@ void scx_bypass(struct scx_sched *sch, bool bypass)
if (cpu_online(cpu) || cpu == smp_processor_id())
resched_curr(rq);
scx_rq_lock_drop(rq);
raw_spin_rq_unlock(rq);
}
@@ -8412,6 +8521,7 @@ static bool kick_one_cpu(s32 cpu, struct scx_sched_pcpu *pcpu, struct rq *this_r
cpumask_clear_cpu(cpu, pcpu->cpus_to_wait);
}
scx_rq_lock_drop(rq);
raw_spin_rq_unlock_irqrestore(rq, flags);
return should_wait;
@@ -8434,6 +8544,7 @@ static void kick_one_cpu_if_idle(s32 cpu, struct scx_sched_pcpu *pcpu,
__scx_add_event(pcpu->sch, SCX_EV_SUB_KICK_DENIED, 1);
}
scx_rq_lock_drop(rq);
raw_spin_rq_unlock_irqrestore(rq, flags);
}
@@ -8481,6 +8592,7 @@ static void kick_cpus_irq_workfn(struct irq_work *irq_work)
raw_spin_rq_lock(this_rq);
this_scx->kick_sync_pending = true;
resched_curr(this_rq);
scx_rq_lock_drop(this_rq);
raw_spin_rq_unlock(this_rq);
}
}
@@ -8887,9 +8999,8 @@ static bool scx_dsq_move(struct bpf_iter_scx_dsq_kern *kit,
{
struct scx_dispatch_q *src_dsq = kit->dsq, *dst_dsq;
struct scx_sched *sch;
struct rq *this_rq, *src_rq, *locked_rq;
struct rq *p_rq, *src_rq, *locked_rq;
bool dispatched = false;
bool in_balance;
unsigned long flags;
/*
@@ -8923,24 +9034,28 @@ static bool scx_dsq_move(struct bpf_iter_scx_dsq_kern *kit,
}
/*
* Can be called from either ops.dispatch() locking this_rq() or any
* context where no rq lock is held. If latter, lock @p's task_rq which
* we'll likely need anyway.
* Can be called from either ops.dispatch() holding the dispatched rq's
* lock or any context where no rq lock is held. If latter, lock @p's
* task_rq which we'll likely need anyway.
*/
src_rq = task_rq(p);
local_irq_save(flags);
this_rq = this_rq();
in_balance = this_rq->scx.flags & SCX_RQ_IN_BALANCE;
if (in_balance) {
if (this_rq != src_rq)
switch_rq_lock(this_rq, src_rq);
/*
* Under core scheduling, dispatch can run for a sibling rq, so the
* locked rq is not necessarily this CPU's.
*/
locked_rq = scx_locked_rq();
if (locked_rq) {
if (locked_rq != src_rq)
switch_rq_lock(locked_rq, src_rq);
} else {
raw_spin_rq_lock(src_rq);
}
locked_rq = src_rq;
p_rq = src_rq;
raw_spin_lock(&src_dsq->lock);
/* did someone else get to it while we dropped the locks? */
@@ -8950,7 +9065,7 @@ static bool scx_dsq_move(struct bpf_iter_scx_dsq_kern *kit,
}
/* @p is still on $src_dsq and stable, determine the destination */
dst_dsq = find_dsq_for_dispatch(sch, this_rq, dsq_id, task_cpu(p));
dst_dsq = find_dsq_for_dispatch(sch, locked_rq ?: this_rq(), dsq_id, task_cpu(p));
/*
* Apply vtime and slice updates before moving. @p is still on $src_dsq
@@ -8963,14 +9078,15 @@ static bool scx_dsq_move(struct bpf_iter_scx_dsq_kern *kit,
scx_set_task_slice(p, kit->slice);
/* execute move */
locked_rq = move_task_between_dsqs(sch, p, enq_flags, src_dsq, dst_dsq);
p_rq = move_task_between_dsqs(sch, p, enq_flags, src_dsq, dst_dsq);
dispatched = true;
out:
if (in_balance) {
if (this_rq != locked_rq)
switch_rq_lock(locked_rq, this_rq);
if (locked_rq) {
if (locked_rq != p_rq)
switch_rq_lock(p_rq, locked_rq);
} else {
raw_spin_rq_unlock_irqrestore(locked_rq, flags);
scx_rq_lock_drop(p_rq);
raw_spin_rq_unlock_irqrestore(p_rq, flags);
}
kit->cursor.flags &= ~(__SCX_DSQ_ITER_HAS_SLICE |
@@ -9464,9 +9580,11 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags)
if (raw_spin_rq_trylock(target_rq)) {
if (can_skip_idle_kick(target_rq)) {
scx_rq_lock_drop(target_rq);
raw_spin_rq_unlock(target_rq);
goto out;
}
scx_rq_lock_drop(target_rq);
raw_spin_rq_unlock(target_rq);
}
cpumask_set_cpu(cpu, pcpu->cpus_to_kick_if_idle);
@@ -9541,6 +9659,10 @@ __bpf_kfunc void scx_bpf_kick_cid(s32 cid, u64 flags, const struct bpf_prog_aux
*
* Return the number of tasks in the DSQ matching @dsq_id. If not found,
* -%ENOENT is returned.
*
* %SCX_DSQ_LOCAL resolves to the local DSQ of the rq the current scheduler
* operation is locked to - e.g. the rq being dispatched for in ops.dispatch() -
* or the calling CPU's when no rq is locked.
*/
__bpf_kfunc s32 scx_bpf_dsq_nr_queued(u64 dsq_id, const struct bpf_prog_aux *aux)
{
@@ -9557,7 +9679,7 @@ __bpf_kfunc s32 scx_bpf_dsq_nr_queued(u64 dsq_id, const struct bpf_prog_aux *aux
}
if (dsq_id == SCX_DSQ_LOCAL) {
ret = READ_ONCE(this_rq()->scx.local_dsq.nr);
ret = READ_ONCE((scx_locked_rq() ?: this_rq())->scx.local_dsq.nr);
goto out;
} else if ((dsq_id & SCX_DSQ_LOCAL_ON) == SCX_DSQ_LOCAL_ON) {
s32 cpu = scx_cpu_ret(sch, dsq_id & SCX_DSQ_LOCAL_CPU_MASK);
@@ -9736,10 +9858,15 @@ __bpf_kfunc struct task_struct *scx_bpf_dsq_peek(u64 dsq_id,
* - User DSQs
*
* Re-enqueues are performed asynchronously. Can be called from anywhere.
*
* %SCX_DSQ_LOCAL resolves to the local DSQ of the rq the current scheduler
* operation is locked to - e.g. the rq being dispatched for in ops.dispatch() -
* or the calling CPU's when no rq is locked.
*/
__bpf_kfunc void scx_bpf_dsq_reenq(u64 dsq_id, u64 reenq_flags,
const struct bpf_prog_aux *aux)
{
struct rq *locked_rq = scx_locked_rq();
struct scx_sched *sch;
struct scx_dispatch_q *dsq;
@@ -9758,8 +9885,8 @@ __bpf_kfunc void scx_bpf_dsq_reenq(u64 dsq_id, u64 reenq_flags,
if (!(reenq_flags & __SCX_REENQ_FILTER_MASK))
reenq_flags |= SCX_REENQ_ANY;
dsq = find_dsq_for_dispatch(sch, this_rq(), dsq_id, smp_processor_id());
schedule_dsq_reenq(sch, dsq, reenq_flags, scx_locked_rq());
dsq = find_dsq_for_dispatch(sch, locked_rq ?: this_rq(), dsq_id, smp_processor_id());
schedule_dsq_reenq(sch, dsq, reenq_flags, locked_rq);
}
/**
@@ -10739,19 +10866,20 @@ int scx_kfunc_context_filter(const struct bpf_prog *prog, u32 kfunc_id)
bool in_idle = btf_id_set8_contains(&scx_kfunc_ids_idle, kfunc_id);
bool in_any = btf_id_set8_contains(&scx_kfunc_ids_any, kfunc_id);
bool in_cpu_only = btf_id_set8_contains(&scx_kfunc_ids_cpu_only, kfunc_id);
bool in_cid = btf_id_set8_contains(&scx_kfunc_ids_cid, kfunc_id);
u32 moff, flags;
/* Not an SCX kfunc - allow. */
if (!(in_unlocked || in_init_cids || in_select_cpu || in_enqueue || in_dispatch ||
in_cpu_release || in_idle || in_any))
in_cpu_release || in_idle || in_any || in_cid))
return 0;
/* SYSCALL progs (e.g. BPF test_run()) may call unlocked and select_cpu kfuncs. */
if (prog->type == BPF_PROG_TYPE_SYSCALL)
return (in_unlocked || in_select_cpu || in_idle || in_any) ? 0 : -EACCES;
return (in_unlocked || in_select_cpu || in_idle || in_any || in_cid) ? 0 : -EACCES;
if (prog->type != BPF_PROG_TYPE_STRUCT_OPS)
return (in_any || in_idle) ? 0 : -EACCES;
return (in_any || in_idle || in_cid) ? 0 : -EACCES;
/*
* add_subprog_and_kfunc() collects all kfunc calls, including dead code
@@ -10786,7 +10914,7 @@ int scx_kfunc_context_filter(const struct bpf_prog *prog, u32 kfunc_id)
return -EACCES;
/* SCX struct_ops: check the per-op allow list. */
if (in_any || in_idle)
if (in_any || in_idle || in_cid)
return 0;
moff = prog->aux->attach_st_ops_member_off;

View File

@@ -13,12 +13,20 @@
#include "internal.h"
#include "cid.h"
/* what dispatch concluded, consumed by the pick that follows */
enum scx_dsp_verdict {
SCX_DSP_NONE, /* nothing to run */
SCX_DSP_LOCAL, /* local DSQ has tasks */
SCX_DSP_PREV, /* keep running @prev */
SCX_DSP_RETRY, /* pick helpers only: restart the pick */
};
/*
* One user of this function is scx_bpf_dispatch() which can be called
* recursively as sub-sched dispatches nest. Always inline to reduce stack usage
* from the call frame.
*/
static __always_inline bool
static __always_inline enum scx_dsp_verdict
scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
struct task_struct *prev, bool nested)
{
@@ -29,12 +37,15 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
scx_task_on_sched(sch, prev);
if (scx_consume_global_dsq(sch, rq))
return true;
return SCX_DSP_LOCAL;
if (scx_bypass_dsp_enabled(sch)) {
/* if @sch is bypassing, only the bypass DSQs are active */
if (scx_bypassing(sch, cpu))
return scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0);
if (scx_bypassing(sch, cpu)) {
if (scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0))
return SCX_DSP_LOCAL;
return SCX_DSP_NONE;
}
#ifdef CONFIG_EXT_SUB_SCHED
/*
@@ -54,13 +65,13 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
if (!(pcpu->bypass_host_seq++ % SCX_BYPASS_HOST_NTH) &&
scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0)) {
__scx_add_event(sch, SCX_EV_SUB_BYPASS_DISPATCH, 1);
return true;
return SCX_DSP_LOCAL;
}
#endif /* CONFIG_EXT_SUB_SCHED */
}
if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !scx_rq_online(rq))
return false;
return SCX_DSP_NONE;
dspc->rq = rq;
@@ -90,14 +101,12 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
scx_flush_dispatch_buf(sch, rq);
if ((prev->scx.flags & SCX_TASK_QUEUED) && prev->scx.slice) {
rq->scx.flags |= SCX_RQ_BAL_KEEP;
return true;
}
if ((prev->scx.flags & SCX_TASK_QUEUED) && prev->scx.slice)
return SCX_DSP_PREV;
if (rq->scx.local_dsq.nr)
return true;
return SCX_DSP_LOCAL;
if (scx_consume_global_dsq(sch, rq))
return true;
return SCX_DSP_LOCAL;
/*
* ops.dispatch() can trap us in this loop by repeatedly
@@ -119,10 +128,11 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
* queued. Without this fallback, bypassed tasks could stall if the host
* scheduler's ops.dispatch() doesn't yield any tasks.
*/
if (scx_bypass_dsp_enabled(sch))
return scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0);
if (scx_bypass_dsp_enabled(sch) &&
scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0))
return SCX_DSP_LOCAL;
return false;
return SCX_DSP_NONE;
}
#endif /* _KERNEL_SCHED_EXT_INLINES_H */

View File

@@ -2222,7 +2222,7 @@ __bpf_kfunc_start_defs();
*/
__bpf_kfunc bool scx_bpf_sub_dispatch(u64 cgroup_id, const struct bpf_prog_aux *aux)
{
struct rq *this_rq = this_rq();
struct rq *rq = scx_locked_rq();
struct scx_sched *parent, *child;
guard(rcu)();
@@ -2246,11 +2246,11 @@ __bpf_kfunc bool scx_bpf_sub_dispatch(u64 cgroup_id, const struct bpf_prog_aux *
* its inserts would only be rejected. ecaps are synced at the top of
* balance_one() before dispatch, so this reflects the in-effect state.
*/
if (scx_missing_caps(child, cpu_of(this_rq), SCX_CAP_BASE))
if (scx_missing_caps(child, cpu_of(rq), SCX_CAP_BASE))
return false;
return scx_dispatch_sched(child, this_rq, this_rq->scx.sub_dispatch_prev,
true);
return scx_dispatch_sched(child, rq, rq->scx.sub_dispatch_prev, true) !=
SCX_DSP_NONE;
}
/* Validate common inputs. On success, *parent_out and *child_out are set. */

View File

@@ -784,7 +784,6 @@ enum scx_rq_flags {
*/
SCX_RQ_ONLINE = 1 << 0,
SCX_RQ_CAN_STOP_TICK = 1 << 1,
SCX_RQ_BAL_KEEP = 1 << 3, /* balance decided to keep current */
SCX_RQ_CLK_VALID = 1 << 5, /* RQ clock is fresh and valid */
SCX_RQ_BAL_CB_PENDING = 1 << 6, /* must queue a cb after dispatching */
SCX_RQ_SUB_IDLE_RENOTIFY = 1 << 7, /* sub-scheds are owed update_idle() */
@@ -824,6 +823,9 @@ struct scx_rq {
bool cpu_released;
u32 flags;
u32 nr_immed; /* ENQ_IMMED tasks on local_dsq */
#ifdef CONFIG_SCHED_CORE
u32 lock_drop_seq; /* nr dispatch lock releases */
#endif
u64 clock; /* current per-rq clock -- see scx_bpf_now() */
#ifdef CONFIG_EXT_SUB_SCHED
struct llist_head ecaps_to_sync; /* pending ecaps syncs */
@@ -1377,6 +1379,7 @@ struct rq {
unsigned int core_forceidle_seq;
unsigned int core_forceidle_occupation;
u64 core_forceidle_start;
unsigned int core_pick_in_flight;
#endif /* CONFIG_SCHED_CORE */
/* Scratch cpumask to be temporarily used under rq_lock */

View File

@@ -183,7 +183,6 @@
#define HAVE___SCX_REENQ_TSR_MASK
#define HAVE_SCX_RQ_ONLINE
#define HAVE_SCX_RQ_CAN_STOP_TICK
#define HAVE_SCX_RQ_BAL_KEEP
#define HAVE_SCX_RQ_CLK_VALID
#define HAVE_SCX_RQ_BAL_CB_PENDING
#define HAVE_SCX_RQ_SUB_IDLE_RENOTIFY

View File

@@ -22,9 +22,6 @@ const volatile u64 __SCX_RQ_CAN_STOP_TICK __weak;
const volatile u64 __SCX_RQ_BAL_PENDING __weak;
#define SCX_RQ_BAL_PENDING __SCX_RQ_BAL_PENDING
const volatile u64 __SCX_RQ_BAL_KEEP __weak;
#define SCX_RQ_BAL_KEEP __SCX_RQ_BAL_KEEP
const volatile u64 __SCX_RQ_BYPASSING __weak;
#define SCX_RQ_BYPASSING __SCX_RQ_BYPASSING

View File

@@ -11,7 +11,6 @@
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_ONLINE); \
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_CAN_STOP_TICK); \
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_BAL_PENDING); \
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_BAL_KEEP); \
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_BYPASSING); \
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_CLK_VALID); \
SCX_ENUM_SET(skel, scx_rq_flags, SCX_RQ_IN_WAKEUP); \