/* @mask is constant, always inline to cull unnecessary branches */ static __always_inline bool scx_kf_allowed(u32 mask)
{ if (unlikely(!(current->scx.kf_mask & mask))) {
scx_kf_error("kfunc with mask 0x%x called from an operation only allowing 0x%x",
mask, current->scx.kf_mask); returnfalse;
}
/* *Enforcenestingboundaries.e.g.Akfuncwhichcanbecalledfrom *DISPATCHmustnotbecalledifwe'rerunningDEQUEUEwhichisnested *insideops.dispatch().Wedon'tneedtocheckboundariesforany *blockingkfuncsastheverifierensuresthey'reonlycalledfrom *sleepableprogs.
*/ if (unlikely(highest_bit(mask) == SCX_KF_CPU_RELEASE &&
(current->scx.kf_mask & higher_bits(SCX_KF_CPU_RELEASE)))) {
scx_kf_error("cpu_release kfunc called from a nested operation"); returnfalse;
}
if (unlikely(highest_bit(mask) == SCX_KF_DISPATCH &&
(current->scx.kf_mask & higher_bits(SCX_KF_DISPATCH)))) {
scx_kf_error("dispatch kfunc called from a nested operation"); returnfalse;
}
returntrue;
}
/* see SCX_CALL_OP_TASK() */ static __always_inline bool scx_kf_allowed_on_arg_tasks(u32 mask, struct task_struct *p)
{ if (!scx_kf_allowed(mask)) returnfalse;
if (unlikely((p != current->scx.kf_tasks[0] &&
p != current->scx.kf_tasks[1]))) {
scx_kf_error("called on a task not being operated on"); returnfalse;
}
if (!is_local) {
raw_spin_lock(&dsq->lock); if (unlikely(dsq->id == SCX_DSQ_INVALID)) {
scx_error(sch, "attempting to dispatch to a destroyed dsq"); /* fall back to the global dsq */
raw_spin_unlock(&dsq->lock);
dsq = find_global_dsq(p);
raw_spin_lock(&dsq->lock);
}
}
if (unlikely((dsq->id & SCX_DSQ_FLAG_BUILTIN) &&
(enq_flags & SCX_ENQ_DSQ_PRIQ))) { /* *SCX_DSQ_LOCALandSCX_DSQ_GLOBALDSQsalwaysconsumefrom *theirFIFOqueues.Toavoidconfusionandaccidentally *starvingvtime-dispatchedtasksbyFIFO-dispatchedtasks,we *disallowanyinternalDSQfromdoingvtimeorderingof *tasks.
*/
scx_error(sch, "cannot use vtime ordering for built-in DSQs");
enq_flags &= ~SCX_ENQ_DSQ_PRIQ;
}
if (enq_flags & SCX_ENQ_DSQ_PRIQ) { struct rb_node *rbp;
/* *APRIQDSQshouldn'tbeusingFIFOenqueueing.Astasksare *linkedtoboththerbtreeandlistonPRIQs,thiscanonlybe *testedeasilywhenaddingthefirsttask.
*/ if (unlikely(RB_EMPTY_ROOT(&dsq->priq) &&
nldsq_next_task(dsq, NULL, false)))
scx_error(sch, "DSQ ID 0x%016llx already had FIFO-enqueued tasks",
dsq->id);
if (!dsq) { /* *If!dsq&&on-list,@pison@rq'sddsp_deferred_locals. *Unlinkingisallthat'sneededtocancel.
*/ if (unlikely(!list_empty(&p->scx.dsq_list.node)))
list_del_init(&p->scx.dsq_list.node);
/* @p must match the task on the enqueue path */ if (unlikely(p != ddsp_task)) { if (IS_ERR(ddsp_task))
scx_kf_error("%s[%d] already direct-dispatched",
p->comm, p->pid); else
scx_kf_error("scheduling for %s[%d] but trying to direct-dispatch %s[%d]",
ddsp_task->comm, ddsp_task->pid,
p->comm, p->pid); return;
}
/* the following marks @p MIGRATING which excludes dequeue */
deactivate_task(src_rq, p, 0);
set_task_cpu(p, cpu_of(dst_rq));
p->scx.sticky_cpu = cpu_of(dst_rq);
if (likely(!atomic_read(&scx_breather_depth))) return;
raw_spin_rq_unlock(rq);
until = ktime_get_ns() + NSEC_PER_MSEC;
do { int cnt = 1024; while (atomic_read(&scx_breather_depth) && --cnt)
cpu_relax();
} while (atomic_read(&scx_breather_depth) &&
time_before64(ktime_get_ns(), until));
/* store_release ensures that dequeue sees the above */
atomic_long_set_release(&p->scx.ops_state, SCX_OPSS_NONE);
/* switch to @src_rq lock */ if (locked_rq != src_rq) {
raw_spin_rq_unlock(locked_rq);
locked_rq = src_rq;
raw_spin_rq_lock(src_rq);
}
/* task_rq couldn't have changed if we're still the holding cpu */ if (likely(p->scx.holding_cpu == raw_smp_processor_id()) &&
!WARN_ON_ONCE(src_rq != task_rq(p))) { /* *If@pisstayingonthesamerq,there'snoneedtogo *throughthefulldeactivate/activatecycle.Optimizeby *abbreviatingmove_remote_task_to_local_dsq().
*/ if (src_rq == dst_rq) {
p->scx.holding_cpu = -1;
dispatch_enqueue(sch, &dst_rq->scx.local_dsq, p,
enq_flags);
} else {
move_remote_task_to_local_dsq(p, enq_flags,
src_rq, dst_rq); /* task has been moved to dst_rq, which is now locked */
locked_rq = dst_rq;
}
/* if the destination CPU is idle, wake it up */ if (sched_class_above(p->sched_class, dst_rq->curr->sched_class))
resched_curr(dst_rq);
}
/* switch back to @rq lock */ if (locked_rq != rq) {
raw_spin_rq_unlock(locked_rq);
raw_spin_rq_lock(rq);
}
}
/* see dequeue_task_scx() on why we skip when !QUEUED */ if (SCX_HAS_OP(sch, running) && (p->scx.flags & SCX_TASK_QUEUED))
SCX_CALL_OP_TASK(sch, SCX_KF_REST, running, rq, p);
/* see dequeue_task_scx() on why we skip when !QUEUED */ if (SCX_HAS_OP(sch, stopping) && (p->scx.flags & SCX_TASK_QUEUED))
SCX_CALL_OP_TASK(sch, SCX_KF_REST, stopping, rq, p, true);
if (p->scx.flags & SCX_TASK_QUEUED) {
set_task_runnable(rq, p);
/* *Ifbalance_scx()istellingustokeeprunning@prev,replenishslice *ifnecessaryandkeeprunning@prev.Otherwise,popthefirstone *fromthelocalDSQ.
*/ if (keep_prev) {
p = prev; if (!p->scx.slice)
refill_task_slice_dfl(p);
} else {
p = first_local_task(rq); if (!p) { if (kick_idle)
scx_bpf_kick_cpu(cpu_of(rq), SCX_KICK_IDLE); return NULL;
}
if (unlikely(!p->scx.slice)) { struct scx_sched *sch = scx_root;
if (!scx_rq_bypassing(rq) && !sch->warned_zero_slice) {
printk_deferred(KERN_WARNING "sched_ext: %s[%d] has zero slice in %s()\n",
p->comm, p->pid, __func__);
sch->warned_zero_slice = true;
}
refill_task_slice_dfl(p);
}
}
/* cycling deq/enq is enough, see the function comment */
sched_deq_and_put_task(p, DEQUEUE_SAVE | DEQUEUE_MOVE, &ctx);
sched_enq_and_set_task(&ctx);
}
/* resched to restore ticks and idle state */ if (cpu_online(cpu) || cpu == smp_processor_id())
resched_curr(rq);
staticconstchar *scx_exit_reason(enum scx_exit_kind kind)
{ switch (kind) { case SCX_EXIT_UNREG: return"unregistered from user space"; case SCX_EXIT_UNREG_BPF: return"unregistered from BPF"; case SCX_EXIT_UNREG_KERN: return"unregistered from the main kernel"; case SCX_EXIT_SYSRQ: return"disabled by sysrq-S"; case SCX_EXIT_ERROR: return"runtime error"; case SCX_EXIT_ERROR_BPF: return"scx_bpf_error"; case SCX_EXIT_ERROR_STALL: return"runnable task stall"; default: return"<UNKNOWN>";
}
}
/* no task is on scx, turn off all the switches and flush in-progress calls */
static_branch_disable(&__scx_enabled);
bitmap_zero(sch->has_op, SCX_OPI_END);
scx_idle_disable();
synchronize_rcu();
trace_sched_ext_dump(line_buf);
} #endif /* @s may be zero sized and seq_buf triggers WARN if so */ if (s->size) {
va_start(args, fmt);
seq_buf_vprintf(s, fmt, args);
va_end(args);
used = seq_buf_used(&ns); if (SCX_HAS_OP(sch, dump_cpu)) {
ops_dump_init(&ns, " ");
SCX_CALL_OP(sch, SCX_KF_REST, dump_cpu, NULL,
&dctx, cpu, idle);
ops_dump_exit();
}
/*
* If idle && nothing generated by ops.dump_cpu(), there's
* nothing interesting. Skip.
*/
if (idle && used == seq_buf_used(&ns))
goto next;
/*
* $s may already have overflowed when $ns was created. If so,
* calling commit on it will trigger BUG.
*/
if (avail) {
seq_buf_commit(&s, seq_buf_used(&ns));
if (seq_buf_has_overflowed(&ns))
seq_buf_set_overflow(&s);
}
if (rq->curr->sched_class == &ext_sched_class)
scx_dump_task(&s, &dctx, rq->curr, '*');
/*
* Set ei->kind and ->reason for scx_dump_state(). They'll be set again
* in scx_disable_workfn().
*/
ei->kind = kind;
ei->reason = scx_exit_reason(ei->kind);
static void check_hotplug_seq(struct scx_sched *sch,
const struct sched_ext_ops *ops)
{
unsigned long long global_hotplug_seq;
/*
* If a hotplug event has occurred between when a scheduler was
* initialized, and when we were able to attach, exit and notify user
* space about it.
*/
if (ops->hotplug_seq) {
global_hotplug_seq = atomic_long_read(&scx_hotplug_seq);
if (ops->hotplug_seq != global_hotplug_seq) {
scx_exit(sch, SCX_EXIT_UNREG_KERN,
SCX_ECODE_ACT_RESTART | SCX_ECODE_RSN_HOTPLUG,
"expected hotplug seq %llu did not match actual %llu",
ops->hotplug_seq, global_hotplug_seq);
}
}
}
static int validate_ops(struct scx_sched *sch, const struct sched_ext_ops *ops)
{
/*
* It doesn't make sense to specify the SCX_OPS_ENQ_LAST flag if the
* ops.enqueue() callback isn't implemented.
*/
if ((ops->flags & SCX_OPS_ENQ_LAST) && !ops->enqueue) {
scx_error(sch, "SCX_OPS_ENQ_LAST requires ops.enqueue() to be implemented");
return -EINVAL;
}
/*
* SCX_OPS_BUILTIN_IDLE_PER_NODE requires built-in CPU idle
* selection policy to be enabled.
*/
if ((ops->flags & SCX_OPS_BUILTIN_IDLE_PER_NODE) &&
(ops->update_idle && !(ops->flags & SCX_OPS_KEEP_BUILTIN_IDLE))) {
scx_error(sch, "SCX_OPS_BUILTIN_IDLE_PER_NODE requires CPU idle selection enabled");
return -EINVAL;
}
if (ops->flags & SCX_OPS_HAS_CGROUP_WEIGHT)
pr_warn("SCX_OPS_HAS_CGROUP_WEIGHT is deprecated and a noop\n");
return 0;
}
static int scx_enable(struct sched_ext_ops *ops, struct bpf_link *link)
{
struct scx_sched *sch;
struct scx_task_iter sti;
struct task_struct *p;
unsigned long timeout;
int i, cpu, ret;
if (!cpumask_equal(housekeeping_cpumask(HK_TYPE_DOMAIN),
cpu_possible_mask)) {
pr_err("sched_ext: Not compatible with \"isolcpus=\" domain isolation\n");
return -EINVAL;
}
mutex_lock(&scx_enable_mutex);
if (scx_enable_state() != SCX_DISABLED) {
ret = -EBUSY;
goto err_unlock;
}
sch = scx_alloc_and_add_sched(ops);
if (IS_ERR(sch)) {
ret = PTR_ERR(sch);
goto err_unlock;
}
/*
* Transition to ENABLING and clear exit info to arm the disable path.
* Failure triggers full disabling from here on.
*/
WARN_ON_ONCE(scx_set_enable_state(SCX_ENABLING) != SCX_DISABLED);
WARN_ON_ONCE(scx_root);
/*
* Keep CPUs stable during enable so that the BPF scheduler can track
* online CPUs by watching ->on/offline_cpu() after ->init().
*/
cpus_read_lock();
/*
* Make the scheduler instance visible. Must be inside cpus_read_lock().
* See handle_hotplug().
*/
rcu_assign_pointer(scx_root, sch);
scx_idle_enable(ops);
if (sch->ops.init) {
ret = SCX_CALL_OP_RET(sch, SCX_KF_UNLOCKED, init, NULL);
if (ret) {
ret = ops_sanitize_err(sch, "init", ret);
cpus_read_unlock();
scx_error(sch, "ops.init() failed (%d)", ret);
goto err_disable;
}
}
for (i = SCX_OPI_CPU_HOTPLUG_BEGIN; i < SCX_OPI_CPU_HOTPLUG_END; i++)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
/*
* Once __scx_enabled is set, %current can be switched to SCX anytime.
* This can lead to stalls as some BPF schedulers (e.g. userspace
* scheduling) may not function correctly before all tasks are switched.
* Init in bypass mode to guarantee forward progress.
*/
scx_bypass(true);
for (i = SCX_OPI_NORMAL_BEGIN; i < SCX_OPI_NORMAL_END; i++)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);
if (sch->ops.cpu_acquire || sch->ops.cpu_release)
sch->ops.flags |= SCX_OPS_HAS_CPU_PREEMPT;
/*
* Lock out forks, cgroup on/offlining and moves before opening the
* floodgate so that they don't wander into the operations prematurely.
*/
percpu_down_write(&scx_fork_rwsem);
/*
* Enable ops for every task. Fork is excluded by scx_fork_rwsem
* preventing new tasks from being added. No need to exclude tasks
* leaving as sched_ext_free() can handle both prepped and enabled
* tasks. Prep all tasks first and then enable them with preemption
* disabled.
*
* All cgroups should be initialized before scx_init_task() so that the
* BPF scheduler can reliably track each task's cgroup membership from
* scx_init_task(). Lock out cgroup on/offlining and task migrations
* while tasks are being initialized so that scx_cgroup_can_attach()
* never sees uninitialized tasks.
*/
scx_cgroup_lock();
ret = scx_cgroup_init(sch);
if (ret)
goto err_disable_unlock_all;
scx_task_iter_start(&sti);
while ((p = scx_task_iter_next_locked(&sti))) {
/*
* @p may already be dead, have lost all its usages counts and
* be waiting for RCU grace period before being freed. @p can't
* be initialized for SCX in such cases and should be ignored.
*/
if (!tryget_task_struct(p))
continue;
scx_task_iter_unlock(&sti);
ret = scx_init_task(p, task_group(p), false);
if (ret) {
put_task_struct(p);
scx_task_iter_relock(&sti);
scx_task_iter_stop(&sti);
scx_error(sch, "ops.init_task() failed (%d) for %s[%d]",
ret, p->comm, p->pid);
goto err_disable_unlock_all;
}
/*
* All tasks are READY. It's safe to turn on scx_enabled() and switch
* all eligible tasks.
*/
WRITE_ONCE(scx_switching_all, !(ops->flags & SCX_OPS_SWITCH_PARTIAL));
static_branch_enable(&__scx_enabled);
/*
* We're fully committed and can't fail. The task READY -> ENABLED
* transitions here are synchronized against sched_ext_free() through
* scx_tasks_lock.
*/
percpu_down_write(&scx_fork_rwsem);
scx_task_iter_start(&sti);
while ((p = scx_task_iter_next_locked(&sti))) {
const struct sched_class *old_class = p->sched_class;
const struct sched_class *new_class =
__setscheduler_class(p->policy, p->prio);
struct sched_enq_and_set_ctx ctx;
err_disable_unlock_all:
scx_cgroup_unlock();
percpu_up_write(&scx_fork_rwsem);
/* we'll soon enter disable path, keep bypass on */
err_disable:
mutex_unlock(&scx_enable_mutex);
/*
* Returning an error code here would not pass all the error information
* to userspace. Record errno using scx_error() for cases scx_error()
* wasn't already invoked and exit indicating success so that the error
* is notified through ops.exit() with all the details.
*
* Flush scx_disable_work to ensure that error is reported before init
* completion. sch's base reference will be put by bpf_scx_unreg().
*/
scx_error(sch, "scx_enable() failed (%d)", ret);
kthread_flush_work(&sch->disable_work);
return 0;
}
switch (moff) {
case offsetof(struct sched_ext_ops, init_task):
#ifdef CONFIG_EXT_GROUP_SCHED
case offsetof(struct sched_ext_ops, cgroup_init):
case offsetof(struct sched_ext_ops, cgroup_exit):
case offsetof(struct sched_ext_ops, cgroup_prep_move):
#endif
case offsetof(struct sched_ext_ops, cpu_online):
case offsetof(struct sched_ext_ops, cpu_offline):
case offsetof(struct sched_ext_ops, init):
case offsetof(struct sched_ext_ops, exit):
break;
default:
if (prog->sleepable)
return -EINVAL;
}
static int bpf_scx_init(struct btf *btf)
{
task_struct_type = btf_type_by_id(btf, btf_tracing_ids[BTF_TRACING_TYPE_TASK]);
return 0;
}
static int bpf_scx_update(void *kdata, void *old_kdata, struct bpf_link *link)
{
/*
* sched_ext does not support updating the actively-loaded BPF
* scheduler, as registering a BPF scheduler can always fail if the
* scheduler returns an error code for e.g. ops.init(), ops.init_task(),
* etc. Similarly, we can always race with unregistration happening
* elsewhere, such as with sysrq.
*/
return -EOPNOTSUPP;
}
static int bpf_scx_validate(void *kdata)
{
return 0;
}
/*
* We can skip idle kicking if @rq is going to go through at least one
* full SCX scheduling cycle before going idle. Just checking whether
* curr is not idle is insufficient because we could be racing
* balance_one() trying to pull the next task from a remote rq, which
* may fail, and @rq may become idle afterwards.
*
* The race window is small and we don't and can't guarantee that @rq is
* only kicked while idle anyway. Skip only when sure.
*/
return !is_idle_task(rq->curr) && !(rq->scx.flags & SCX_RQ_IN_BALANCE);
}
/*
* During CPU hotplug, a CPU may depend on kicking itself to make
* forward progress. Allow kicking self regardless of online state.
*/
if (cpu_online(cpu) || cpu == cpu_of(this_rq)) {
if (cpumask_test_cpu(cpu, this_scx->cpus_to_preempt)) {
if (rq->curr->sched_class == &ext_sched_class)
rq->curr->scx.slice = 0;
cpumask_clear_cpu(cpu, this_scx->cpus_to_preempt);
}
for_each_cpu(cpu, this_scx->cpus_to_wait) {
unsigned long *wait_pnt_seq = &cpu_rq(cpu)->scx.pnt_seq;
if (cpu != cpu_of(this_rq)) {
/*
* Pairs with smp_store_release() issued by this CPU in
* switch_class() on the resched path.
*
* We busy-wait here to guarantee that no other task can
* be scheduled on our core before the target CPU has
* entered the resched path.
*/
while (smp_load_acquire(wait_pnt_seq) == pseqs[cpu])
cpu_relax();
}
/**
* print_scx_info - print out sched_ext scheduler state
* @log_lvl: the log level to use when printing
* @p: target task
*
* If a sched_ext scheduler is enabled, print the name and state of the
* scheduler. If @p is on sched_ext, print further information about the task.
*
* This function can be safely called on any task as long as the task_struct
* itself is accessible. While safe, this function isn't synchronized and may
* print out mixups or garbages of limited length.
*/
void print_scx_info(const char *log_lvl, struct task_struct *p)
{
struct scx_sched *sch = scx_root;
enum scx_enable_state state = scx_enable_state();
const char *all = READ_ONCE(scx_switching_all) ? "+all" : "";
char runnable_at_buf[22] = "?";
struct sched_class *class;
unsigned long runnable_at;
if (state == SCX_DISABLED)
return;
/*
* Carefully check if the task was running on sched_ext, and then
* carefully copy the time it's been runnable, and its state.
*/
if (copy_from_kernel_nofault(&class, &p->sched_class, sizeof(class)) ||
class != &ext_sched_class) {
printk("%sSched_ext: %s (%s%s)", log_lvl, sch->ops.name,
scx_enable_state_str[state], all);
return;
}
if (!copy_from_kernel_nofault(&runnable_at, &p->scx.runnable_at,
sizeof(runnable_at)))
scnprintf(runnable_at_buf, sizeof(runnable_at_buf), "%+ldms",
jiffies_delta_msecs(runnable_at, jiffies));
/* print everything onto one line to conserve console space */
printk("%sSched_ext: %s (%s%s), task: runnable_at=%s",
log_lvl, sch->ops.name, scx_enable_state_str[state], all,
runnable_at_buf);
}
static int scx_pm_handler(struct notifier_block *nb, unsigned long event, void *ptr)
{
/*
* SCX schedulers often have userspace components which are sometimes
* involved in critial scheduling paths. PM operations involve freezing
* userspace which can lead to scheduling misbehaviors including stalls.
* Let's bypass while PM operations are in progress.
*/
switch (event) {
case PM_HIBERNATION_PREPARE:
case PM_SUSPEND_PREPARE:
case PM_RESTORE_PREPARE:
scx_bypass(true);
break;
case PM_POST_HIBERNATION:
case PM_POST_SUSPEND:
case PM_POST_RESTORE:
scx_bypass(false);
break;
}
/*
* The following is to prevent the compiler from optimizing out the enum
* definitions so that BPF scheduler implementations can use them
* through the generated vmlinux.h.
*/
WRITE_ONCE(v, SCX_ENQ_WAKEUP | SCX_DEQ_SLEEP | SCX_KICK_PREEMPT |
SCX_TG_ONLINE);
/********************************************************************************
* Helpers that can be called from the BPF scheduler.
*/
static bool scx_dsq_insert_preamble(struct task_struct *p, u64 enq_flags)
{
if (!scx_kf_allowed(SCX_KF_ENQUEUE | SCX_KF_DISPATCH))
return false;
lockdep_assert_irqs_disabled();
if (unlikely(!p)) {
scx_kf_error("called with NULL task");
return false;
}
/**
* scx_bpf_dsq_insert - Insert a task into the FIFO queue of a DSQ
* @p: task_struct to insert
* @dsq_id: DSQ to insert into
* @slice: duration @p can run for in nsecs, 0 to keep the current value
* @enq_flags: SCX_ENQ_*
*
* Insert @p into the FIFO queue of the DSQ identified by @dsq_id. It is safe to
* call this function spuriously. Can be called from ops.enqueue(),
* ops.select_cpu(), and ops.dispatch().
*
* When called from ops.select_cpu() or ops.enqueue(), it's for direct dispatch
* and @p must match the task being enqueued.
*
* When called from ops.select_cpu(), @enq_flags and @dsp_id are stored, and @p
* will be directly inserted into the corresponding dispatch queue after
* ops.select_cpu() returns. If @p is inserted into SCX_DSQ_LOCAL, it will be
* inserted into the local DSQ of the CPU returned by ops.select_cpu().
* @enq_flags are OR'd with the enqueue flags on the enqueue path before the
* task is inserted.
*
* When called from ops.dispatch(), there are no restrictions on @p or @dsq_id
* and this function can be called upto ops.dispatch_max_batch times to insert
* multiple tasks. scx_bpf_dispatch_nr_slots() returns the number of the
* remaining slots. scx_bpf_dsq_move_to_local() flushes the batch and resets the
* counter.
*
* This function doesn't have any locking restrictions and may be called under
* BPF locks (in the future when BPF introduces more flexible locking).
*
* @p is allowed to run for @slice. The scheduling path is triggered on slice
* exhaustion. If zero, the current residual slice is maintained. If
* %SCX_SLICE_INF, @p never expires and the BPF scheduler must kick the CPU with
* scx_bpf_kick_cpu() to trigger scheduling.
*/
__bpf_kfunc void scx_bpf_dsq_insert(struct task_struct *p, u64 dsq_id, u64 slice,
u64 enq_flags)
{
if (!scx_dsq_insert_preamble(p, enq_flags))
return;
/**
* scx_bpf_dsq_insert_vtime - Insert a task into the vtime priority queue of a DSQ
* @p: task_struct to insert
* @dsq_id: DSQ to insert into
* @slice: duration @p can run for in nsecs, 0 to keep the current value
* @vtime: @p's ordering inside the vtime-sorted queue of the target DSQ
* @enq_flags: SCX_ENQ_*
*
* Insert @p into the vtime priority queue of the DSQ identified by @dsq_id.
* Tasks queued into the priority queue are ordered by @vtime. All other aspects
* are identical to scx_bpf_dsq_insert().
*
* @vtime ordering is according to time_before64() which considers wrapping. A
* numerically larger vtime may indicate an earlier position in the ordering and
* vice-versa.
*
* A DSQ can only be used as a FIFO or priority queue at any given time and this
* function must not be called on a DSQ which already has one or more FIFO tasks
* queued and vice-versa. Also, the built-in DSQs (SCX_DSQ_LOCAL and
* SCX_DSQ_GLOBAL) cannot be used as priority queues.
*/
__bpf_kfunc void scx_bpf_dsq_insert_vtime(struct task_struct *p, u64 dsq_id,
u64 slice, u64 vtime, u64 enq_flags)
{
if (!scx_dsq_insert_preamble(p, enq_flags))
return;
if (!scx_kf_allowed_if_unlocked() && !scx_kf_allowed(SCX_KF_DISPATCH))
return false;
/*
* Can be called from either ops.dispatch() locking this_rq() or any
* context where no rq lock is held. If latter, lock @p's task_rq which
* we'll likely need anyway.
*/
src_rq = task_rq(p);
if (in_balance) {
if (this_rq != src_rq) {
raw_spin_rq_unlock(this_rq);
raw_spin_rq_lock(src_rq);
}
} else {
raw_spin_rq_lock(src_rq);
}
/*
* If the BPF scheduler keeps calling this function repeatedly, it can
* cause similar live-lock conditions as consume_dispatch_q(). Insert a
* breather if necessary.
*/
scx_breather(src_rq);
/*
* Did someone else get to it? @p could have already left $src_dsq, got
* re-enqueud, or be in the process of being consumed by someone else.
*/
if (unlikely(p->scx.dsq != src_dsq ||
u32_before(kit->cursor.priv, p->scx.dsq_seq) ||
p->scx.holding_cpu >= 0) ||
WARN_ON_ONCE(src_rq != task_rq(p))) {
raw_spin_unlock(&src_dsq->lock);
goto out;
}
/* @p is still on $src_dsq and stable, determine the destination */
dst_dsq = find_dsq_for_dispatch(sch, this_rq, dsq_id, p);
/*
* Apply vtime and slice updates before moving so that the new time is
* visible before inserting into $dst_dsq. @p is still on $src_dsq but
* this is safe as we're locking it.
*/
if (kit->cursor.flags & __SCX_DSQ_ITER_HAS_VTIME)
p->scx.dsq_vtime = kit->vtime;
if (kit->cursor.flags & __SCX_DSQ_ITER_HAS_SLICE)
p->scx.slice = kit->slice;
/**
* scx_bpf_dispatch_nr_slots - Return the number of remaining dispatch slots
*
* Can only be called from ops.dispatch().
*/
__bpf_kfunc u32 scx_bpf_dispatch_nr_slots(void)
{
if (!scx_kf_allowed(SCX_KF_DISPATCH))
return 0;
/**
* scx_bpf_dispatch_cancel - Cancel the latest dispatch
*
* Cancel the latest dispatch. Can be called multiple times to cancel further
* dispatches. Can only be called from ops.dispatch().
*/
__bpf_kfunc void scx_bpf_dispatch_cancel(void)
{
struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
if (!scx_kf_allowed(SCX_KF_DISPATCH))
return;
if (dspc->cursor > 0)
dspc->cursor--;
else
scx_kf_error("dispatch buffer underflow");
}
/**
* scx_bpf_dsq_move_to_local - move a task from a DSQ to the current CPU's local DSQ
* @dsq_id: DSQ to move task from
*
* Move a task from the non-local DSQ identified by @dsq_id to the current CPU's
* local DSQ for execution. Can only be called from ops.dispatch().
*
* This function flushes the in-flight dispatches from scx_bpf_dsq_insert()
* before trying to move from the specified DSQ. It may also grab rq locks and
* thus can't be called under any BPF locks.
*
* Returns %true if a task has been moved, %false if there isn't any task to
* move.
*/
__bpf_kfunc bool scx_bpf_dsq_move_to_local(u64 dsq_id)
{
struct scx_sched *sch = scx_root;
struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
struct scx_dispatch_q *dsq;
if (!scx_kf_allowed(SCX_KF_DISPATCH))
return false;
flush_dispatch_buf(sch, dspc->rq);
dsq = find_user_dsq(sch, dsq_id);
if (unlikely(!dsq)) {
scx_error(sch, "invalid DSQ ID 0x%016llx", dsq_id);
return false;
}
if (consume_dispatch_q(sch, dspc->rq, dsq)) {
/*
* A successfully consumed task can be dequeued before it starts
* running while the CPU is trying to migrate other dispatched
* tasks. Bump nr_tasks to tell balance_scx() to retry on empty
* local DSQ.
*/
dspc->nr_tasks++;
return true;
} else {
return false;
}
}
/**
* scx_bpf_dsq_move_set_slice - Override slice when moving between DSQs
* @it__iter: DSQ iterator in progress
* @slice: duration the moved task can run for in nsecs
*
* Override the slice of the next task that will be moved from @it__iter using
* scx_bpf_dsq_move[_vtime](). If this function is not called, the previous
* slice duration is kept.
*/
__bpf_kfunc void scx_bpf_dsq_move_set_slice(struct bpf_iter_scx_dsq *it__iter,
u64 slice)
{
struct bpf_iter_scx_dsq_kern *kit = (void *)it__iter;
/**
* scx_bpf_dsq_move_set_vtime - Override vtime when moving between DSQs
* @it__iter: DSQ iterator in progress
* @vtime: task's ordering inside the vtime-sorted queue of the target DSQ
*
* Override the vtime of the next task that will be moved from @it__iter using
* scx_bpf_dsq_move_vtime(). If this function is not called, the previous slice
* vtime is kept. If scx_bpf_dsq_move() is used to dispatch the next task, the
* override is ignored and cleared.
*/
__bpf_kfunc void scx_bpf_dsq_move_set_vtime(struct bpf_iter_scx_dsq *it__iter,
u64 vtime)
{
struct bpf_iter_scx_dsq_kern *kit = (void *)it__iter;
/**
* scx_bpf_dsq_move - Move a task from DSQ iteration to a DSQ
* @it__iter: DSQ iterator in progress
* @p: task to transfer
* @dsq_id: DSQ to move @p to
* @enq_flags: SCX_ENQ_*
*
* Transfer @p which is on the DSQ currently iterated by @it__iter to the DSQ
* specified by @dsq_id. All DSQs - local DSQs, global DSQ and user DSQs - can
* be the destination.
*
* For the transfer to be successful, @p must still be on the DSQ and have been
* queued before the DSQ iteration started. This function doesn't care whether
* @p was obtained from the DSQ iteration. @p just has to be on the DSQ and have
* been queued before the iteration started.
*
* @p's slice is kept by default. Use scx_bpf_dsq_move_set_slice() to update.
*
* Can be called from ops.dispatch() or any BPF context which doesn't hold a rq
* lock (e.g. BPF timers or SYSCALL programs).
*
* Returns %true if @p has been consumed, %false if @p had already been consumed
* or dequeued.
*/
__bpf_kfunc bool scx_bpf_dsq_move(struct bpf_iter_scx_dsq *it__iter,
struct task_struct *p, u64 dsq_id,
u64 enq_flags)
{
return scx_dsq_move((struct bpf_iter_scx_dsq_kern *)it__iter,
p, dsq_id, enq_flags);
}
/**
* scx_bpf_dsq_move_vtime - Move a task from DSQ iteration to a PRIQ DSQ
* @it__iter: DSQ iterator in progress
* @p: task to transfer
* @dsq_id: DSQ to move @p to
* @enq_flags: SCX_ENQ_*
*
* Transfer @p which is on the DSQ currently iterated by @it__iter to the
* priority queue of the DSQ specified by @dsq_id. The destination must be a
* user DSQ as only user DSQs support priority queue.
*
* @p's slice and vtime are kept by default. Use scx_bpf_dsq_move_set_slice()
* and scx_bpf_dsq_move_set_vtime() to update.
*
* All other aspects are identical to scx_bpf_dsq_move(). See
* scx_bpf_dsq_insert_vtime() for more information on @vtime.
*/
__bpf_kfunc bool scx_bpf_dsq_move_vtime(struct bpf_iter_scx_dsq *it__iter,
struct task_struct *p, u64 dsq_id,
u64 enq_flags)
{
return scx_dsq_move((struct bpf_iter_scx_dsq_kern *)it__iter,
p, dsq_id, enq_flags | SCX_ENQ_DSQ_PRIQ);
}
/**
* scx_bpf_reenqueue_local - Re-enqueue tasks on a local DSQ
*
* Iterate over all of the tasks currently enqueued on the local DSQ of the
* caller's CPU, and re-enqueue them in the BPF scheduler. Returns the number of
* processed tasks. Can only be called from ops.cpu_release().
*/
__bpf_kfunc u32 scx_bpf_reenqueue_local(void)
{
LIST_HEAD(tasks);
u32 nr_enqueued = 0;
struct rq *rq;
struct task_struct *p, *n;
if (!scx_kf_allowed(SCX_KF_CPU_RELEASE))
return 0;
/*
* The BPF scheduler may choose to dispatch tasks back to
* @rq->scx.local_dsq. Move all candidate tasks off to a private list
* first to avoid processing the same tasks repeatedly.
*/
list_for_each_entry_safe(p, n, &rq->scx.local_dsq.list,
scx.dsq_list.node) {
/*
* If @p is being migrated, @p's current CPU may not agree with
* its allowed CPUs and the migration_cpu_stop is about to
* deactivate and re-activate @p anyway. Skip re-enqueueing.
*
* While racing sched property changes may also dequeue and
* re-enqueue a migrating task while its current CPU and allowed
* CPUs disagree, they use %ENQUEUE_RESTORE which is bypassed to
* the current local DSQ for running tasks and thus are not
* visible to the BPF scheduler.
*/
if (p->migration_pending)
continue;
/**
* scx_bpf_create_dsq - Create a custom DSQ
* @dsq_id: DSQ to create
* @node: NUMA node to allocate from
*
* Create a custom DSQ identified by @dsq_id. Can be called from any sleepable
* scx callback, and any BPF_PROG_TYPE_SYSCALL prog.
*/
__bpf_kfunc s32 scx_bpf_create_dsq(u64 dsq_id, s32 node)
{
struct scx_dispatch_q *dsq;
struct scx_sched *sch;
s32 ret;
/**
* scx_bpf_kick_cpu - Trigger reschedule on a CPU
* @cpu: cpu to kick
* @flags: %SCX_KICK_* flags
*
* Kick @cpu into rescheduling. This can be used to wake up an idle CPU or
* trigger rescheduling on a busy CPU. This can be called from any online
* scx_ops operation and the actual kicking is performed asynchronously through
* an irq work.
*/
__bpf_kfunc void scx_bpf_kick_cpu(s32 cpu, u64 flags)
{
struct rq *this_rq;
unsigned long irq_flags;
if (!kf_cpu_valid(cpu, NULL))
return;
local_irq_save(irq_flags);
this_rq = this_rq();
/*
* While bypassing for PM ops, IRQ handling may not be online which can
* lead to irq_work_queue() malfunction such as infinite busy wait for
* IRQ status update. Suppress kicking.
*/
if (scx_rq_bypassing(this_rq))
goto out;
/*
* Actual kicking is bounced to kick_cpus_irq_workfn() to avoid nesting
* rq locks. We can probably be smarter and avoid bouncing if called
* from ops which don't hold a rq lock.
*/
if (flags & SCX_KICK_IDLE) {
struct rq *target_rq = cpu_rq(cpu);
if (unlikely(flags & (SCX_KICK_PREEMPT | SCX_KICK_WAIT)))
scx_kf_error("PREEMPT/WAIT cannot be used with SCX_KICK_IDLE");
if (raw_spin_rq_trylock(target_rq)) {
if (can_skip_idle_kick(target_rq)) {
raw_spin_rq_unlock(target_rq);
goto out;
}
raw_spin_rq_unlock(target_rq);
}
cpumask_set_cpu(cpu, this_rq->scx.cpus_to_kick_if_idle);
} else {
cpumask_set_cpu(cpu, this_rq->scx.cpus_to_kick);
if (flags & SCX_KICK_PREEMPT)
cpumask_set_cpu(cpu, this_rq->scx.cpus_to_preempt);
if (flags & SCX_KICK_WAIT)
cpumask_set_cpu(cpu, this_rq->scx.cpus_to_wait);
}
/**
* scx_bpf_dsq_nr_queued - Return the number of queued tasks
* @dsq_id: id of the DSQ
*
* Return the number of tasks in the DSQ matching @dsq_id. If not found,
* -%ENOENT is returned.
*/
__bpf_kfunc s32 scx_bpf_dsq_nr_queued(u64 dsq_id)
{
struct scx_sched *sch;
struct scx_dispatch_q *dsq;
s32 ret;
preempt_disable();
sch = rcu_dereference_sched(scx_root);
if (unlikely(!sch)) {
ret = -ENODEV;
goto out;
}
if (dsq_id == SCX_DSQ_LOCAL) {
ret = READ_ONCE(this_rq()->scx.local_dsq.nr);
goto out;
} else if ((dsq_id & SCX_DSQ_LOCAL_ON) == SCX_DSQ_LOCAL_ON) {
s32 cpu = dsq_id & SCX_DSQ_LOCAL_CPU_MASK;
if (ops_cpu_valid(sch, cpu, NULL)) {
ret = READ_ONCE(cpu_rq(cpu)->scx.local_dsq.nr);
goto out;
}
} else {
dsq = find_user_dsq(sch, dsq_id);
if (dsq) {
ret = READ_ONCE(dsq->nr);
goto out;
}
}
ret = -ENOENT;
out:
preempt_enable();
return ret;
}
/**
* scx_bpf_destroy_dsq - Destroy a custom DSQ
* @dsq_id: DSQ to destroy
*
* Destroy the custom DSQ identified by @dsq_id. Only DSQs created with
* scx_bpf_create_dsq() can be destroyed. The caller must ensure that the DSQ is
* empty and no further tasks are dispatched to it. Ignored if called on a DSQ
* which doesn't exist. Can be called from any online scx_ops operations.
*/
__bpf_kfunc void scx_bpf_destroy_dsq(u64 dsq_id)
{
struct scx_sched *sch;
rcu_read_lock();
sch = rcu_dereference(scx_root);
if (sch)
destroy_dsq(sch, dsq_id);
rcu_read_unlock();
}
/**
* bpf_iter_scx_dsq_new - Create a DSQ iterator
* @it: iterator to initialize
* @dsq_id: DSQ to iterate
* @flags: %SCX_DSQ_ITER_*
*
* Initialize BPF iterator @it which can be used with bpf_for_each() to walk
* tasks in the DSQ specified by @dsq_id. Iteration using @it only includes
* tasks which are already queued when this function is invoked.
*/
__bpf_kfunc int bpf_iter_scx_dsq_new(struct bpf_iter_scx_dsq *it, u64 dsq_id,
u64 flags)
{
struct bpf_iter_scx_dsq_kern *kit = (void *)it;
struct scx_sched *sch;
/**
* bpf_iter_scx_dsq_next - Progress a DSQ iterator
* @it: iterator to progress
*
* Return the next task. See bpf_iter_scx_dsq_new().
*/
__bpf_kfunc struct task_struct *bpf_iter_scx_dsq_next(struct bpf_iter_scx_dsq *it)
{
struct bpf_iter_scx_dsq_kern *kit = (void *)it;
bool rev = kit->cursor.flags & SCX_DSQ_ITER_REV;
struct task_struct *p;
unsigned long flags;
if (!kit->dsq)
return NULL;
raw_spin_lock_irqsave(&kit->dsq->lock, flags);
if (list_empty(&kit->cursor.node))
p = NULL;
else
p = container_of(&kit->cursor, struct task_struct, scx.dsq_list);
/*
* Only tasks which were queued before the iteration started are
* visible. This bounds BPF iterations and guarantees that vtime never
* jumps in the other direction while iterating.
*/
do {
p = nldsq_next_task(kit->dsq, p, rev);
} while (p && unlikely(u32_before(kit->cursor.priv, p->scx.dsq_seq)));
if (p) {
if (rev)
list_move_tail(&kit->cursor.node, &p->scx.dsq_list.node);
else
list_move(&kit->cursor.node, &p->scx.dsq_list.node);
} else {
list_del_init(&kit->cursor.node);
}
/**
* scx_bpf_exit_bstr - Gracefully exit the BPF scheduler.
* @exit_code: Exit value to pass to user space via struct scx_exit_info.
* @fmt: error message format string
* @data: format string parameters packaged using ___bpf_fill() macro
* @data__sz: @data len, must end in '__sz' for the verifier
*
* Indicate that the BPF scheduler wants to exit gracefully, and initiate ops
* disabling.
*/
__bpf_kfunc void scx_bpf_exit_bstr(s64 exit_code, char *fmt,
unsigned long long *data, u32 data__sz)
{
unsigned long flags;
/**
* scx_bpf_error_bstr - Indicate fatal error
* @fmt: error message format string
* @data: format string parameters packaged using ___bpf_fill() macro
* @data__sz: @data len, must end in '__sz' for the verifier
*
* Indicate that the BPF scheduler encountered a fatal error and initiate ops
* disabling.
*/
__bpf_kfunc void scx_bpf_error_bstr(char *fmt, unsigned long long *data,
u32 data__sz)
{
unsigned long flags;
/**
* scx_bpf_dump_bstr - Generate extra debug dump specific to the BPF scheduler
* @fmt: format string
* @data: format string parameters packaged using ___bpf_fill() macro
* @data__sz: @data len, must end in '__sz' for the verifier
*
* To be called through scx_bpf_dump() helper from ops.dump(), dump_cpu() and
* dump_task() to generate extra debug dump specific to the BPF scheduler.
*
* The extra dump may be multiple lines. A single line may be split over
* multiple calls. The last line is automatically terminated.
*/
__bpf_kfunc void scx_bpf_dump_bstr(char *fmt, unsigned long long *data,
u32 data__sz)
{
struct scx_dump_data *dd = &scx_dump_data;
struct scx_bstr_buf *buf = &dd->buf;
s32 ret;
if (raw_smp_processor_id() != dd->cpu) {
scx_kf_error("scx_bpf_dump() must only be called from ops.dump() and friends");
return;
}
/* append the formatted string to the line buf */
ret = __bstr_format(buf->data, buf->line + dd->cursor,
sizeof(buf->line) - dd->cursor, fmt, data, data__sz);
if (ret < 0) {
dump_line(dd->s, "%s[!] (\"%s\", %p, %u) failed to format (%d)",
dd->prefix, fmt, data, data__sz, ret);
return;
}
/*
* If the line buf overflowed or ends in a newline, flush it into the
* dump. This is to allow the caller to generate a single line over
* multiple calls. As ops_dump_flush() can also handle multiple lines in
* the line buf, the only case which can lead to an unexpected
* truncation is when the caller keeps generating newlines in the middle
* instead of the end consecutively. Don't do that.
*/
if (dd->cursor >= sizeof(buf->line) || buf->line[dd->cursor - 1] == '\n')
ops_dump_flush();
}
/**
* scx_bpf_cpuperf_cap - Query the maximum relative capacity of a CPU
* @cpu: CPU of interest
*
* Return the maximum relative capacity of @cpu in relation to the most
* performant CPU in the system. The return value is in the range [1,
* %SCX_CPUPERF_ONE]. See scx_bpf_cpuperf_cur().
*/
__bpf_kfunc u32 scx_bpf_cpuperf_cap(s32 cpu)
{
if (kf_cpu_valid(cpu, NULL))
return arch_scale_cpu_capacity(cpu);
else
return SCX_CPUPERF_ONE;
}
/**
* scx_bpf_cpuperf_cur - Query the current relative performance of a CPU
* @cpu: CPU of interest
*
* Return the current relative performance of @cpu in relation to its maximum.
* The return value is in the range [1, %SCX_CPUPERF_ONE].
*
* The current performance level of a CPU in relation to the maximum performance
* available in the system can be calculated as follows:
*
* scx_bpf_cpuperf_cap() * scx_bpf_cpuperf_cur() / %SCX_CPUPERF_ONE
*
* The result is in the range [1, %SCX_CPUPERF_ONE].
*/
__bpf_kfunc u32 scx_bpf_cpuperf_cur(s32 cpu)
{
if (kf_cpu_valid(cpu, NULL))
return arch_scale_freq_capacity(cpu);
else
return SCX_CPUPERF_ONE;
}
/**
* scx_bpf_cpuperf_set - Set the relative performance target of a CPU
* @cpu: CPU of interest
* @perf: target performance level [0, %SCX_CPUPERF_ONE]
*
* Set the target performance level of @cpu to @perf. @perf is in linear
* relative scale between 0 and %SCX_CPUPERF_ONE. This determines how the
* schedutil cpufreq governor chooses the target frequency.
*
* The actual performance level chosen, CPU grouping, and the overhead and
* latency of the operations are dependent on the hardware and cpufreq driver in
* use. Consult hardware and cpufreq documentation for more information. The
* current performance level can be monitored using scx_bpf_cpuperf_cur().
*/
__bpf_kfunc void scx_bpf_cpuperf_set(s32 cpu, u32 perf)
{
if (unlikely(perf > SCX_CPUPERF_ONE)) {
scx_kf_error("Invalid cpuperf target %u for CPU %d", perf, cpu);
return;
}
/*
* When called with an rq lock held, restrict the operation
* to the corresponding CPU to prevent ABBA deadlocks.
*/
if (locked_rq && rq != locked_rq) {
scx_kf_error("Invalid target CPU %d", cpu);
return;
}
/*
* If no rq lock is held, allow to operate on any CPU by
* acquiring the corresponding rq lock.
*/
if (!locked_rq) {
rq_lock_irqsave(rq, &rf);
update_rq_clock(rq);
}
if (!locked_rq)
rq_unlock_irqrestore(rq, &rf);
}
}
/**
* scx_bpf_nr_node_ids - Return the number of possible node IDs
*
* All valid node IDs in the system are smaller than the returned value.
*/
__bpf_kfunc u32 scx_bpf_nr_node_ids(void)
{
return nr_node_ids;
}
/**
* scx_bpf_nr_cpu_ids - Return the number of possible CPU IDs
*
* All valid CPU IDs in the system are smaller than the returned value.
*/
__bpf_kfunc u32 scx_bpf_nr_cpu_ids(void)
{
return nr_cpu_ids;
}
/**
* scx_bpf_get_possible_cpumask - Get a referenced kptr to cpu_possible_mask
*/
__bpf_kfunc const struct cpumask *scx_bpf_get_possible_cpumask(void)
{
return cpu_possible_mask;
}
/**
* scx_bpf_get_online_cpumask - Get a referenced kptr to cpu_online_mask
*/
__bpf_kfunc const struct cpumask *scx_bpf_get_online_cpumask(void)
{
return cpu_online_mask;
}
/**
* scx_bpf_put_cpumask - Release a possible/online cpumask
* @cpumask: cpumask to release
*/
__bpf_kfunc void scx_bpf_put_cpumask(const struct cpumask *cpumask)
{
/*
* Empty function body because we aren't actually acquiring or releasing
* a reference to a global cpumask, which is read-only in the caller and
* is never released. The acquire / release semantics here are just used
* to make the cpumask is a trusted pointer in the caller.
*/
}
/**
* scx_bpf_task_running - Is task currently running?
* @p: task of interest
*/
__bpf_kfunc bool scx_bpf_task_running(const struct task_struct *p)
{
return task_rq(p)->curr == p;
}
/**
* scx_bpf_task_cpu - CPU a task is currently associated with
* @p: task of interest
*/
__bpf_kfunc s32 scx_bpf_task_cpu(const struct task_struct *p)
{
return task_cpu(p);
}
/**
* scx_bpf_cpu_rq - Fetch the rq of a CPU
* @cpu: CPU of the rq
*/
__bpf_kfunc struct rq *scx_bpf_cpu_rq(s32 cpu)
{
if (!kf_cpu_valid(cpu, NULL))
return NULL;
return cpu_rq(cpu);
}
/**
* scx_bpf_task_cgroup - Return the sched cgroup of a task
* @p: task of interest
*
* @p->sched_task_group->css.cgroup represents the cgroup @p is associated with
* from the scheduler's POV. SCX operations should use this function to
* determine @p's current cgroup as, unlike following @p->cgroups,
* @p->sched_task_group is protected by @p's rq lock and thus atomic w.r.t. all
* rq-locked operations. Can be called on the parameter tasks of rq-locked
* operations. The restriction guarantees that @p's rq is locked by the caller.
*/
#ifdef CONFIG_CGROUP_SCHED
__bpf_kfunc struct cgroup *scx_bpf_task_cgroup(struct task_struct *p)
{
struct task_group *tg = p->sched_task_group;
struct cgroup *cgrp = &cgrp_dfl_root.cgrp;
if (!scx_kf_allowed_on_arg_tasks(__SCX_KF_RQ_LOCKED, p))
goto out;
cgrp = tg_cgrp(tg);
out:
cgroup_get(cgrp);
return cgrp;
}
#endif
/**
* scx_bpf_now - Returns a high-performance monotonically non-decreasing
* clock for the current CPU. The clock returned is in nanoseconds.
*
* It provides the following properties:
*
* 1) High performance: Many BPF schedulers call bpf_ktime_get_ns() frequently
* to account for execution time and track tasks' runtime properties.
* Unfortunately, in some hardware platforms, bpf_ktime_get_ns() -- which
* eventually reads a hardware timestamp counter -- is neither performant nor
* scalable. scx_bpf_now() aims to provide a high-performance clock by
* using the rq clock in the scheduler core whenever possible.
*
* 2) High enough resolution for the BPF scheduler use cases: In most BPF
* scheduler use cases, the required clock resolution is lower than the most
* accurate hardware clock (e.g., rdtsc in x86). scx_bpf_now() basically
* uses the rq clock in the scheduler core whenever it is valid. It considers
* that the rq clock is valid from the time the rq clock is updated
* (update_rq_clock) until the rq is unlocked (rq_unpin_lock).
*
* 3) Monotonically non-decreasing clock for the same CPU: scx_bpf_now()
* guarantees the clock never goes backward when comparing them in the same
* CPU. On the other hand, when comparing clocks in different CPUs, there
* is no such guarantee -- the clock can go backward. It provides a
* monotonically *non-decreasing* clock so that it would provide the same
* clock values in two different scx_bpf_now() calls in the same CPU
* during the same period of when the rq clock is valid.
*/
__bpf_kfunc u64 scx_bpf_now(void)
{
struct rq *rq;
u64 clock;
preempt_disable();
rq = this_rq();
if (smp_load_acquire(&rq->scx.flags) & SCX_RQ_CLK_VALID) {
/*
* If the rq clock is valid, use the cached rq clock.
*
* Note that scx_bpf_now() is re-entrant between a process
* context and an interrupt context (e.g., timer interrupt).
* However, we don't need to consider the race between them
* because such race is not observable from a caller.
*/
clock = READ_ONCE(rq->scx.clock);
} else {
/*
* Otherwise, return a fresh rq clock.
*
* The rq clock is updated outside of the rq lock.
* In this case, keep the updated rq clock invalid so the next
* kfunc call outside the rq lock gets a fresh rq clock.
*/
clock = sched_clock_cpu(cpu_of(rq));
}
/*
* scx_bpf_events - Get a system-wide event counter to
* @events: output buffer from a BPF program
* @events__sz: @events len, must end in '__sz'' for the verifier
*/
__bpf_kfunc void scx_bpf_events(struct scx_event_stats *events,
size_t events__sz)
{
struct scx_sched *sch;
struct scx_event_stats e_sys;
/*
* We cannot entirely trust a BPF-provided size since a BPF program
* might be compiled against a different vmlinux.h, of which
* scx_event_stats would be larger (a newer vmlinux.h) or smaller
* (an older vmlinux.h). Hence, we use the smaller size to avoid
* memory corruption.
*/
events__sz = min(events__sz, sizeof(*events));
memcpy(events, &e_sys, events__sz);
}
/*
* kfunc registration can't be done from init_sched_ext_class() as
* register_btf_kfunc_id_set() needs most of the system to be up.
*
* Some kfuncs are context-sensitive and can only be called from
* specific SCX ops. They are grouped into BTF sets accordingly.
* Unfortunately, BPF currently doesn't have a way of enforcing such
* restrictions. Eventually, the verifier should be able to enforce
* them. For now, register them the same and make each kfunc explicitly
* check using scx_kf_allowed().
*/
if ((ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
&scx_kfunc_set_enqueue_dispatch)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
&scx_kfunc_set_dispatch)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
&scx_kfunc_set_cpu_release)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
&scx_kfunc_set_unlocked)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL,
&scx_kfunc_set_unlocked)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
&scx_kfunc_set_any)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_TRACING,
&scx_kfunc_set_any)) ||
(ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL,
&scx_kfunc_set_any))) {
pr_err("sched_ext: Failed to register kfunc sets (%d)\n", ret);
return ret;
}
ret = scx_idle_init();
if (ret) {
pr_err("sched_ext: Failed to initialize idle tracking (%d)\n", ret);
return ret;
}
ret = register_bpf_struct_ops(&bpf_sched_ext_ops, sched_ext_ops);
if (ret) {
pr_err("sched_ext: Failed to register struct_ops (%d)\n", ret);
return ret;
}
ret = register_pm_notifier(&scx_pm_notifier);
if (ret) {
pr_err("sched_ext: Failed to register PM notifier (%d)\n", ret);
return ret;
}
scx_kset = kset_create_and_add("sched_ext", &scx_uevent_ops, kernel_kobj);
if (!scx_kset) {
pr_err("sched_ext: Failed to create /sys/kernel/sched_ext\n");
return -ENOMEM;
}
ret = sysfs_create_group(&scx_kset->kobj, &scx_global_attr_group);
if (ret < 0) {
pr_err("sched_ext: Failed to add global attributes\n");
return ret;
}
return 0;
}
__initcall(scx_init);
Messung V0.5 in Prozent
¤ Dauer der Verarbeitung: 0.307 Sekunden
(vorverarbeitet am 2026-10-11)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.