summaryrefslogtreecommitdiff
path: root/include/linux
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-08-20 11:01:37 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-08-20 11:01:37 -0700
commit11260c335ec6071af5543aef73000b28f041c124 (patch)
tree72eda88ab241d491ec3b6ec52f31b5f55a4324eb /include/linux
parent40d8c81577db09b71ee5402ba336b642d32d6a82 (diff)
parentfab183d632628381b466a41479489541ac0e29a0 (diff)
Merge tag 'sched_ext-for-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext
Pull sched_ext updates from Tejun Heo: "Most of this cycle completes the enqueue-path support for hierarchical sub-scheduling, which makes sub-scheduler support feature complete: a root BPF scheduler can now hand a cgroup subtree over to a nested sub-scheduler together with revocable CPU grants, and the sub-scheduler owns all scheduling decisions for its tasks on those CPUs. Development volume was high and a number of changes plugging holes in the new support landed late in the cycle. Also included are core scheduling fixes that were completed too late for the v7.2 release and are routed through this pull request. Sub-scheduler CPU delegation: - Parent schedulers now grant and revoke per-CPU capabilities (enqueueing, preemption, CPU frequency control) on their children, enforced on every path a scheduler can reach a CPU through. Previously only dispatching could be delegated; this lets sub-schedulers fully schedule their CPUs. - Rescue execution: a task whose scheduler doesn't have access to the CPUs the task needs to run on starved until the watchdog ejected the whole scheduler. The kernel now runs such tasks directly on a small bandwidth budget, turning a scheduler-killing failure into bounded degradation. - Cgroup integration: tasks migrating across a sub-scheduler boundary weren't re-homed to the new owner, causing wrong-scheduler scheduling and a use-after-free. Sub-schedulers now take over their cgroup subtree and receive its cgroup callbacks. - Arena objects now cross the kernel/BPF boundary as typed pointer arguments, translated transparently by the BPF tree's new arena argument support, replacing untyped arguments with manual translation. - scx_qmap now demonstrates full hierarchical sub-scheduling. Other fixes and updates: - Robustness improvements: the abort path is now NMI-safe, fixing deadlocks when errors are raised from NMI context and making hardlockup recovery direct. Reenqueue loops that could monopolize a CPU ahead of the watchdog now eject the offending scheduler, and stalls are blamed on the scheduler actually responsible. - Hardening: BPF-writable arena memory is validated before kernel use, and task slice and vtime writes got explicit synchronization rules, closing corruption vectors open to buggy or malicious schedulers. - Core scheduling: sched_ext dispatching can drop the rq lock inside the core-wide pick, which let interleaving selections corrupt each other's state and hard-hang the machine. The selection now restarts when the lock was released. The task ordering callback was also invoked with its arguments swapped, and the default ordering is updated to work across sub-scheduler boundaries. The fixes are marked for stable. - Other fixes headed for stable: a task init leak on fork failure during enable, tooling compat macros that silently failed to detect newer kernels, and a crash on reenqueueing against a destroyed dispatch queue. - Tooling: scx_pair moves off deprecated callbacks, and the deprecated scx_bpf_cpu_rq() kfunc is removed" * tag 'sched_ext-for-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext: (144 commits) sched_ext: Drop the dead SCX_DEQ_CORE_SCHED_EXEC test in dequeue_task_scx() sched_ext: Make core-sched task ordering hierarchy-aware sched_ext: Use runnable_at for the default core-sched task ordering sched_ext: Fix inverted ops.core_sched_before() invocation sched_ext: Move the config-off sub-cap kfunc stubs into sub.c sched_ext: Rename balance-era identifiers to dispatch terms sched_ext: Drop the stale keep_prev fixup in dispatch_pick() sched_ext: Keep kick_sync waiting on the rq's own CPU sched_ext: Make SCHED_CLASS_EXT select GENERIC_ALLOCATOR sched_ext/scx_flatcg: Fix cvtime true-up on slice expiry sched_ext: Don't BUG_ON a destroyed DSQ in process_deferred_reenq_users sched_ext: Fix scx_bpf_dsq_move_to_local___v2 compat detection sched_ext: Make scx_bpf_events() read the calling scheduler's counters sched_ext: Drop unlocked scx_rq_clock_invalidate() from scx_root_disable() selftests/sched_ext: Fix flaky ddsp failure tests on busy systems selftests/sched_ext: Make numa idle validation race-free sched_ext: Fix scx_bpf_dsq_reenq___compat kfunc extern prototype sched_ext/scx_flatcg: expire cached hweights on weight changes sched_ext: Fix exit_task leak on fork failure during enable sched_ext: fix stale references in doc comments ...
Diffstat (limited to 'include/linux')
-rw-r--r--include/linux/cgroup.h26
-rw-r--r--include/linux/sched/ext.h61
2 files changed, 77 insertions, 10 deletions
diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h
index b905208942bf..303655d9deaf 100644
--- a/include/linux/cgroup.h
+++ b/include/linux/cgroup.h
@@ -82,12 +82,38 @@ enum cgroup_lifetime_events {
CGROUP_LIFETIME_OFFLINE,
};
+/*
+ * Events on cgroup_task_notifier, data is struct cgroup_task_migrate_ctx.
+ * MIGRATING fires per task before the migration commits and an error return
+ * from the chain fails the migration, in which case tasks that were already
+ * notified receive MIGRATE_CANCELED. MIGRATED fires per task after the
+ * migration is committed and can't fail. Only migrations that change a task's
+ * dfl cgroup are reported.
+ */
+enum cgroup_task_events {
+ CGROUP_TASK_MIGRATING,
+ CGROUP_TASK_MIGRATED,
+ CGROUP_TASK_MIGRATE_CANCELED,
+};
+
+/*
+ * @src_dcgrp and @dst_dcgrp are @task's dfl cgroups before and after the
+ * migration. @src_dcgrp is NULL for CGROUP_TASK_MIGRATED as per-task sources
+ * are not tracked past the commit point.
+ */
+struct cgroup_task_migrate_ctx {
+ struct task_struct *task;
+ struct cgroup *src_dcgrp;
+ struct cgroup *dst_dcgrp;
+};
+
extern struct file_system_type cgroup_fs_type;
extern struct cgroup_root cgrp_dfl_root;
extern struct css_set init_css_set;
extern struct mutex cgroup_mutex;
extern spinlock_t css_set_lock;
extern struct blocking_notifier_head cgroup_lifetime_notifier;
+extern struct blocking_notifier_head cgroup_task_notifier;
#define SUBSYS(_x) extern struct cgroup_subsys _x ## _cgrp_subsys;
#include <linux/cgroup_subsys.h>
diff --git a/include/linux/sched/ext.h b/include/linux/sched/ext.h
index 87e353f7e011..582d7cd4a983 100644
--- a/include/linux/sched/ext.h
+++ b/include/linux/sched/ext.h
@@ -58,6 +58,8 @@ enum scx_dsq_id_flags {
SCX_DSQ_GLOBAL = SCX_DSQ_FLAG_BUILTIN | 1,
SCX_DSQ_LOCAL = SCX_DSQ_FLAG_BUILTIN | 2,
SCX_DSQ_BYPASS = SCX_DSQ_FLAG_BUILTIN | 3,
+ SCX_DSQ_REJECT = SCX_DSQ_FLAG_BUILTIN | 4, /* internal - see find_dsq_for_dispatch() */
+ SCX_DSQ_RESCUE = SCX_DSQ_FLAG_BUILTIN | 5, /* internal - see find_dsq_for_dispatch() */
SCX_DSQ_LOCAL_ON = SCX_DSQ_FLAG_BUILTIN | SCX_DSQ_FLAG_LOCAL_ON,
SCX_DSQ_LOCAL_CPU_MASK = 0xffffffffLLU,
};
@@ -101,6 +103,7 @@ enum scx_ent_flags {
SCX_TASK_DEQD_FOR_SLEEP = 1 << 3, /* last dequeue was for SLEEP */
SCX_TASK_SUB_INIT = 1 << 4, /* task being initialized for a sub sched */
SCX_TASK_IMMED = 1 << 5, /* task is on local DSQ with %SCX_ENQ_IMMED */
+ SCX_TASK_PROTECTED = 1 << 6, /* slice and DSQ head position protected */
/*
* Bits 8 to 10 are used to carry task state:
@@ -124,7 +127,7 @@ enum scx_ent_flags {
SCX_TASK_DEAD = 5 << SCX_TASK_STATE_SHIFT,
/*
- * Bits 12 and 13 are used to carry reenqueue reason. In addition to
+ * Bits 12 to 14 are used to carry reenqueue reason. In addition to
* %SCX_ENQ_REENQ flag, ops.enqueue() can also test for
* %SCX_TASK_REENQ_REASON_NONE to distinguish reenqueues.
*
@@ -132,15 +135,17 @@ enum scx_ent_flags {
* KFUNC reenqueued by scx_bpf_dsq_reenq() and friends
* IMMED reenqueued due to failed ENQ_IMMED
* PREEMPTED preempted while running
+ * CAP sub-sched cap miss, see p->scx.reenq_reason_*
*/
SCX_TASK_REENQ_REASON_SHIFT = 12,
- SCX_TASK_REENQ_REASON_BITS = 2,
+ SCX_TASK_REENQ_REASON_BITS = 3,
SCX_TASK_REENQ_REASON_MASK = ((1 << SCX_TASK_REENQ_REASON_BITS) - 1) << SCX_TASK_REENQ_REASON_SHIFT,
SCX_TASK_REENQ_NONE = 0 << SCX_TASK_REENQ_REASON_SHIFT,
SCX_TASK_REENQ_KFUNC = 1 << SCX_TASK_REENQ_REASON_SHIFT,
SCX_TASK_REENQ_IMMED = 2 << SCX_TASK_REENQ_REASON_SHIFT,
SCX_TASK_REENQ_PREEMPTED = 3 << SCX_TASK_REENQ_REASON_SHIFT,
+ SCX_TASK_REENQ_CAP = 4 << SCX_TASK_REENQ_REASON_SHIFT,
/* iteration cursor, not a task */
SCX_TASK_CURSOR = 1 << 31,
@@ -189,22 +194,26 @@ struct sched_ext_entity {
atomic_long_t ops_state;
u64 ddsp_dsq_id;
u64 ddsp_enq_flags;
+ u64 ddsp_slice;
+ u64 ddsp_vtime;
struct scx_dsq_list_node dsq_list; /* dispatch order */
struct rb_node dsq_priq; /* p->scx.dsq_vtime order */
u32 dsq_seq;
u32 dsq_flags; /* protected by DSQ lock */
u32 flags; /* protected by rq lock */
u32 weight;
+ u32 reenq_cnt; /* reenqueues since last run */
s32 sticky_cpu;
s32 holding_cpu;
s32 selected_cpu;
+ s32 runnable_cpu; /* cpu @p is runnable on, -1 if not */
struct task_struct *kf_tasks[2]; /* see SCX_CALL_OP_TASK() */
struct list_head runnable_node; /* rq->scx.runnable_list */
unsigned long runnable_at;
-#ifdef CONFIG_SCHED_CORE
- u64 core_sched_at; /* see scx_prio_less() */
+#ifdef CONFIG_EXT_SUB_SCHED
+ unsigned long rescue_at; /* queued on a rescue DSQ at, jiffies */
#endif
/*
@@ -219,10 +228,11 @@ struct sched_ext_entity {
/* BPF scheduler modifiable fields */
/*
- * Runtime budget in nsecs. This is usually set through
- * scx_bpf_dsq_insert() but can also be modified directly by the BPF
- * scheduler. Automatically decreased by SCX as the task executes. On
- * depletion, a scheduling event is triggered.
+ * Runtime budget in nsecs - how long the task may hold its cpu. Owned
+ * by the task's scheduler. Set it when enqueuing via
+ * scx_bpf_dsq_insert(), or otherwise via scx_bpf_task_set_slice().
+ * Automatically decreased as the task executes. On depletion a
+ * scheduling event is triggered.
*
* This value is cleared to zero if the task is preempted by
* %SCX_KICK_PREEMPT and shouldn't be used to determine how long the
@@ -240,6 +250,22 @@ struct sched_ext_entity {
u64 dsq_vtime;
/*
+ * Out-of-band slice request from scx_bpf_task_set_slice() when the
+ * caller does not hold the rq lock, applied under the rq lock at the
+ * next slice consideration. One atomic64 packs the pending flag, the
+ * issuing sch's id, and the requested slice. See scx_slice_oob_consts.
+ */
+ atomic64_t slice_oob;
+
+ /*
+ * Sub-sched cap rejected reenq context, valid only while
+ * %SCX_TASK_REENQ_CAP is set. @reenq_reason_caps is the SCX_CAP_* bits
+ * that were needed but missing. @reenq_reason_cid is the target cid.
+ */
+ u64 reenq_reason_caps;
+ s32 reenq_reason_cid;
+
+ /*
* If set, reject future sched_setscheduler(2) calls updating the policy
* to %SCHED_EXT with -%EACCES.
*
@@ -263,7 +289,7 @@ void sched_ext_dead(struct task_struct *p);
void print_scx_info(const char *log_lvl, struct task_struct *p);
void scx_softlockup(u32 dur_s);
bool scx_hardlockup(int cpu);
-bool scx_rcu_cpu_stall(void);
+bool scx_rcu_cpu_stall(const struct cpumask *stalled_mask);
#else /* !CONFIG_SCHED_CLASS_EXT */
@@ -271,12 +297,27 @@ static inline void sched_ext_dead(struct task_struct *p) {}
static inline void print_scx_info(const char *log_lvl, struct task_struct *p) {}
static inline void scx_softlockup(u32 dur_s) {}
static inline bool scx_hardlockup(int cpu) { return false; }
-static inline bool scx_rcu_cpu_stall(void) { return false; }
+static inline bool scx_rcu_cpu_stall(const struct cpumask *stalled_mask) { return false; }
#endif /* CONFIG_SCHED_CLASS_EXT */
struct scx_task_group {
#ifdef CONFIG_EXT_GROUP_SCHED
+ /*
+ * The sched this tg is on, NULL if none. SCX_TG_INITED tracks whether
+ * ops.cgroup_init() succeeded on it. When a child sched exits and its
+ * tgs move to the parent, a failed init leaves the tg on the parent
+ * with INITED clear (see scx_cgroup_return_subtree()).
+ *
+ * This is tracked separately from cgrp->scx_sched because the tg
+ * hierarchy can diverge from the cgroup2 hierarchy in both lifetime and
+ * shape. A tg stays online past its cgroup's removal while the
+ * cgrp->scx_sched rewrites visit only live cgroups, leaving a removed
+ * cgroup's pointer stale. The cpu controller can also be mounted on
+ * cgroup1.
+ */
+ struct scx_sched *sched;
+
u32 flags; /* SCX_TG_* */
u32 weight;
u64 bw_period_us;