mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH sched_ext/for-7.4] sched_ext: Clear a sub-scheduler's caps before ops.sub_detach()
@ 2026-10-06 20:31 Tejun Heo
  2026-10-06 22:14 ` Tejun Heo
  0 siblings, 1 reply; 2+ messages in thread
From: Tejun Heo @ 2026-10-06 20:31 UTC (permalink / raw)
  To: David Vernet, Andrea Righi, Changwoo Min
  Cc: sched-ext, linux-kernel, Emil Tsalapatis, David Dai, Tao Cui

A sub-scheduler can leave standing state behind on the cids delegated to it,
cpuperf targets being the obvious case, and the kernel does not track who
wrote it. Restoring it is the parent's job and ops.sub_detach() is the place
for it. Today the parent can't rely on that: the dying sub-scheduler keeps
every cap until it is marked dead after ops.exit(), so its ops.exit() or a
late timer can write after the parent has cleaned up.

Clear every cap of the sub-scheduler before ops.sub_detach(). The pshard
caps go first as a pending sync recomputes ecaps from them, and ecaps are
then zeroed directly. A grant racing the clear is refused under the child's
pshard lock, so nothing hands the caps back.

Fixes: 86094b95efcf ("sched_ext: Add per-shard cap delegation for sub-schedulers")
Signed-off-by: Tejun Heo <tj@kernel.org>
---
Applied to sched_ext/for-7.4.

 kernel/sched/ext/internal.h |    7 ++++++
 kernel/sched/ext/sub.c      |   49 ++++++++++++++++++++++++++++++++++++++++----
 2 files changed, 52 insertions(+), 4 deletions(-)

--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -828,6 +828,10 @@ struct sched_ext_ops {
 	/**
 	 * @sub_detach: Detach a sub-scheduler
 	 * @args: argument container, see the struct definition
+	 *
+	 * The sub-scheduler holds no caps by this point and can no longer
+	 * affect any cid. Whatever was delegated to it, e.g. cpuperf targets,
+	 * is the parent's to restore here.
 	 */
 	void (*sub_detach)(struct scx_sub_detach_args *args);
 
@@ -911,6 +915,9 @@ struct sched_ext_ops {
 	 * ops.exit() is also called on ops.init() failure, which is a bit
 	 * unusual. This is to allow rich reporting through @info on how
 	 * ops.init() failed.
+	 *
+	 * A sub-scheduler holds no caps by the time ops.exit() runs. Its
+	 * cap-gated kfuncs are denied.
 	 */
 	void (*exit)(struct scx_exit_info *info);
 
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -1154,6 +1154,34 @@ void scx_offline_ecaps(struct rq *rq)
 }
 
 /*
+ * Clear every cap @sch holds. The pshard caps go first as they are the source a
+ * pending sync recomputes ecaps from. ecaps are then zeroed directly for the
+ * cap checks.
+ */
+static void clear_all_caps(struct scx_sched *sch)
+{
+	s32 si, cpu;
+	u32 cap_bit;
+
+	/* enable may have failed before scx_alloc_pshards() */
+	if (!sch->pshard)
+		return;
+
+	for (si = 0; si < sch->nr_pshards; si++) {
+		struct scx_pshard *ps = sch->pshard[si];
+
+		guard(raw_spinlock_irqsave)(&ps->lock);
+		for (cap_bit = 0; cap_bit < __SCX_NR_CAPS; cap_bit++)
+			scx_cmask_clear(&ps->caps[cap_bit].cmask);
+	}
+
+	for_each_possible_cpu(cpu) {
+		guard(rq_lock_irqsave)(cpu_rq(cpu));
+		WRITE_ONCE(per_cpu_ptr(sch->pcpu, cpu)->ecaps, 0);
+	}
+}
+
+/*
  * @pcpu's sched was unhashed before the grace period, so nothing re-queues its
  * sync node. Remove the node from @rq's pending list so the pcpu can be freed.
  */
@@ -1628,6 +1656,12 @@ dump:
 
 	scx_unlink_sched(sch);
 
+	/*
+	 * The parent restores what it delegated in ops.sub_detach(). @sch must
+	 * hold no caps by then.
+	 */
+	clear_all_caps(sch);
+
 	mutex_unlock(&scx_enable_mutex);
 
 	/*
@@ -2306,8 +2340,8 @@ static s32 sub_cap_preamble(u64 cgroup_i
  * All-or-nothing keeps the caller-visible result binary per cid, so the denied
  * mask is one mask to interpret rather than a per-cap matrix.
  *
- * Return 0 on full success, -EPERM if any cid was refused, or a negative
- * errno on other failures.
+ * Return 0 on full success, -EPERM if any cid was refused, -ENODEV if the child
+ * is gone or being disabled, or a negative errno on other failures.
  */
 __bpf_kfunc s32 scx_bpf_sub_grant(u64 cgroup_id, u64 caps,
 				  const struct scx_cmask *cmask__arena,
@@ -2361,6 +2395,12 @@ __bpf_kfunc s32 scx_bpf_sub_grant(u64 cg
 		scoped_guard (raw_spinlock, &pps->lock) {
 			guard(raw_spinlock_nested)(&cps->lock);
 
+			/* the child is being disabled, see clear_all_caps() */
+			if (unlikely(READ_ONCE(child->aborting))) {
+				ret = -ENODEV;
+				goto out;
+			}
+
 			/*
 			 * Narrow granted_cids to cids the parent holds every
 			 * requested cap on. All-or-nothing per cid.
@@ -2413,9 +2453,10 @@ __bpf_kfunc s32 scx_bpf_sub_grant(u64 cg
 		}
 	}
 
+	ret = any_denied ? -EPERM : 0;
+out:
 	caps_updated_deliver(&to_deliver);
-
-	return any_denied ? -EPERM : 0;
+	return ret;
 }
 
 /**

^ permalink raw reply	[flat|nested] 2+ messages in thread

* Re: [PATCH sched_ext/for-7.4] sched_ext: Clear a sub-scheduler's caps before ops.sub_detach()
  2026-10-06 20:31 [PATCH sched_ext/for-7.4] sched_ext: Clear a sub-scheduler's caps before ops.sub_detach() Tejun Heo
@ 2026-10-06 22:14 ` Tejun Heo
  0 siblings, 0 replies; 2+ messages in thread
From: Tejun Heo @ 2026-10-06 22:14 UTC (permalink / raw)
  To: David Vernet, Andrea Righi, Changwoo Min
  Cc: sched-ext, linux-kernel, Emil Tsalapatis, David Dai, Tao Cui

On Tue, Oct 06, 2026 at 10:31:48AM -1000, Tejun Heo wrote:
> Applied to sched_ext/for-7.4.

Applying this was an accident and premature. I'm going to leave it as-is
but please holler if you see any issues.

Thanks.

--
tejun

^ permalink raw reply	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-10-06 22:14 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-06 20:31 [PATCH sched_ext/for-7.4] sched_ext: Clear a sub-scheduler's caps before ops.sub_detach() Tejun Heo
2026-10-06 22:14 ` Tejun Heo

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®