summaryrefslogtreecommitdiff
path: root/tools/testing/selftests/sched_ext
diff options
context:
space:
mode:
Diffstat (limited to 'tools/testing/selftests/sched_ext')
-rw-r--r--tools/testing/selftests/sched_ext/Makefile10
-rw-r--r--tools/testing/selftests/sched_ext/allowed_cpus.bpf.c49
-rw-r--r--tools/testing/selftests/sched_ext/cyclic_kick_wait.bpf.c68
-rw-r--r--tools/testing/selftests/sched_ext/cyclic_kick_wait.c194
-rw-r--r--tools/testing/selftests/sched_ext/ddsp_bogus_dsq_fail.bpf.c20
-rw-r--r--tools/testing/selftests/sched_ext/ddsp_vtimelocal_fail.bpf.c13
-rw-r--r--tools/testing/selftests/sched_ext/dequeue.bpf.c389
-rw-r--r--tools/testing/selftests/sched_ext/dequeue.c275
-rw-r--r--tools/testing/selftests/sched_ext/dequeue_iter.bpf.c73
-rw-r--r--tools/testing/selftests/sched_ext/dequeue_iter.c79
-rw-r--r--tools/testing/selftests/sched_ext/enable_cmask.bpf.c217
-rw-r--r--tools/testing/selftests/sched_ext/enable_cmask.c138
-rw-r--r--tools/testing/selftests/sched_ext/exit.bpf.c2
-rw-r--r--tools/testing/selftests/sched_ext/exit.c3
-rw-r--r--tools/testing/selftests/sched_ext/exit_test.h2
-rw-r--r--tools/testing/selftests/sched_ext/init_enable_count.c35
-rw-r--r--tools/testing/selftests/sched_ext/maximal.bpf.c17
-rw-r--r--tools/testing/selftests/sched_ext/maximal.c3
-rw-r--r--tools/testing/selftests/sched_ext/nohz_tick.bpf.c65
-rw-r--r--tools/testing/selftests/sched_ext/nohz_tick.c347
-rw-r--r--tools/testing/selftests/sched_ext/non_scx_kfunc_deny.bpf.c39
-rw-r--r--tools/testing/selftests/sched_ext/non_scx_kfunc_deny.c47
-rw-r--r--tools/testing/selftests/sched_ext/numa.bpf.c41
-rw-r--r--tools/testing/selftests/sched_ext/peek_dsq.bpf.c14
-rw-r--r--tools/testing/selftests/sched_ext/prog_run.c34
-rw-r--r--tools/testing/selftests/sched_ext/reload_loop.c3
-rw-r--r--tools/testing/selftests/sched_ext/rt_stall.bpf.c23
-rw-r--r--tools/testing/selftests/sched_ext/rt_stall.c293
-rw-r--r--tools/testing/selftests/sched_ext/runner.c51
-rw-r--r--tools/testing/selftests/sched_ext/select_cpu_dfl.c54
-rw-r--r--tools/testing/selftests/sched_ext/select_cpu_vtime.bpf.c8
-rw-r--r--tools/testing/selftests/sched_ext/total_bw.c480
-rw-r--r--tools/testing/selftests/sched_ext/util.c4
-rw-r--r--tools/testing/selftests/sched_ext/util.h2
34 files changed, 2994 insertions, 98 deletions
diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile
index 5fe45f9c5f8f..4e06d0baaeec 100644
--- a/tools/testing/selftests/sched_ext/Makefile
+++ b/tools/testing/selftests/sched_ext/Makefile
@@ -93,6 +93,8 @@ BPF_CFLAGS = -g -D__TARGET_ARCH_$(SRCARCH) \
$(CLANG_SYS_INCLUDES) \
-Wall -Wno-compare-distinct-pointer-types \
-Wno-incompatible-function-pointer-types \
+ -Wno-microsoft-anon-tag \
+ -fms-extensions \
-O2 -mcpu=v3
# sort removes libbpf duplicates when not cross-building
@@ -161,10 +163,13 @@ all_test_bpfprogs := $(foreach prog,$(wildcard *.bpf.c),$(INCLUDE_DIR)/$(patsubs
auto-test-targets := \
create_dsq \
+ dequeue \
+ dequeue_iter \
enq_last_no_enq_fails \
ddsp_bogus_dsq_fail \
ddsp_vtimelocal_fail \
dsp_local_on \
+ enable_cmask \
enq_select_cpu \
exit \
hotplug \
@@ -172,6 +177,8 @@ auto-test-targets := \
maximal \
maybe_null \
minimal \
+ non_scx_kfunc_deny \
+ nohz_tick \
numa \
allowed_cpus \
peek_dsq \
@@ -183,7 +190,10 @@ auto-test-targets := \
select_cpu_dispatch_bad_dsq \
select_cpu_dispatch_dbl_dsp \
select_cpu_vtime \
+ rt_stall \
test_example \
+ total_bw \
+ cyclic_kick_wait \
testcase-targets := $(addsuffix .o,$(addprefix $(SCXOBJ_DIR)/,$(auto-test-targets)))
diff --git a/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c b/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c
index 35923e74a2ec..9dd72d0da29b 100644
--- a/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c
+++ b/tools/testing/selftests/sched_ext/allowed_cpus.bpf.c
@@ -15,15 +15,48 @@ UEI_DEFINE(uei);
private(PREF_CPUS) struct bpf_cpumask __kptr * allowed_cpumask;
static void
-validate_idle_cpu(const struct task_struct *p, const struct cpumask *allowed, s32 cpu)
+validate_local_idle_state(void)
{
- if (scx_bpf_test_and_clear_cpu_idle(cpu))
- scx_bpf_error("CPU %d should be marked as busy", cpu);
+ const struct cpumask *idle;
+ struct task_struct *curr;
+ s32 cpu = bpf_get_smp_processor_id();
+ bool cpu_is_idle, curr_is_idle;
- if (bpf_cpumask_subset(allowed, p->cpus_ptr) &&
- !bpf_cpumask_test_cpu(cpu, allowed))
+ bpf_rcu_read_lock();
+ curr = scx_bpf_cpu_curr(cpu);
+ curr_is_idle = curr && (curr->flags & PF_IDLE);
+ bpf_rcu_read_unlock();
+
+ idle = scx_bpf_get_idle_cpumask();
+ cpu_is_idle = bpf_cpumask_test_cpu(cpu, idle);
+ scx_bpf_put_idle_cpumask(idle);
+
+ /*
+ * Unlike a remote selected CPU, the local CPU cannot go through an
+ * idle re-pick while this callback is running. If it is running a
+ * non-idle scheduling context, it must not be advertised as idle.
+ */
+ if (!curr_is_idle && cpu_is_idle)
+ scx_bpf_error("running CPU %d should be marked as busy", cpu);
+}
+
+static void
+validate_selected_cpu(const struct task_struct *p, s32 cpu)
+{
+ const struct cpumask *allowed = cast_mask(allowed_cpumask);
+
+ if (!allowed) {
+ scx_bpf_error("allowed domain not initialized");
+ return;
+ }
+
+ if (!bpf_cpumask_test_cpu(cpu, allowed))
scx_bpf_error("CPU %d not in the allowed domain for %d (%s)",
cpu, p->pid, p->comm);
+
+ if (!bpf_cpumask_test_cpu(cpu, p->cpus_ptr))
+ scx_bpf_error("CPU %d not in the affinity mask for %d (%s)",
+ cpu, p->pid, p->comm);
}
s32 BPF_STRUCT_OPS(allowed_cpus_select_cpu,
@@ -32,6 +65,7 @@ s32 BPF_STRUCT_OPS(allowed_cpus_select_cpu,
const struct cpumask *allowed;
s32 cpu;
+ validate_local_idle_state();
allowed = cast_mask(allowed_cpumask);
if (!allowed) {
scx_bpf_error("allowed domain not initialized");
@@ -43,7 +77,7 @@ s32 BPF_STRUCT_OPS(allowed_cpus_select_cpu,
*/
cpu = scx_bpf_select_cpu_and(p, prev_cpu, wake_flags, allowed, 0);
if (cpu >= 0) {
- validate_idle_cpu(p, allowed, cpu);
+ validate_selected_cpu(p, cpu);
scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL, 0);
return cpu;
@@ -59,6 +93,7 @@ void BPF_STRUCT_OPS(allowed_cpus_enqueue, struct task_struct *p, u64 enq_flags)
scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, 0);
+ validate_local_idle_state();
allowed = cast_mask(allowed_cpumask);
if (!allowed) {
scx_bpf_error("allowed domain not initialized");
@@ -71,7 +106,7 @@ void BPF_STRUCT_OPS(allowed_cpus_enqueue, struct task_struct *p, u64 enq_flags)
*/
cpu = scx_bpf_select_cpu_and(p, prev_cpu, 0, allowed, 0);
if (cpu >= 0) {
- validate_idle_cpu(p, allowed, cpu);
+ validate_selected_cpu(p, cpu);
scx_bpf_kick_cpu(cpu, SCX_KICK_IDLE);
}
}
diff --git a/tools/testing/selftests/sched_ext/cyclic_kick_wait.bpf.c b/tools/testing/selftests/sched_ext/cyclic_kick_wait.bpf.c
new file mode 100644
index 000000000000..cb34d3335917
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/cyclic_kick_wait.bpf.c
@@ -0,0 +1,68 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Stress concurrent SCX_KICK_WAIT calls to reproduce wait-cycle deadlock.
+ *
+ * Three CPUs are designated from userspace. Every enqueue from one of the
+ * three CPUs kicks the next CPU in the ring with SCX_KICK_WAIT, creating a
+ * persistent A -> B -> C -> A wait cycle pressure.
+ */
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+const volatile s32 test_cpu_a;
+const volatile s32 test_cpu_b;
+const volatile s32 test_cpu_c;
+
+u64 nr_enqueues;
+u64 nr_wait_kicks;
+
+UEI_DEFINE(uei);
+
+static s32 target_cpu(s32 cpu)
+{
+ if (cpu == test_cpu_a)
+ return test_cpu_b;
+ if (cpu == test_cpu_b)
+ return test_cpu_c;
+ if (cpu == test_cpu_c)
+ return test_cpu_a;
+ return -1;
+}
+
+void BPF_STRUCT_OPS(cyclic_kick_wait_enqueue, struct task_struct *p,
+ u64 enq_flags)
+{
+ s32 this_cpu = bpf_get_smp_processor_id();
+ s32 tgt;
+
+ __sync_fetch_and_add(&nr_enqueues, 1);
+
+ if (p->flags & PF_KTHREAD) {
+ scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL, SCX_SLICE_INF,
+ enq_flags | SCX_ENQ_PREEMPT);
+ return;
+ }
+
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+
+ tgt = target_cpu(this_cpu);
+ if (tgt < 0 || tgt == this_cpu)
+ return;
+
+ __sync_fetch_and_add(&nr_wait_kicks, 1);
+ scx_bpf_kick_cpu(tgt, SCX_KICK_WAIT);
+}
+
+void BPF_STRUCT_OPS(cyclic_kick_wait_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops cyclic_kick_wait_ops = {
+ .enqueue = cyclic_kick_wait_enqueue,
+ .exit = cyclic_kick_wait_exit,
+ .name = "cyclic_kick_wait",
+ .timeout_ms = 1000U,
+};
diff --git a/tools/testing/selftests/sched_ext/cyclic_kick_wait.c b/tools/testing/selftests/sched_ext/cyclic_kick_wait.c
new file mode 100644
index 000000000000..c2e5aa9de715
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/cyclic_kick_wait.c
@@ -0,0 +1,194 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Test SCX_KICK_WAIT forward progress under cyclic wait pressure.
+ *
+ * SCX_KICK_WAIT busy-waits until the target CPU enters the scheduling path.
+ * If multiple CPUs form a wait cycle (A waits for B, B waits for C, C waits
+ * for A), all CPUs deadlock unless the implementation breaks the cycle.
+ *
+ * This test creates that scenario: three CPUs are arranged in a ring. The BPF
+ * scheduler's ops.enqueue() kicks the next CPU in the ring with SCX_KICK_WAIT
+ * on every enqueue. Userspace pins 4 worker threads per CPU that loop calling
+ * sched_yield(), generating a steady stream of enqueues and thus sustained
+ * A->B->C->A kick_wait cycle pressure. The test passes if the system remains
+ * responsive for 5 seconds without the scheduler being killed by the watchdog.
+ */
+#define _GNU_SOURCE
+
+#include <bpf/bpf.h>
+#include <errno.h>
+#include <pthread.h>
+#include <sched.h>
+#include <scx/common.h>
+#include <stdint.h>
+#include <string.h>
+#include <time.h>
+#include <unistd.h>
+
+#include "scx_test.h"
+#include "cyclic_kick_wait.bpf.skel.h"
+
+#define WORKERS_PER_CPU 4
+#define NR_TEST_CPUS 3
+#define NR_WORKERS (NR_TEST_CPUS * WORKERS_PER_CPU)
+
+struct worker_ctx {
+ pthread_t tid;
+ int cpu;
+ volatile bool stop;
+ volatile __u64 iters;
+ bool started;
+};
+
+static void *worker_fn(void *arg)
+{
+ struct worker_ctx *worker = arg;
+ cpu_set_t mask;
+
+ CPU_ZERO(&mask);
+ CPU_SET(worker->cpu, &mask);
+
+ if (sched_setaffinity(0, sizeof(mask), &mask))
+ return (void *)(uintptr_t)errno;
+
+ while (!worker->stop) {
+ sched_yield();
+ worker->iters++;
+ }
+
+ return NULL;
+}
+
+static int join_worker(struct worker_ctx *worker)
+{
+ void *ret;
+ struct timespec ts;
+ int err;
+
+ if (!worker->started)
+ return 0;
+
+ if (clock_gettime(CLOCK_REALTIME, &ts))
+ return -errno;
+
+ ts.tv_sec += 2;
+ err = pthread_timedjoin_np(worker->tid, &ret, &ts);
+ if (err == ETIMEDOUT)
+ pthread_detach(worker->tid);
+ if (err)
+ return -err;
+
+ if ((uintptr_t)ret)
+ return -(int)(uintptr_t)ret;
+
+ return 0;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+ struct cyclic_kick_wait *skel;
+
+ skel = cyclic_kick_wait__open();
+ SCX_FAIL_IF(!skel, "Failed to open skel");
+ SCX_ENUM_INIT(skel);
+
+ *ctx = skel;
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct cyclic_kick_wait *skel = ctx;
+ struct worker_ctx workers[NR_WORKERS] = {};
+ struct bpf_link *link = NULL;
+ enum scx_test_status status = SCX_TEST_PASS;
+ int test_cpus[NR_TEST_CPUS];
+ int nr_cpus = 0;
+ cpu_set_t mask;
+ int ret, i;
+
+ if (sched_getaffinity(0, sizeof(mask), &mask)) {
+ SCX_ERR("Failed to get affinity (%d)", errno);
+ return SCX_TEST_FAIL;
+ }
+
+ for (i = 0; i < CPU_SETSIZE; i++) {
+ if (CPU_ISSET(i, &mask))
+ test_cpus[nr_cpus++] = i;
+ if (nr_cpus == NR_TEST_CPUS)
+ break;
+ }
+
+ if (nr_cpus < NR_TEST_CPUS)
+ return SCX_TEST_SKIP;
+
+ skel->rodata->test_cpu_a = test_cpus[0];
+ skel->rodata->test_cpu_b = test_cpus[1];
+ skel->rodata->test_cpu_c = test_cpus[2];
+
+ if (cyclic_kick_wait__load(skel)) {
+ SCX_ERR("Failed to load skel");
+ return SCX_TEST_FAIL;
+ }
+
+ link = bpf_map__attach_struct_ops(skel->maps.cyclic_kick_wait_ops);
+ if (!link) {
+ SCX_ERR("Failed to attach scheduler");
+ return SCX_TEST_FAIL;
+ }
+
+ for (i = 0; i < NR_WORKERS; i++)
+ workers[i].cpu = test_cpus[i / WORKERS_PER_CPU];
+
+ for (i = 0; i < NR_WORKERS; i++) {
+ ret = pthread_create(&workers[i].tid, NULL, worker_fn, &workers[i]);
+ if (ret) {
+ SCX_ERR("Failed to create worker thread %d (%d)", i, ret);
+ status = SCX_TEST_FAIL;
+ goto out;
+ }
+ workers[i].started = true;
+ }
+
+ sleep(5);
+
+ if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) {
+ SCX_ERR("Scheduler exited unexpectedly (kind=%llu code=%lld)",
+ (unsigned long long)skel->data->uei.kind,
+ (long long)skel->data->uei.exit_code);
+ status = SCX_TEST_FAIL;
+ }
+
+out:
+ for (i = 0; i < NR_WORKERS; i++)
+ workers[i].stop = true;
+
+ for (i = 0; i < NR_WORKERS; i++) {
+ ret = join_worker(&workers[i]);
+ if (ret && status == SCX_TEST_PASS) {
+ SCX_ERR("Failed to join worker thread %d (%d)", i, ret);
+ status = SCX_TEST_FAIL;
+ }
+ }
+
+ if (link)
+ bpf_link__destroy(link);
+
+ return status;
+}
+
+static void cleanup(void *ctx)
+{
+ struct cyclic_kick_wait *skel = ctx;
+
+ cyclic_kick_wait__destroy(skel);
+}
+
+struct scx_test cyclic_kick_wait = {
+ .name = "cyclic_kick_wait",
+ .description = "Verify SCX_KICK_WAIT forward progress under a 3-CPU wait cycle",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&cyclic_kick_wait)
diff --git a/tools/testing/selftests/sched_ext/ddsp_bogus_dsq_fail.bpf.c b/tools/testing/selftests/sched_ext/ddsp_bogus_dsq_fail.bpf.c
index 6f4c3f5a1c5d..7ef9de7b27eb 100644
--- a/tools/testing/selftests/sched_ext/ddsp_bogus_dsq_fail.bpf.c
+++ b/tools/testing/selftests/sched_ext/ddsp_bogus_dsq_fail.bpf.c
@@ -14,18 +14,16 @@ s32 BPF_STRUCT_OPS(ddsp_bogus_dsq_fail_select_cpu, struct task_struct *p,
s32 prev_cpu, u64 wake_flags)
{
s32 cpu = scx_bpf_pick_idle_cpu(p->cpus_ptr, 0);
+ if (cpu < 0)
+ cpu = prev_cpu;
- if (cpu >= 0) {
- /*
- * If we dispatch to a bogus DSQ that will fall back to the
- * builtin global DSQ, we fail gracefully.
- */
- scx_bpf_dsq_insert_vtime(p, 0xcafef00d, SCX_SLICE_DFL,
- p->scx.dsq_vtime, 0);
- return cpu;
- }
-
- return prev_cpu;
+ /*
+ * If we dispatch to a bogus DSQ that will fall back to the
+ * builtin global DSQ, we fail gracefully.
+ */
+ scx_bpf_dsq_insert_vtime(p, 0xcafef00d, SCX_SLICE_DFL,
+ p->scx.dsq_vtime, 0);
+ return cpu;
}
void BPF_STRUCT_OPS(ddsp_bogus_dsq_fail_exit, struct scx_exit_info *ei)
diff --git a/tools/testing/selftests/sched_ext/ddsp_vtimelocal_fail.bpf.c b/tools/testing/selftests/sched_ext/ddsp_vtimelocal_fail.bpf.c
index e4a55027778f..82dca4cdc0a6 100644
--- a/tools/testing/selftests/sched_ext/ddsp_vtimelocal_fail.bpf.c
+++ b/tools/testing/selftests/sched_ext/ddsp_vtimelocal_fail.bpf.c
@@ -14,15 +14,14 @@ s32 BPF_STRUCT_OPS(ddsp_vtimelocal_fail_select_cpu, struct task_struct *p,
s32 prev_cpu, u64 wake_flags)
{
s32 cpu = scx_bpf_pick_idle_cpu(p->cpus_ptr, 0);
+ if (cpu < 0)
+ cpu = prev_cpu;
- if (cpu >= 0) {
- /* Shouldn't be allowed to vtime dispatch to a builtin DSQ. */
- scx_bpf_dsq_insert_vtime(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL,
- p->scx.dsq_vtime, 0);
- return cpu;
- }
+ /* Shouldn't be allowed to vtime dispatch to a builtin DSQ. */
+ scx_bpf_dsq_insert_vtime(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL,
+ p->scx.dsq_vtime, 0);
- return prev_cpu;
+ return cpu;
}
void BPF_STRUCT_OPS(ddsp_vtimelocal_fail_exit, struct scx_exit_info *ei)
diff --git a/tools/testing/selftests/sched_ext/dequeue.bpf.c b/tools/testing/selftests/sched_ext/dequeue.bpf.c
new file mode 100644
index 000000000000..624e2ccb0688
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/dequeue.bpf.c
@@ -0,0 +1,389 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A scheduler that validates ops.dequeue() is called correctly:
+ * - Tasks dispatched to terminal DSQs (local, global) bypass the BPF
+ * scheduler entirely: no ops.dequeue() should be called
+ * - Tasks dispatched to user DSQs from ops.enqueue() enter BPF custody:
+ * ops.dequeue() must be called when they leave custody
+ * - Every ops.enqueue() dispatch to non-terminal DSQs is followed by
+ * exactly one ops.dequeue() (validate 1:1 pairing and state machine)
+ *
+ * Copyright (c) 2026 NVIDIA Corporation.
+ */
+
+#include <scx/common.bpf.h>
+
+#define SHARED_DSQ 0
+
+/*
+ * BPF internal queue.
+ *
+ * Tasks are stored here and consumed from ops.dispatch(), validating that
+ * tasks on BPF internal structures still get ops.dequeue() when they
+ * leave.
+ */
+struct {
+ __uint(type, BPF_MAP_TYPE_QUEUE);
+ __uint(max_entries, 32768);
+ __type(value, s32);
+} global_queue SEC(".maps");
+
+char _license[] SEC("license") = "GPL";
+
+UEI_DEFINE(uei);
+
+/*
+ * Counters to track the lifecycle of tasks:
+ * - enqueue_cnt: Number of times ops.enqueue() was called
+ * - dequeue_cnt: Number of times ops.dequeue() was called (any type)
+ * - dispatch_dequeue_cnt: Number of regular dispatch dequeues (no flag)
+ * - change_dequeue_cnt: Number of property change dequeues
+ * - bpf_queue_full: Number of times the BPF internal queue was full
+ */
+u64 enqueue_cnt, dequeue_cnt, dispatch_dequeue_cnt, change_dequeue_cnt, bpf_queue_full;
+
+/*
+ * Test scenarios:
+ * 0) Dispatch to local DSQ from ops.select_cpu() (terminal DSQ, bypasses BPF
+ * scheduler, no dequeue callbacks)
+ * 1) Dispatch to global DSQ from ops.select_cpu() (terminal DSQ, bypasses BPF
+ * scheduler, no dequeue callbacks)
+ * 2) Dispatch to shared user DSQ from ops.select_cpu() (enters BPF scheduler,
+ * dequeue callbacks expected)
+ * 3) Dispatch to local DSQ from ops.enqueue() (terminal DSQ, bypasses BPF
+ * scheduler, no dequeue callbacks)
+ * 4) Dispatch to global DSQ from ops.enqueue() (terminal DSQ, bypasses BPF
+ * scheduler, no dequeue callbacks)
+ * 5) Dispatch to shared user DSQ from ops.enqueue() (enters BPF scheduler,
+ * dequeue callbacks expected)
+ * 6) BPF internal queue from ops.enqueue(): store task PIDs in ops.enqueue(),
+ * consume in ops.dispatch() and dispatch to local DSQ (validates dequeue
+ * for tasks stored in internal BPF data structures)
+ */
+u32 test_scenario;
+
+/*
+ * Per-task state to track lifecycle and validate workflow semantics.
+ * State transitions:
+ * NONE -> ENQUEUED (on enqueue)
+ * NONE -> DISPATCHED (on direct dispatch to terminal DSQ)
+ * ENQUEUED -> DISPATCHED (on dispatch dequeue)
+ * DISPATCHED -> NONE (on property change dequeue or re-enqueue)
+ * ENQUEUED -> NONE (on property change dequeue before dispatch)
+ */
+enum task_state {
+ TASK_NONE = 0,
+ TASK_ENQUEUED,
+ TASK_DISPATCHED,
+};
+
+struct task_ctx {
+ enum task_state state; /* Current state in the workflow */
+ u64 enqueue_seq; /* Sequence number for debugging */
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_TASK_STORAGE);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __type(key, int);
+ __type(value, struct task_ctx);
+} task_ctx_stor SEC(".maps");
+
+static struct task_ctx *try_lookup_task_ctx(struct task_struct *p)
+{
+ return bpf_task_storage_get(&task_ctx_stor, p, 0, 0);
+}
+
+s32 BPF_STRUCT_OPS(dequeue_select_cpu, struct task_struct *p,
+ s32 prev_cpu, u64 wake_flags)
+{
+ struct task_ctx *tctx;
+
+ tctx = try_lookup_task_ctx(p);
+ if (!tctx)
+ return prev_cpu;
+
+ switch (test_scenario) {
+ case 0:
+ /*
+ * Direct dispatch to the local DSQ.
+ *
+ * Task bypasses BPF scheduler entirely: no enqueue
+ * tracking, no ops.dequeue() callbacks.
+ */
+ scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL, 0);
+ tctx->state = TASK_DISPATCHED;
+ break;
+ case 1:
+ /*
+ * Direct dispatch to the global DSQ.
+ *
+ * Task bypasses BPF scheduler entirely: no enqueue
+ * tracking, no ops.dequeue() callbacks.
+ */
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, 0);
+ tctx->state = TASK_DISPATCHED;
+ break;
+ case 2:
+ /*
+ * Dispatch to a shared user DSQ.
+ *
+ * Task enters BPF scheduler management: track
+ * enqueue/dequeue lifecycle and validate state
+ * transitions.
+ */
+ if (tctx->state == TASK_ENQUEUED)
+ scx_bpf_error("%d (%s): enqueue while in ENQUEUED state seq=%llu",
+ p->pid, p->comm, tctx->enqueue_seq);
+
+ scx_bpf_dsq_insert(p, SHARED_DSQ, SCX_SLICE_DFL, 0);
+
+ __sync_fetch_and_add(&enqueue_cnt, 1);
+
+ tctx->state = TASK_ENQUEUED;
+ tctx->enqueue_seq++;
+ break;
+ }
+
+ return prev_cpu;
+}
+
+void BPF_STRUCT_OPS(dequeue_enqueue, struct task_struct *p, u64 enq_flags)
+{
+ struct task_ctx *tctx;
+ s32 pid = p->pid;
+
+ tctx = try_lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ switch (test_scenario) {
+ case 3:
+ /*
+ * Direct dispatch to the local DSQ.
+ *
+ * Task bypasses BPF scheduler entirely: no enqueue
+ * tracking, no ops.dequeue() callbacks.
+ */
+ scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL, SCX_SLICE_DFL, enq_flags);
+ tctx->state = TASK_DISPATCHED;
+ break;
+ case 4:
+ /*
+ * Direct dispatch to the global DSQ.
+ *
+ * Task bypasses BPF scheduler entirely: no enqueue
+ * tracking, no ops.dequeue() callbacks.
+ */
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+ tctx->state = TASK_DISPATCHED;
+ break;
+ case 5:
+ /*
+ * Dispatch to shared user DSQ.
+ *
+ * Task enters BPF scheduler management: track
+ * enqueue/dequeue lifecycle and validate state
+ * transitions.
+ */
+ if (tctx->state == TASK_ENQUEUED)
+ scx_bpf_error("%d (%s): enqueue while in ENQUEUED state seq=%llu",
+ p->pid, p->comm, tctx->enqueue_seq);
+
+ scx_bpf_dsq_insert(p, SHARED_DSQ, SCX_SLICE_DFL, enq_flags);
+
+ __sync_fetch_and_add(&enqueue_cnt, 1);
+
+ tctx->state = TASK_ENQUEUED;
+ tctx->enqueue_seq++;
+ break;
+ case 6:
+ /*
+ * Store task in BPF internal queue.
+ *
+ * Task enters BPF scheduler management: track
+ * enqueue/dequeue lifecycle and validate state
+ * transitions.
+ */
+ if (tctx->state == TASK_ENQUEUED)
+ scx_bpf_error("%d (%s): enqueue while in ENQUEUED state seq=%llu",
+ p->pid, p->comm, tctx->enqueue_seq);
+
+ if (bpf_map_push_elem(&global_queue, &pid, 0)) {
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+ __sync_fetch_and_add(&bpf_queue_full, 1);
+
+ tctx->state = TASK_DISPATCHED;
+ } else {
+ __sync_fetch_and_add(&enqueue_cnt, 1);
+
+ tctx->state = TASK_ENQUEUED;
+ tctx->enqueue_seq++;
+ }
+ break;
+ default:
+ /* For all other scenarios, dispatch to the global DSQ */
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+ tctx->state = TASK_DISPATCHED;
+ break;
+ }
+
+ scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_IDLE);
+}
+
+void BPF_STRUCT_OPS(dequeue_dequeue, struct task_struct *p, u64 deq_flags)
+{
+ struct task_ctx *tctx;
+
+ __sync_fetch_and_add(&dequeue_cnt, 1);
+
+ tctx = try_lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ /*
+ * For scenarios 0, 1, 3, and 4 (terminal DSQs: local and global),
+ * ops.dequeue() should never be called because tasks bypass the
+ * BPF scheduler entirely. If we get here, it's a kernel bug.
+ */
+ if (test_scenario == 0 || test_scenario == 3) {
+ scx_bpf_error("%d (%s): dequeue called for local DSQ scenario",
+ p->pid, p->comm);
+ return;
+ }
+
+ if (test_scenario == 1 || test_scenario == 4) {
+ scx_bpf_error("%d (%s): dequeue called for global DSQ scenario",
+ p->pid, p->comm);
+ return;
+ }
+
+ if (deq_flags & SCX_DEQ_SCHED_CHANGE) {
+ /*
+ * Property change interrupting the workflow. Valid from
+ * both ENQUEUED and DISPATCHED states. Transitions task
+ * back to NONE state.
+ */
+ __sync_fetch_and_add(&change_dequeue_cnt, 1);
+
+ /* Validate state transition */
+ if (tctx->state != TASK_ENQUEUED && tctx->state != TASK_DISPATCHED)
+ scx_bpf_error("%d (%s): invalid property change dequeue state=%d seq=%llu",
+ p->pid, p->comm, tctx->state, tctx->enqueue_seq);
+
+ /*
+ * Transition back to NONE: task outside scheduler control.
+ *
+ * Scenario 6: dispatch() checks tctx->state after popping a
+ * PID, if the task is in state NONE, it was dequeued by
+ * property change and must not be dispatched (this
+ * prevents "target CPU not allowed").
+ */
+ tctx->state = TASK_NONE;
+ } else {
+ /*
+ * Regular dispatch dequeue: kernel is moving the task from
+ * BPF custody to a terminal DSQ. Normally we come from
+ * ENQUEUED state. We can also see TASK_NONE if the task
+ * was dequeued by property change (SCX_DEQ_SCHED_CHANGE)
+ * while it was already on a DSQ (dispatched but not yet
+ * consumed); in that case we just leave state as NONE.
+ */
+ __sync_fetch_and_add(&dispatch_dequeue_cnt, 1);
+
+ /*
+ * Must be ENQUEUED (normal path) or NONE (already dequeued
+ * by property change while on a DSQ).
+ */
+ if (tctx->state != TASK_ENQUEUED && tctx->state != TASK_NONE)
+ scx_bpf_error("%d (%s): dispatch dequeue from state %d seq=%llu",
+ p->pid, p->comm, tctx->state, tctx->enqueue_seq);
+
+ if (tctx->state == TASK_ENQUEUED)
+ tctx->state = TASK_DISPATCHED;
+
+ /* NONE: leave as-is, task was already property-change dequeued */
+ }
+}
+
+void BPF_STRUCT_OPS(dequeue_dispatch, s32 cpu, struct task_struct *prev)
+{
+ if (test_scenario == 6) {
+ struct task_ctx *tctx;
+ struct task_struct *p;
+ s32 pid;
+
+ if (bpf_map_pop_elem(&global_queue, &pid))
+ return;
+
+ p = bpf_task_from_pid(pid);
+ if (!p)
+ return;
+
+ /*
+ * If the task was dequeued by property change
+ * (ops.dequeue() set tctx->state = TASK_NONE), skip
+ * dispatch.
+ */
+ tctx = try_lookup_task_ctx(p);
+ if (!tctx || tctx->state == TASK_NONE) {
+ bpf_task_release(p);
+ return;
+ }
+
+ /*
+ * Dispatch to this CPU's local DSQ if allowed, otherwise
+ * fallback to the global DSQ.
+ */
+ if (bpf_cpumask_test_cpu(cpu, p->cpus_ptr))
+ scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL_ON | cpu, SCX_SLICE_DFL, 0);
+ else
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, 0);
+
+ bpf_task_release(p);
+ } else {
+ scx_bpf_dsq_move_to_local(SHARED_DSQ, 0);
+ }
+}
+
+s32 BPF_STRUCT_OPS(dequeue_init_task, struct task_struct *p,
+ struct scx_init_task_args *args)
+{
+ struct task_ctx *tctx;
+
+ tctx = bpf_task_storage_get(&task_ctx_stor, p, 0,
+ BPF_LOCAL_STORAGE_GET_F_CREATE);
+ if (!tctx)
+ return -ENOMEM;
+
+ return 0;
+}
+
+s32 BPF_STRUCT_OPS_SLEEPABLE(dequeue_init)
+{
+ s32 ret;
+
+ ret = scx_bpf_create_dsq(SHARED_DSQ, -1);
+ if (ret)
+ return ret;
+
+ return 0;
+}
+
+void BPF_STRUCT_OPS(dequeue_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops dequeue_ops = {
+ .select_cpu = (void *)dequeue_select_cpu,
+ .enqueue = (void *)dequeue_enqueue,
+ .dequeue = (void *)dequeue_dequeue,
+ .dispatch = (void *)dequeue_dispatch,
+ .init_task = (void *)dequeue_init_task,
+ .init = (void *)dequeue_init,
+ .exit = (void *)dequeue_exit,
+ .flags = SCX_OPS_ENQ_LAST,
+ .name = "dequeue_test",
+};
diff --git a/tools/testing/selftests/sched_ext/dequeue.c b/tools/testing/selftests/sched_ext/dequeue.c
new file mode 100644
index 000000000000..383d06e972a4
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/dequeue.c
@@ -0,0 +1,275 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2025 NVIDIA Corporation.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <unistd.h>
+#include <signal.h>
+#include <time.h>
+#include <bpf/bpf.h>
+#include <scx/common.h>
+#include <sys/wait.h>
+#include <sched.h>
+#include <pthread.h>
+#include "scx_test.h"
+#include "dequeue.bpf.skel.h"
+
+#define NUM_WORKERS 8
+#define AFFINITY_HAMMER_MS 500
+
+/*
+ * Worker function that creates enqueue/dequeue events via CPU work and
+ * sleep.
+ */
+static void worker_fn(int id)
+{
+ int i;
+ volatile int sum = 0;
+
+ for (i = 0; i < 1000; i++) {
+ volatile int j;
+
+ /* Do some work to trigger scheduling events */
+ for (j = 0; j < 10000; j++)
+ sum += j;
+ asm volatile("" : : "r"(sum));
+
+ /* Sleep to trigger dequeue */
+ usleep(1000 + (id * 100));
+ }
+
+ exit(0);
+}
+
+/*
+ * This thread changes workers' affinity from outside so that some changes
+ * hit tasks while they are still in the scheduler's queue and trigger
+ * property-change dequeues.
+ */
+static void *affinity_hammer_fn(void *arg)
+{
+ pid_t *pids = arg;
+ cpu_set_t cpuset;
+ int i = 0, n = NUM_WORKERS;
+ struct timespec start, now;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+ while (1) {
+ int w = i % n;
+ int cpu = (i / n) % 4;
+
+ CPU_ZERO(&cpuset);
+ CPU_SET(cpu, &cpuset);
+ sched_setaffinity(pids[w], sizeof(cpuset), &cpuset);
+ i++;
+
+ /* Check elapsed time every 256 iterations to limit gettime cost */
+ if ((i & 255) == 0) {
+ long long elapsed_ms;
+
+ clock_gettime(CLOCK_MONOTONIC, &now);
+ elapsed_ms = (now.tv_sec - start.tv_sec) * 1000LL +
+ (now.tv_nsec - start.tv_nsec) / 1000000;
+ if (elapsed_ms >= AFFINITY_HAMMER_MS)
+ break;
+ }
+ }
+ return NULL;
+}
+
+static enum scx_test_status run_scenario(struct dequeue *skel, u32 scenario,
+ const char *scenario_name)
+{
+ struct bpf_link *link;
+ pid_t pids[NUM_WORKERS];
+ pthread_t hammer;
+
+ int i, status;
+ u64 enq_start, deq_start,
+ dispatch_deq_start, change_deq_start, bpf_queue_full_start;
+ u64 enq_delta, deq_delta,
+ dispatch_deq_delta, change_deq_delta, bpf_queue_full_delta;
+
+ /* Set the test scenario */
+ skel->bss->test_scenario = scenario;
+
+ /* Record starting counts */
+ enq_start = skel->bss->enqueue_cnt;
+ deq_start = skel->bss->dequeue_cnt;
+ dispatch_deq_start = skel->bss->dispatch_dequeue_cnt;
+ change_deq_start = skel->bss->change_dequeue_cnt;
+ bpf_queue_full_start = skel->bss->bpf_queue_full;
+
+ link = bpf_map__attach_struct_ops(skel->maps.dequeue_ops);
+ SCX_FAIL_IF(!link, "Failed to attach struct_ops for scenario %s", scenario_name);
+
+ /* Fork worker processes to generate enqueue/dequeue events */
+ for (i = 0; i < NUM_WORKERS; i++) {
+ pids[i] = fork();
+ SCX_FAIL_IF(pids[i] < 0, "Failed to fork worker %d", i);
+
+ if (pids[i] == 0) {
+ worker_fn(i);
+ /* Should not reach here */
+ exit(1);
+ }
+ }
+
+ /*
+ * Run an "affinity hammer" so that some property changes hit tasks
+ * while they are still in BPF custody (e.g., in user DSQ or BPF
+ * queue), triggering SCX_DEQ_SCHED_CHANGE dequeues.
+ */
+ SCX_FAIL_IF(pthread_create(&hammer, NULL, affinity_hammer_fn, pids) != 0,
+ "Failed to create affinity hammer thread");
+ pthread_join(hammer, NULL);
+
+ /* Wait for all workers to complete */
+ for (i = 0; i < NUM_WORKERS; i++) {
+ SCX_FAIL_IF(waitpid(pids[i], &status, 0) != pids[i],
+ "Failed to wait for worker %d", i);
+ SCX_FAIL_IF(status != 0, "Worker %d exited with status %d", i, status);
+ }
+
+ bpf_link__destroy(link);
+
+ SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG));
+
+ /* Calculate deltas */
+ enq_delta = skel->bss->enqueue_cnt - enq_start;
+ deq_delta = skel->bss->dequeue_cnt - deq_start;
+ dispatch_deq_delta = skel->bss->dispatch_dequeue_cnt - dispatch_deq_start;
+ change_deq_delta = skel->bss->change_dequeue_cnt - change_deq_start;
+ bpf_queue_full_delta = skel->bss->bpf_queue_full - bpf_queue_full_start;
+
+ printf("%s:\n", scenario_name);
+ printf(" enqueues: %lu\n", (unsigned long)enq_delta);
+ printf(" dequeues: %lu (dispatch: %lu, property_change: %lu)\n",
+ (unsigned long)deq_delta,
+ (unsigned long)dispatch_deq_delta,
+ (unsigned long)change_deq_delta);
+ printf(" BPF queue full: %lu\n", (unsigned long)bpf_queue_full_delta);
+
+ /*
+ * Validate enqueue/dequeue lifecycle tracking.
+ *
+ * For scenarios 0, 1, 3, 4 (local and global DSQs from
+ * ops.select_cpu() and ops.enqueue()), both enqueues and dequeues
+ * should be 0 because tasks bypass the BPF scheduler entirely:
+ * tasks never enter BPF scheduler's custody.
+ *
+ * For scenarios 2, 5, 6 (user DSQ or BPF internal queue) we expect
+ * both enqueues and dequeues.
+ *
+ * The BPF code does strict state machine validation with
+ * scx_bpf_error() to ensure the workflow semantics are correct.
+ *
+ * If we reach this point without errors, the semantics are
+ * validated correctly.
+ */
+ if (scenario == 0 || scenario == 1 ||
+ scenario == 3 || scenario == 4) {
+ /* Tasks bypass BPF scheduler completely */
+ SCX_EQ(enq_delta, 0);
+ SCX_EQ(deq_delta, 0);
+ SCX_EQ(dispatch_deq_delta, 0);
+ SCX_EQ(change_deq_delta, 0);
+ } else {
+ /*
+ * User DSQ from ops.enqueue() or ops.select_cpu(): tasks
+ * enter BPF scheduler's custody.
+ *
+ * Also validate 1:1 enqueue/dequeue pairing.
+ */
+ SCX_GT(enq_delta, 0);
+ SCX_GT(deq_delta, 0);
+ SCX_EQ(enq_delta, deq_delta);
+ }
+
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+ struct dequeue *skel;
+
+ skel = dequeue__open();
+ SCX_FAIL_IF(!skel, "Failed to open skel");
+ SCX_ENUM_INIT(skel);
+ SCX_FAIL_IF(dequeue__load(skel), "Failed to load skel");
+
+ *ctx = skel;
+
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct dequeue *skel = ctx;
+ enum scx_test_status status;
+
+ status = run_scenario(skel, 0, "Scenario 0: Local DSQ from ops.select_cpu()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ status = run_scenario(skel, 1, "Scenario 1: Global DSQ from ops.select_cpu()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ status = run_scenario(skel, 2, "Scenario 2: User DSQ from ops.select_cpu()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ status = run_scenario(skel, 3, "Scenario 3: Local DSQ from ops.enqueue()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ status = run_scenario(skel, 4, "Scenario 4: Global DSQ from ops.enqueue()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ status = run_scenario(skel, 5, "Scenario 5: User DSQ from ops.enqueue()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ status = run_scenario(skel, 6, "Scenario 6: BPF queue from ops.enqueue()");
+ if (status != SCX_TEST_PASS)
+ return status;
+
+ printf("\n=== Summary ===\n");
+ printf("Total enqueues: %lu\n", (unsigned long)skel->bss->enqueue_cnt);
+ printf("Total dequeues: %lu\n", (unsigned long)skel->bss->dequeue_cnt);
+ printf(" Dispatch dequeues: %lu (no flag, normal workflow)\n",
+ (unsigned long)skel->bss->dispatch_dequeue_cnt);
+ printf(" Property change dequeues: %lu (SCX_DEQ_SCHED_CHANGE flag)\n",
+ (unsigned long)skel->bss->change_dequeue_cnt);
+ printf(" BPF queue full: %lu\n",
+ (unsigned long)skel->bss->bpf_queue_full);
+ printf("\nAll scenarios passed - no state machine violations detected\n");
+ printf("-> Validated: Local DSQ dispatch bypasses BPF scheduler\n");
+ printf("-> Validated: Global DSQ dispatch bypasses BPF scheduler\n");
+ printf("-> Validated: User DSQ dispatch triggers ops.dequeue() callbacks\n");
+ printf("-> Validated: Dispatch dequeues have no flags (normal workflow)\n");
+ printf("-> Validated: Property change dequeues have SCX_DEQ_SCHED_CHANGE flag\n");
+ printf("-> Validated: No duplicate enqueues or invalid state transitions\n");
+
+ return SCX_TEST_PASS;
+}
+
+static void cleanup(void *ctx)
+{
+ struct dequeue *skel = ctx;
+
+ dequeue__destroy(skel);
+}
+
+struct scx_test dequeue_test = {
+ .name = "dequeue",
+ .description = "Verify ops.dequeue() semantics",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+
+REGISTER_SCX_TEST(&dequeue_test)
diff --git a/tools/testing/selftests/sched_ext/dequeue_iter.bpf.c b/tools/testing/selftests/sched_ext/dequeue_iter.bpf.c
new file mode 100644
index 000000000000..76c2c71a90b6
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/dequeue_iter.bpf.c
@@ -0,0 +1,73 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * ops.dequeue() of this scheduler iterates the user DSQ it consumes
+ * tasks from with bpf_iter_scx_dsq, which takes the DSQ lock.
+ * On a kernel that still runs ops.dequeue() with that lock held, the
+ * iteration self-deadlocks the CPU - this test wedges the system on
+ * unfixed kernels instead of failing cleanly.
+ *
+ * Copyright (c) 2026 fangqiurong <fangqiurong@kylinos.cn>
+ */
+
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+UEI_DEFINE(uei);
+
+#define TEST_DSQ_ID 1000
+
+u64 dq_count;
+
+s32 BPF_STRUCT_OPS_SLEEPABLE(dequeue_iter_init)
+{
+ return scx_bpf_create_dsq(TEST_DSQ_ID, -1);
+}
+
+s32 BPF_STRUCT_OPS(dequeue_iter_select_cpu, struct task_struct *p,
+ s32 prev_cpu, u64 wake_flags)
+{
+ return prev_cpu;
+}
+
+void BPF_STRUCT_OPS(dequeue_iter_enqueue, struct task_struct *p, u64 enq_flags)
+{
+ scx_bpf_dsq_insert(p, TEST_DSQ_ID, SCX_SLICE_DFL, enq_flags);
+}
+
+void BPF_STRUCT_OPS(dequeue_iter_dispatch, s32 cpu, struct task_struct *task)
+{
+ scx_bpf_dsq_move_to_local(TEST_DSQ_ID, 0);
+}
+
+void BPF_STRUCT_OPS(dequeue_iter_dequeue, struct task_struct *p, u64 deq_flags)
+{
+ struct bpf_iter_scx_dsq it;
+ struct task_struct *t;
+
+ if (!bpf_iter_scx_dsq_new(&it, TEST_DSQ_ID, 0)) {
+ while ((t = bpf_iter_scx_dsq_next(&it)))
+ ;
+ }
+ bpf_iter_scx_dsq_destroy(&it);
+
+ __sync_fetch_and_add(&dq_count, 1);
+}
+
+void BPF_STRUCT_OPS(dequeue_iter_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+ scx_bpf_destroy_dsq(TEST_DSQ_ID);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops dequeue_iter_ops = {
+ .init = (void *)dequeue_iter_init,
+ .select_cpu = (void *)dequeue_iter_select_cpu,
+ .enqueue = (void *)dequeue_iter_enqueue,
+ .dispatch = (void *)dequeue_iter_dispatch,
+ .dequeue = (void *)dequeue_iter_dequeue,
+ .exit = (void *)dequeue_iter_exit,
+ .timeout_ms = 1000U,
+ .name = "dequeue_iter",
+};
diff --git a/tools/testing/selftests/sched_ext/dequeue_iter.c b/tools/testing/selftests/sched_ext/dequeue_iter.c
new file mode 100644
index 000000000000..f60711bf3b39
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/dequeue_iter.c
@@ -0,0 +1,79 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 fangqiurong <fangqiurong@kylinos.cn>
+ */
+#include <bpf/bpf.h>
+#include <scx/common.h>
+#include <time.h>
+#include <unistd.h>
+#include "dequeue_iter.bpf.skel.h"
+#include "scx_test.h"
+
+#define DQ_TARGET 10
+#define DQ_DEADLINE_MS 3000
+
+static unsigned long long now_ms(void)
+{
+ struct timespec ts;
+
+ clock_gettime(CLOCK_MONOTONIC, &ts);
+
+ return ts.tv_sec * 1000ULL + ts.tv_nsec / 1000000;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+ struct dequeue_iter *skel;
+
+ skel = dequeue_iter__open();
+ SCX_FAIL_IF(!skel, "Failed to open");
+ SCX_ENUM_INIT(skel);
+ SCX_FAIL_IF(dequeue_iter__load(skel), "Failed to load skel");
+
+ *ctx = skel;
+
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct dequeue_iter *skel = ctx;
+ struct bpf_link *link;
+ unsigned long long end;
+
+ link = bpf_map__attach_struct_ops(skel->maps.dequeue_iter_ops);
+ SCX_FAIL_IF(!link, "Failed to attach scheduler");
+
+ end = now_ms() + DQ_DEADLINE_MS;
+ while (skel->bss->dq_count < DQ_TARGET && !UEI_EXITED(skel, uei) &&
+ now_ms() < end)
+ usleep(100);
+
+ bpf_link__destroy(link);
+
+ SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG));
+
+ if (skel->bss->dq_count < DQ_TARGET) {
+ SCX_ERR("ops.dequeue() fired only %llu times",
+ (unsigned long long)skel->bss->dq_count);
+ return SCX_TEST_FAIL;
+ }
+
+ return SCX_TEST_PASS;
+}
+
+static void cleanup(void *ctx)
+{
+ struct dequeue_iter *skel = ctx;
+
+ dequeue_iter__destroy(skel);
+}
+
+struct scx_test dequeue_iter = {
+ .name = "dequeue_iter",
+ .description = "Verify ops.dequeue() can iterate its source user DSQ",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&dequeue_iter)
diff --git a/tools/testing/selftests/sched_ext/enable_cmask.bpf.c b/tools/testing/selftests/sched_ext/enable_cmask.bpf.c
new file mode 100644
index 000000000000..0068f3c7ab3c
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/enable_cmask.bpf.c
@@ -0,0 +1,217 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A cid-form scheduler checking the cmask cid-form ops.enable() receives: the
+ * header, every cid bit against p->cpus_ptr, and that set_cmask() follows with
+ * the same mask before set_weight() and before the task first becomes runnable,
+ * and never runs before enable().
+ *
+ * Copyright (c) 2026 Tejun Heo <tj@kernel.org>
+ */
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+struct {
+ __uint(type, BPF_MAP_TYPE_ARENA);
+ __uint(map_flags, BPF_F_MMAPABLE);
+ __uint(max_entries, 1 << 16);
+} arena SEC(".maps");
+
+struct task_ctx {
+ u64 enable_fp; /* fingerprint of the mask enable() received */
+ bool enabled;
+ bool pending; /* enable() ran, the initial set_cmask() hasn't */
+};
+
+struct {
+ __uint(type, BPF_MAP_TYPE_TASK_STORAGE);
+ __uint(map_flags, BPF_F_NO_PREALLOC);
+ __type(key, int);
+ __type(value, struct task_ctx);
+} task_ctx_stor SEC(".maps");
+
+/* details of a cid bit mismatch, filled by check_mask() */
+struct mask_mismatch {
+ s32 cid;
+ bool want;
+ bool got;
+};
+
+u64 nr_enable, nr_initial_set_cmask, nr_set_cmask, nr_set_weight;
+
+UEI_DEFINE(uei);
+
+static struct task_ctx *lookup_task_ctx(struct task_struct *p)
+{
+ struct task_ctx *tctx;
+
+ tctx = bpf_task_storage_get(&task_ctx_stor, p, 0, 0);
+ if (!tctx)
+ scx_bpf_error("task_ctx lookup failed for %s[%d]", p->comm, p->pid);
+ return tctx;
+}
+
+/*
+ * Verify @m's header and every cid bit against @p's cpumask and fingerprint the
+ * bits into @fp. Return 0 on success, -EINVAL on a bad header, -ENOENT on a cid
+ * without a cpu and -EIO on a bit mismatch with the details in @mm.
+ */
+static int check_mask(struct task_struct *p, const struct scx_cmask __arena *m, u64 *fp,
+ struct mask_mismatch *mm)
+{
+ u32 nr_cids = scx_bpf_nr_cids();
+ u64 h = 0;
+ s32 cid;
+
+ if (m->base || m->nr_cids != nr_cids || m->alloc_words != CMASK_NR_WORDS(nr_cids))
+ return -EINVAL;
+
+ bpf_for(cid, 0, nr_cids) {
+ bool want, got;
+ s32 cpu;
+
+ cpu = scx_bpf_cid_to_cpu(cid);
+ if (cpu < 0)
+ return -ENOENT;
+ want = bpf_cpumask_test_cpu(cpu, p->cpus_ptr);
+ got = cmask_test(cid, m);
+ if (want != got) {
+ mm->cid = cid;
+ mm->want = want;
+ mm->got = got;
+ return -EIO;
+ }
+ h = h * 31 + got;
+ }
+
+ *fp = h;
+ return 0;
+}
+
+s32 BPF_STRUCT_OPS_SLEEPABLE(enable_cmask_init_task, struct task_struct *p,
+ struct scx_init_task_args *args)
+{
+ if (!bpf_task_storage_get(&task_ctx_stor, p, 0, BPF_LOCAL_STORAGE_GET_F_CREATE))
+ return -ENOMEM;
+ return 0;
+}
+
+void BPF_STRUCT_OPS(enable_cmask_enable, struct task_struct *p, struct scx_enable_args *args)
+{
+ struct scx_cmask __arena *m = (struct scx_cmask __arena *)args->cmask_arena_addr;
+ struct mask_mismatch mm = {};
+ struct task_ctx *tctx;
+ int ret;
+
+ asm volatile("" :: "r"(&arena));
+ tctx = lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ __sync_fetch_and_add(&nr_enable, 1);
+ if (tctx->enabled || tctx->pending) {
+ scx_bpf_error("enable: %s[%d] enabled twice", p->comm, p->pid);
+ return;
+ }
+
+ ret = check_mask(p, m, &tctx->enable_fp, &mm);
+ if (ret) {
+ scx_bpf_error("enable: %s[%d] cmask check failed %d cid=%d want=%d got=%d",
+ p->comm, p->pid, ret, mm.cid, mm.want, mm.got);
+ return;
+ }
+ tctx->enabled = true;
+ tctx->pending = true;
+}
+
+void BPF_STRUCT_OPS(enable_cmask_set_cmask, struct task_struct *p,
+ struct scx_cmask __arena *m)
+{
+ struct mask_mismatch mm = {};
+ struct task_ctx *tctx;
+ u64 fp;
+ int ret;
+
+ asm volatile("" :: "r"(&arena));
+ tctx = lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ __sync_fetch_and_add(&nr_set_cmask, 1);
+ if (!tctx->enabled) {
+ scx_bpf_error("set_cmask: %s[%d] not enabled", p->comm, p->pid);
+ return;
+ }
+
+ ret = check_mask(p, m, &fp, &mm);
+ if (ret) {
+ scx_bpf_error("set_cmask: %s[%d] cmask check failed %d cid=%d want=%d got=%d",
+ p->comm, p->pid, ret, mm.cid, mm.want, mm.got);
+ return;
+ }
+
+ if (tctx->pending) {
+ if (fp != tctx->enable_fp) {
+ scx_bpf_error("set_cmask: %s[%d] initial mask differs from enable()",
+ p->comm, p->pid);
+ return;
+ }
+ tctx->pending = false;
+ __sync_fetch_and_add(&nr_initial_set_cmask, 1);
+ }
+}
+
+void BPF_STRUCT_OPS(enable_cmask_set_weight, struct task_struct *p, u32 weight)
+{
+ struct task_ctx *tctx;
+
+ tctx = lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ __sync_fetch_and_add(&nr_set_weight, 1);
+ if (tctx->pending)
+ scx_bpf_error("set_weight: %s[%d] before the initial set_cmask()", p->comm,
+ p->pid);
+}
+
+void BPF_STRUCT_OPS(enable_cmask_runnable, struct task_struct *p, u64 enq_flags)
+{
+ struct task_ctx *tctx;
+
+ tctx = lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ if (tctx->pending)
+ scx_bpf_error("runnable: %s[%d] before the initial set_cmask()", p->comm,
+ p->pid);
+}
+
+void BPF_STRUCT_OPS(enable_cmask_disable, struct task_struct *p)
+{
+ struct task_ctx *tctx;
+
+ tctx = lookup_task_ctx(p);
+ if (!tctx)
+ return;
+
+ tctx->enabled = false;
+ tctx->pending = false;
+}
+
+void BPF_STRUCT_OPS(enable_cmask_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SCX_OPS_CID_DEFINE(enable_cmask_ops,
+ .init_task = (void *)enable_cmask_init_task,
+ .enable = (void *)enable_cmask_enable,
+ .set_cmask = (void *)enable_cmask_set_cmask,
+ .set_weight = (void *)enable_cmask_set_weight,
+ .runnable = (void *)enable_cmask_runnable,
+ .disable = (void *)enable_cmask_disable,
+ .exit = (void *)enable_cmask_exit,
+ .flags = SCX_OPS_SWITCH_PARTIAL,
+ .name = "enable_cmask");
diff --git a/tools/testing/selftests/sched_ext/enable_cmask.c b/tools/testing/selftests/sched_ext/enable_cmask.c
new file mode 100644
index 000000000000..556bcad4431d
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/enable_cmask.c
@@ -0,0 +1,138 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Tejun Heo <tj@kernel.org> */
+#define _GNU_SOURCE
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <time.h>
+#include <unistd.h>
+#include <sys/wait.h>
+#include <bpf/bpf.h>
+#include <scx/common.h>
+#include "enable_cmask.bpf.skel.h"
+#include "scx_test.h"
+
+#define SCHED_EXT 7
+#define NR_CHILDREN 8
+#define MAX_CPUS 1024
+
+static int cpus[MAX_CPUS];
+static int nr_cpus;
+
+static void spin_ms(int ms)
+{
+ struct timespec start, now;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+ do {
+ clock_gettime(CLOCK_MONOTONIC, &now);
+ } while ((now.tv_sec - start.tv_sec) * 1000 +
+ (now.tv_nsec - start.tv_nsec) / 1000000 < ms);
+}
+
+static int pin(pid_t pid, int idx)
+{
+ cpu_set_t set;
+
+ CPU_ZERO(&set);
+ CPU_SET(cpus[idx % nr_cpus], &set);
+ return sched_setaffinity(pid, sizeof(set), &set);
+}
+
+/*
+ * Pin, switch to SCHED_EXT for a class-switch enable, fork a grandchild that
+ * inherits the policy for a fork-path enable, then change affinity a few times
+ * while running for set_cmask() on live tasks.
+ */
+static int child(int idx)
+{
+ struct sched_param param = {};
+ int i, status;
+ pid_t pid;
+
+ if (pin(0, idx) || sched_setscheduler(0, SCHED_EXT, &param))
+ return 1;
+
+ pid = fork();
+ if (pid < 0)
+ return 1;
+ if (!pid) {
+ spin_ms(20);
+ return 0;
+ }
+
+ for (i = 1; i <= 4; i++) {
+ if (pin(0, idx + i))
+ return 1;
+ spin_ms(5);
+ }
+
+ return waitpid(pid, &status, 0) == pid && !status ? 0 : 1;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct enable_cmask *skel;
+ struct bpf_link *link;
+ pid_t pids[NR_CHILDREN];
+ cpu_set_t set;
+ int i, status, failed = 0;
+
+ if (!__COMPAT_struct_has_field("scx_enable_args", "cmask_arena_addr"))
+ return SCX_TEST_SKIP;
+
+ SCX_FAIL_IF(sched_getaffinity(0, sizeof(set), &set), "Failed to read affinity");
+ for (i = 0; i < MAX_CPUS && i < CPU_SETSIZE; i++)
+ if (CPU_ISSET(i, &set))
+ cpus[nr_cpus++] = i;
+ if (nr_cpus < 2)
+ return SCX_TEST_SKIP;
+
+ skel = enable_cmask__open();
+ SCX_FAIL_IF(!skel, "Failed to open");
+ SCX_ENUM_INIT(skel);
+ SCX_FAIL_IF(enable_cmask__load(skel), "Failed to load skel");
+
+ link = bpf_map__attach_struct_ops(skel->maps.enable_cmask_ops);
+ SCX_FAIL_IF(!link, "Failed to attach struct_ops");
+
+ for (i = 0; i < NR_CHILDREN; i++) {
+ pids[i] = fork();
+ SCX_FAIL_IF(pids[i] < 0, "Failed to fork");
+ if (!pids[i])
+ exit(child(i));
+ }
+
+ /* affinity changes from the outside race with the children's own */
+ for (i = 0; i < NR_CHILDREN; i++)
+ pin(pids[i], i + NR_CHILDREN);
+
+ for (i = 0; i < NR_CHILDREN; i++) {
+ if (waitpid(pids[i], &status, 0) != pids[i] || status)
+ failed++;
+ }
+
+ bpf_link__destroy(link);
+
+ SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG));
+ SCX_EQ(failed, 0);
+ SCX_GE(skel->bss->nr_enable, 2 * NR_CHILDREN);
+ SCX_EQ(skel->bss->nr_initial_set_cmask, skel->bss->nr_enable);
+ SCX_GT(skel->bss->nr_set_cmask, skel->bss->nr_initial_set_cmask);
+ SCX_GE(skel->bss->nr_set_weight, skel->bss->nr_enable);
+ printf("enable=%lu initial_set_cmask=%lu set_cmask=%lu set_weight=%lu\n",
+ (unsigned long)skel->bss->nr_enable,
+ (unsigned long)skel->bss->nr_initial_set_cmask,
+ (unsigned long)skel->bss->nr_set_cmask,
+ (unsigned long)skel->bss->nr_set_weight);
+
+ enable_cmask__destroy(skel);
+ return SCX_TEST_PASS;
+}
+
+struct scx_test enable_cmask = {
+ .name = "enable_cmask",
+ .description = "Check the cid-form ops.enable() cmask and the set_cmask() after it",
+ .run = run,
+};
+REGISTER_SCX_TEST(&enable_cmask)
diff --git a/tools/testing/selftests/sched_ext/exit.bpf.c b/tools/testing/selftests/sched_ext/exit.bpf.c
index 4bc36182d3ff..2e848820a44b 100644
--- a/tools/testing/selftests/sched_ext/exit.bpf.c
+++ b/tools/testing/selftests/sched_ext/exit.bpf.c
@@ -41,7 +41,7 @@ void BPF_STRUCT_OPS(exit_dispatch, s32 cpu, struct task_struct *p)
if (exit_point == EXIT_DISPATCH)
EXIT_CLEANLY();
- scx_bpf_dsq_move_to_local(DSQ_ID);
+ scx_bpf_dsq_move_to_local(DSQ_ID, 0);
}
void BPF_STRUCT_OPS(exit_enable, struct task_struct *p)
diff --git a/tools/testing/selftests/sched_ext/exit.c b/tools/testing/selftests/sched_ext/exit.c
index ee25824b1cbe..01b17092d5c8 100644
--- a/tools/testing/selftests/sched_ext/exit.c
+++ b/tools/testing/selftests/sched_ext/exit.c
@@ -31,9 +31,10 @@ static enum scx_test_status run(void *ctx)
continue;
skel = exit__open();
+ SCX_FAIL_IF(!skel, "Failed to open");
SCX_ENUM_INIT(skel);
skel->rodata->exit_point = tc;
- exit__load(skel);
+ SCX_FAIL_IF(exit__load(skel), "Failed to load skel");
link = bpf_map__attach_struct_ops(skel->maps.exit_ops);
if (!link) {
SCX_ERR("Failed to attach scheduler");
diff --git a/tools/testing/selftests/sched_ext/exit_test.h b/tools/testing/selftests/sched_ext/exit_test.h
index 94f0268b9cb8..2723e0fda801 100644
--- a/tools/testing/selftests/sched_ext/exit_test.h
+++ b/tools/testing/selftests/sched_ext/exit_test.h
@@ -17,4 +17,4 @@ enum exit_test_case {
NUM_EXITS,
};
-#endif // # __EXIT_TEST_H__
+#endif // __EXIT_TEST_H__
diff --git a/tools/testing/selftests/sched_ext/init_enable_count.c b/tools/testing/selftests/sched_ext/init_enable_count.c
index eddf9e0e26e7..44577e30e764 100644
--- a/tools/testing/selftests/sched_ext/init_enable_count.c
+++ b/tools/testing/selftests/sched_ext/init_enable_count.c
@@ -4,6 +4,7 @@
* Copyright (c) 2023 David Vernet <dvernet@meta.com>
* Copyright (c) 2023 Tejun Heo <tj@kernel.org>
*/
+#include <signal.h>
#include <stdio.h>
#include <unistd.h>
#include <sched.h>
@@ -23,6 +24,9 @@ static enum scx_test_status run_test(bool global)
int ret, i, status;
struct sched_param param = {};
pid_t pids[num_pre_forks];
+ int pipe_fds[2];
+
+ SCX_FAIL_IF(pipe(pipe_fds) < 0, "Failed to create pipe");
skel = init_enable_count__open();
SCX_FAIL_IF(!skel, "Failed to open");
@@ -38,26 +42,35 @@ static enum scx_test_status run_test(bool global)
* ensure (at least in practical terms) that there are more tasks that
* transition from SCHED_OTHER -> SCHED_EXT than there are tasks that
* take the fork() path either below or in other processes.
+ *
+ * All children will block on read() on the pipe until the parent closes
+ * the write end after attaching the scheduler, which signals all of
+ * them to exit simultaneously. Auto-reap so we don't have to wait on
+ * them.
*/
+ signal(SIGCHLD, SIG_IGN);
for (i = 0; i < num_pre_forks; i++) {
- pids[i] = fork();
- SCX_FAIL_IF(pids[i] < 0, "Failed to fork child");
- if (pids[i] == 0) {
- sleep(1);
+ pid_t pid = fork();
+
+ SCX_FAIL_IF(pid < 0, "Failed to fork child");
+ if (pid == 0) {
+ char buf;
+
+ close(pipe_fds[1]);
+ if (read(pipe_fds[0], &buf, 1) < 0)
+ exit(1);
+ close(pipe_fds[0]);
exit(0);
}
}
+ close(pipe_fds[0]);
link = bpf_map__attach_struct_ops(skel->maps.init_enable_count_ops);
SCX_FAIL_IF(!link, "Failed to attach struct_ops");
- for (i = 0; i < num_pre_forks; i++) {
- SCX_FAIL_IF(waitpid(pids[i], &status, 0) != pids[i],
- "Failed to wait for pre-forked child\n");
-
- SCX_FAIL_IF(status != 0, "Pre-forked child %d exited with status %d\n", i,
- status);
- }
+ /* Signal all pre-forked children to exit. */
+ close(pipe_fds[1]);
+ signal(SIGCHLD, SIG_DFL);
bpf_link__destroy(link);
SCX_GE(skel->bss->init_task_cnt, num_pre_forks);
diff --git a/tools/testing/selftests/sched_ext/maximal.bpf.c b/tools/testing/selftests/sched_ext/maximal.bpf.c
index 01cf4f3da4e0..04a369078aac 100644
--- a/tools/testing/selftests/sched_ext/maximal.bpf.c
+++ b/tools/testing/selftests/sched_ext/maximal.bpf.c
@@ -30,7 +30,7 @@ void BPF_STRUCT_OPS(maximal_dequeue, struct task_struct *p, u64 deq_flags)
void BPF_STRUCT_OPS(maximal_dispatch, s32 cpu, struct task_struct *prev)
{
- scx_bpf_dsq_move_to_local(DSQ_ID);
+ scx_bpf_dsq_move_to_local(DSQ_ID, 0);
}
void BPF_STRUCT_OPS(maximal_runnable, struct task_struct *p, u64 enq_flags)
@@ -67,13 +67,12 @@ void BPF_STRUCT_OPS(maximal_set_cpumask, struct task_struct *p,
void BPF_STRUCT_OPS(maximal_update_idle, s32 cpu, bool idle)
{}
-void BPF_STRUCT_OPS(maximal_cpu_acquire, s32 cpu,
- struct scx_cpu_acquire_args *args)
-{}
-
-void BPF_STRUCT_OPS(maximal_cpu_release, s32 cpu,
- struct scx_cpu_release_args *args)
-{}
+SEC("tp_btf/sched_switch")
+int BPF_PROG(maximal_sched_switch, bool preempt, struct task_struct *prev,
+ struct task_struct *next, unsigned int prev_state)
+{
+ return 0;
+}
void BPF_STRUCT_OPS(maximal_cpu_online, s32 cpu)
{}
@@ -150,8 +149,6 @@ struct sched_ext_ops maximal_ops = {
.set_weight = (void *) maximal_set_weight,
.set_cpumask = (void *) maximal_set_cpumask,
.update_idle = (void *) maximal_update_idle,
- .cpu_acquire = (void *) maximal_cpu_acquire,
- .cpu_release = (void *) maximal_cpu_release,
.cpu_online = (void *) maximal_cpu_online,
.cpu_offline = (void *) maximal_cpu_offline,
.init_task = (void *) maximal_init_task,
diff --git a/tools/testing/selftests/sched_ext/maximal.c b/tools/testing/selftests/sched_ext/maximal.c
index c6be50a9941d..1dc369224670 100644
--- a/tools/testing/selftests/sched_ext/maximal.c
+++ b/tools/testing/selftests/sched_ext/maximal.c
@@ -19,6 +19,9 @@ static enum scx_test_status setup(void **ctx)
SCX_ENUM_INIT(skel);
SCX_FAIL_IF(maximal__load(skel), "Failed to load skel");
+ bpf_map__set_autoattach(skel->maps.maximal_ops, false);
+ SCX_FAIL_IF(maximal__attach(skel), "Failed to attach skel");
+
*ctx = skel;
return SCX_TEST_PASS;
diff --git a/tools/testing/selftests/sched_ext/nohz_tick.bpf.c b/tools/testing/selftests/sched_ext/nohz_tick.bpf.c
new file mode 100644
index 000000000000..6998c5dd6bcb
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/nohz_tick.bpf.c
@@ -0,0 +1,65 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES
+ *
+ * Exercise tick dependency transitions between infinite and finite slices.
+ */
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+const volatile s32 test_cpu;
+bool finite_phase;
+u64 nr_inf_running;
+u64 nr_finite_running;
+u64 nr_finite_ticks;
+
+UEI_DEFINE(uei);
+
+s32 BPF_STRUCT_OPS(nohz_tick_select_cpu, struct task_struct *p, s32 prev_cpu,
+ u64 wake_flags)
+{
+ return prev_cpu;
+}
+
+void BPF_STRUCT_OPS(nohz_tick_enqueue, struct task_struct *p, u64 enq_flags)
+{
+ u64 slice = finite_phase ? 1000000ULL : SCX_SLICE_INF;
+
+ scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, slice, enq_flags);
+ if (enq_flags & SCX_ENQ_LAST)
+ scx_bpf_kick_cpu(test_cpu, SCX_KICK_IDLE);
+}
+
+void BPF_STRUCT_OPS(nohz_tick_running, struct task_struct *p)
+{
+ if (bpf_get_smp_processor_id() != test_cpu)
+ return;
+
+ if (finite_phase)
+ __sync_fetch_and_add(&nr_finite_running, 1);
+ else
+ __sync_fetch_and_add(&nr_inf_running, 1);
+}
+
+void BPF_STRUCT_OPS(nohz_tick_tick, struct task_struct *p)
+{
+ if (bpf_get_smp_processor_id() == test_cpu && finite_phase)
+ __sync_fetch_and_add(&nr_finite_ticks, 1);
+}
+
+void BPF_STRUCT_OPS(nohz_tick_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops nohz_tick_ops = {
+ .select_cpu = (void *)nohz_tick_select_cpu,
+ .enqueue = (void *)nohz_tick_enqueue,
+ .running = (void *)nohz_tick_running,
+ .tick = (void *)nohz_tick_tick,
+ .exit = (void *)nohz_tick_exit,
+ .name = "nohz_tick",
+ .timeout_ms = 1000U,
+};
diff --git a/tools/testing/selftests/sched_ext/nohz_tick.c b/tools/testing/selftests/sched_ext/nohz_tick.c
new file mode 100644
index 000000000000..028f54391c2c
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/nohz_tick.c
@@ -0,0 +1,347 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES
+ *
+ * Validate that a finite-slice EXT task restarts the scheduler tick when it
+ * follows an infinite-slice EXT task and an idle interval on a NOHZ_FULL CPU.
+ */
+#define _GNU_SOURCE
+
+#include <bpf/bpf.h>
+#include <errno.h>
+#include <sched.h>
+#include <signal.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <sys/prctl.h>
+#include <sys/wait.h>
+#include <unistd.h>
+
+#include <scx/common.h>
+
+#include "nohz_tick.bpf.skel.h"
+#include "scx_test.h"
+
+#ifndef SCHED_EXT
+#define SCHED_EXT 7
+#endif
+
+#define MIN_FINITE_TICKS 3
+#define PHASE_TIMEOUT_MS 1000
+
+struct nohz_tick_ctx {
+ struct nohz_tick *skel;
+ cpu_set_t original_mask;
+ int test_cpu;
+};
+
+static int first_allowed_cpu(const cpu_set_t *mask, int first, int last)
+{
+ int cpu;
+
+ for (cpu = first; cpu <= last && cpu < CPU_SETSIZE; cpu++)
+ if (CPU_ISSET(cpu, mask))
+ return cpu;
+
+ return -1;
+}
+
+static int find_nohz_full_cpu(const cpu_set_t *allowed)
+{
+ char buf[4096], *cur, *end;
+ FILE *file;
+
+ file = fopen("/sys/devices/system/cpu/nohz_full", "r");
+ if (!file)
+ return -1;
+ if (!fgets(buf, sizeof(buf), file)) {
+ fclose(file);
+ return -1;
+ }
+ fclose(file);
+
+ cur = buf;
+ while (*cur) {
+ long first, last;
+ int cpu;
+
+ while (*cur == ' ' || *cur == '\t' || *cur == ',')
+ cur++;
+ if (*cur < '0' || *cur > '9')
+ break;
+
+ errno = 0;
+ first = strtol(cur, &end, 10);
+ if (errno || end == cur || first < 0 || first >= CPU_SETSIZE)
+ return -1;
+ cur = end;
+ last = first;
+ if (*cur == '-') {
+ cur++;
+ errno = 0;
+ last = strtol(cur, &end, 10);
+ if (errno || end == cur || last < first)
+ return -1;
+ cur = end;
+ }
+
+ cpu = first_allowed_cpu(allowed, first, last);
+ if (cpu >= 0)
+ return cpu;
+ }
+
+ return -1;
+}
+
+static pid_t start_worker(int cpu)
+{
+ struct sched_param param = {};
+ cpu_set_t mask;
+ pid_t parent;
+ pid_t pid;
+
+ parent = getpid();
+ pid = fork();
+ if (pid != 0)
+ return pid;
+ if (prctl(PR_SET_PDEATHSIG, SIGKILL) || getppid() != parent)
+ _exit(1);
+
+ /*
+ * Become EXT before touching the target so it stays idle until wakeup.
+ */
+ if (sched_setscheduler(0, SCHED_EXT, &param))
+ _exit(1);
+
+ CPU_ZERO(&mask);
+ CPU_SET(cpu, &mask);
+ if (sched_setaffinity(0, sizeof(mask), &mask))
+ _exit(1);
+
+ for (;;)
+ asm volatile("" ::: "memory");
+}
+
+static void stop_worker(pid_t pid)
+{
+ if (pid <= 0)
+ return;
+
+ kill(pid, SIGKILL);
+ waitpid(pid, NULL, 0);
+}
+
+static int pause_worker(pid_t pid)
+{
+ int status;
+
+ if (kill(pid, SIGSTOP))
+ return -errno;
+ if (waitpid(pid, &status, WUNTRACED) != pid)
+ return -errno;
+ if (!WIFSTOPPED(status))
+ return -ECHILD;
+
+ return 0;
+}
+
+static bool wait_for_counter(const u64 *counter, u64 value, int timeout_ms)
+{
+ int elapsed;
+
+ for (elapsed = 0; elapsed < timeout_ms; elapsed++) {
+ if (__atomic_load_n(counter, __ATOMIC_RELAXED) >= value)
+ return true;
+ usleep(1000);
+ }
+
+ return false;
+}
+
+static enum scx_test_status setup(void **ctx_ptr)
+{
+ struct nohz_tick_ctx *ctx;
+ cpu_set_t controller_mask;
+ int cpu;
+
+ ctx = calloc(1, sizeof(*ctx));
+ SCX_FAIL_IF(!ctx, "Failed to allocate context");
+ if (sched_getaffinity(0, sizeof(ctx->original_mask),
+ &ctx->original_mask)) {
+ free(ctx);
+ SCX_FAIL("Failed to get affinity (%d)", errno);
+ }
+
+ cpu = find_nohz_full_cpu(&ctx->original_mask);
+ if (cpu < 0) {
+ fprintf(stderr, "SKIP: no allowed NOHZ_FULL CPU\n");
+ free(ctx);
+ return SCX_TEST_SKIP;
+ }
+
+ controller_mask = ctx->original_mask;
+ CPU_CLR(cpu, &controller_mask);
+ if (CPU_COUNT(&controller_mask) == 0) {
+ fprintf(stderr, "SKIP: no housekeeping CPU available\n");
+ free(ctx);
+ return SCX_TEST_SKIP;
+ }
+
+ ctx->test_cpu = cpu;
+ ctx->skel = nohz_tick__open();
+ if (!ctx->skel) {
+ free(ctx);
+ SCX_FAIL("Failed to open skeleton");
+ }
+
+ SCX_ENUM_INIT(ctx->skel);
+ ctx->skel->rodata->test_cpu = cpu;
+ ctx->skel->struct_ops.nohz_tick_ops->flags |= SCX_OPS_SWITCH_PARTIAL |
+ SCX_OPS_ENQ_LAST;
+ if (nohz_tick__load(ctx->skel)) {
+ nohz_tick__destroy(ctx->skel);
+ free(ctx);
+ SCX_FAIL("Failed to load skeleton");
+ }
+
+ if (sched_setaffinity(0, sizeof(controller_mask), &controller_mask)) {
+ nohz_tick__destroy(ctx->skel);
+ free(ctx);
+ SCX_FAIL("Failed to move controller off CPU %d (%d)", cpu, errno);
+ }
+
+ *ctx_ptr = ctx;
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx_ptr)
+{
+ struct nohz_tick_ctx *ctx = ctx_ptr;
+ struct nohz_tick *skel = ctx->skel;
+ struct bpf_link *link = NULL;
+ enum scx_test_status status = SCX_TEST_FAIL;
+ pid_t finite_worker = -1;
+ pid_t inf_worker = -1;
+ u64 finite_running;
+ u64 finite_ticks;
+ int ret;
+
+ link = bpf_map__attach_struct_ops(skel->maps.nohz_tick_ops);
+ if (!link) {
+ SCX_ERR("Failed to attach scheduler");
+ goto out;
+ }
+
+ /*
+ * Establish SCX_RQ_CAN_STOP_TICK with an infinite-slice task.
+ */
+ inf_worker = start_worker(ctx->test_cpu);
+ if (inf_worker < 0) {
+ SCX_ERR("Failed to start infinite-slice worker (%d)", errno);
+ goto out;
+ }
+ if (!wait_for_counter(&skel->bss->nr_inf_running, 1,
+ PHASE_TIMEOUT_MS)) {
+ SCX_ERR("Infinite-slice worker was not scheduled");
+ goto out;
+ }
+
+ /* Block without exiting so the rq retains the infinite-slice state. */
+ ret = pause_worker(inf_worker);
+ if (ret) {
+ SCX_ERR("Failed to stop infinite-slice worker (%d)", ret);
+ goto out;
+ }
+
+ /* Let the target enter idle with its tick stopped. */
+ usleep(100000);
+
+ /*
+ * The next EXT task receives a finite slice and must restart the tick.
+ */
+ __atomic_store_n(&skel->bss->finite_phase, true, __ATOMIC_RELEASE);
+ finite_worker = start_worker(ctx->test_cpu);
+ if (finite_worker < 0) {
+ SCX_ERR("Failed to start finite-slice worker (%d)", errno);
+ goto out;
+ }
+ if (!wait_for_counter(&skel->bss->nr_finite_running, 1,
+ PHASE_TIMEOUT_MS)) {
+ SCX_ERR("Finite-slice worker was not scheduled");
+ goto out;
+ }
+ if (!wait_for_counter(&skel->bss->nr_finite_ticks, MIN_FINITE_TICKS,
+ PHASE_TIMEOUT_MS)) {
+ SCX_ERR("Finite-slice worker received only %llu scheduler ticks",
+ (unsigned long long)skel->bss->nr_finite_ticks);
+ goto out;
+ }
+ stop_worker(finite_worker);
+ finite_worker = -1;
+
+ /*
+ * Leave the CPU idle after a finite-slice task. The next finite-slice
+ * task must restart the tick even though the slice type is unchanged.
+ */
+ usleep(100000);
+ finite_running = __atomic_load_n(&skel->bss->nr_finite_running,
+ __ATOMIC_RELAXED);
+ finite_ticks = __atomic_load_n(&skel->bss->nr_finite_ticks,
+ __ATOMIC_RELAXED);
+
+ finite_worker = start_worker(ctx->test_cpu);
+ if (finite_worker < 0) {
+ SCX_ERR("Failed to start second finite-slice worker (%d)", errno);
+ goto out;
+ }
+ if (!wait_for_counter(&skel->bss->nr_finite_running,
+ finite_running + 1, PHASE_TIMEOUT_MS)) {
+ SCX_ERR("Second finite-slice worker was not scheduled");
+ goto out;
+ }
+ if (!wait_for_counter(&skel->bss->nr_finite_ticks,
+ finite_ticks + MIN_FINITE_TICKS,
+ PHASE_TIMEOUT_MS)) {
+ SCX_ERR("Second finite-slice worker received only %llu scheduler ticks",
+ (unsigned long long)(skel->bss->nr_finite_ticks -
+ finite_ticks));
+ goto out;
+ }
+
+ if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_NONE)) {
+ SCX_ERR("Scheduler exited unexpectedly (kind=%llu code=%lld)",
+ (unsigned long long)skel->data->uei.kind,
+ (long long)skel->data->uei.exit_code);
+ goto out;
+ }
+
+ fprintf(stderr, "CPU %d received %llu finite-slice ticks\n",
+ ctx->test_cpu,
+ (unsigned long long)skel->bss->nr_finite_ticks);
+ status = SCX_TEST_PASS;
+out:
+ stop_worker(finite_worker);
+ stop_worker(inf_worker);
+ if (link)
+ bpf_link__destroy(link);
+ return status;
+}
+
+static void cleanup(void *ctx_ptr)
+{
+ struct nohz_tick_ctx *ctx = ctx_ptr;
+
+ sched_setaffinity(0, sizeof(ctx->original_mask), &ctx->original_mask);
+ nohz_tick__destroy(ctx->skel);
+ free(ctx);
+}
+
+struct scx_test nohz_tick = {
+ .name = "nohz_tick",
+ .description = "Verify finite EXT slices restart the NOHZ_FULL tick",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&nohz_tick)
diff --git a/tools/testing/selftests/sched_ext/non_scx_kfunc_deny.bpf.c b/tools/testing/selftests/sched_ext/non_scx_kfunc_deny.bpf.c
new file mode 100644
index 000000000000..0d6fcc8e5eb6
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/non_scx_kfunc_deny.bpf.c
@@ -0,0 +1,39 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Verify that context-sensitive SCX kfuncs (even "unlocked" ones) are
+ * restricted to only SCX struct_ops programs. Non-SCX struct_ops programs,
+ * such as TCP congestion control programs, should be rejected by the BPF
+ * verifier when attempting to call these kfuncs.
+ *
+ * Copyright (C) 2026 Ching-Chun (Jim) Huang <jserv@ccns.ncku.edu.tw>
+ * Copyright (C) 2026 Cheng-Yang Chou <yphbchou0911@gmail.com>
+ */
+
+#include <scx/common.bpf.h>
+
+SEC("struct_ops/ssthresh")
+__u32 BPF_PROG(tcp_ca_ssthresh, struct sock *sk)
+{
+ /*
+ * This call should be rejected by the verifier because this is a
+ * TCP congestion control program (non-SCX struct_ops).
+ */
+ scx_bpf_kick_cpu(0, 0);
+ return 2;
+}
+
+SEC("struct_ops/cong_avoid")
+void BPF_PROG(tcp_ca_cong_avoid, struct sock *sk, __u32 ack, __u32 acked) {}
+
+SEC("struct_ops/undo_cwnd")
+__u32 BPF_PROG(tcp_ca_undo_cwnd, struct sock *sk) { return 2; }
+
+SEC(".struct_ops")
+struct tcp_congestion_ops tcp_non_scx_ca = {
+ .ssthresh = (void *)tcp_ca_ssthresh,
+ .cong_avoid = (void *)tcp_ca_cong_avoid,
+ .undo_cwnd = (void *)tcp_ca_undo_cwnd,
+ .name = "tcp_kfunc_deny",
+};
+
+char _license[] SEC("license") = "GPL";
diff --git a/tools/testing/selftests/sched_ext/non_scx_kfunc_deny.c b/tools/testing/selftests/sched_ext/non_scx_kfunc_deny.c
new file mode 100644
index 000000000000..1c031575fb87
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/non_scx_kfunc_deny.c
@@ -0,0 +1,47 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Verify that context-sensitive SCX kfuncs (even "unlocked" ones) are
+ * restricted to only SCX struct_ops programs. Non-SCX struct_ops programs,
+ * such as TCP congestion control programs, should be rejected by the BPF
+ * verifier when attempting to call these kfuncs.
+ *
+ * Copyright (C) 2026 Ching-Chun (Jim) Huang <jserv@ccns.ncku.edu.tw>
+ * Copyright (C) 2026 Cheng-Yang Chou <yphbchou0911@gmail.com>
+ */
+
+#include <bpf/bpf.h>
+#include <scx/common.h>
+#include <unistd.h>
+#include <errno.h>
+#include <stdio.h>
+#include "non_scx_kfunc_deny.bpf.skel.h"
+#include "scx_test.h"
+
+static enum scx_test_status run(void *ctx)
+{
+ struct non_scx_kfunc_deny *skel;
+ int err;
+
+ skel = non_scx_kfunc_deny__open();
+ if (!skel) {
+ SCX_ERR("Failed to open skel");
+ return SCX_TEST_FAIL;
+ }
+
+ err = non_scx_kfunc_deny__load(skel);
+ non_scx_kfunc_deny__destroy(skel);
+
+ if (err == 0) {
+ SCX_ERR("non-SCX BPF program loaded when it should have been rejected");
+ return SCX_TEST_FAIL;
+ }
+
+ return SCX_TEST_PASS;
+}
+
+struct scx_test non_scx_kfunc_deny = {
+ .name = "non_scx_kfunc_deny",
+ .description = "Verify that non-SCX struct_ops programs cannot call SCX kfuncs",
+ .run = run,
+};
+REGISTER_SCX_TEST(&non_scx_kfunc_deny)
diff --git a/tools/testing/selftests/sched_ext/numa.bpf.c b/tools/testing/selftests/sched_ext/numa.bpf.c
index a79d86ed54a1..679b51d38089 100644
--- a/tools/testing/selftests/sched_ext/numa.bpf.c
+++ b/tools/testing/selftests/sched_ext/numa.bpf.c
@@ -19,24 +19,42 @@ UEI_DEFINE(uei);
const volatile unsigned int __COMPAT_SCX_PICK_IDLE_IN_NODE;
-static bool is_cpu_idle(s32 cpu, int node)
+static void validate_local_idle_state(void)
{
const struct cpumask *idle_cpumask;
- bool idle;
+ struct task_struct *curr;
+ s32 cpu = bpf_get_smp_processor_id();
+ int node = __COMPAT_scx_bpf_cpu_node(cpu);
+ bool cpu_is_idle, curr_is_idle;
+
+ bpf_rcu_read_lock();
+ curr = scx_bpf_cpu_curr(cpu);
+ curr_is_idle = curr && (curr->flags & PF_IDLE);
+ bpf_rcu_read_unlock();
idle_cpumask = __COMPAT_scx_bpf_get_idle_cpumask_node(node);
- idle = bpf_cpumask_test_cpu(cpu, idle_cpumask);
+ cpu_is_idle = bpf_cpumask_test_cpu(cpu, idle_cpumask);
scx_bpf_put_cpumask(idle_cpumask);
- return idle;
+ /*
+ * Unlike a remote picked CPU, the local CPU cannot go through an
+ * idle re-pick while this callback is running. If it is running a
+ * non-idle scheduling context, it must not be advertised as idle
+ * in its node's idle cpumask.
+ */
+ if (!curr_is_idle && cpu_is_idle)
+ scx_bpf_error("running CPU %d should be marked as busy", cpu);
}
s32 BPF_STRUCT_OPS(numa_select_cpu,
struct task_struct *p, s32 prev_cpu, u64 wake_flags)
{
- int node = __COMPAT_scx_bpf_cpu_node(scx_bpf_task_cpu(p));
+ s32 task_cpu = scx_bpf_task_cpu(p);
+ int node = __COMPAT_scx_bpf_cpu_node(task_cpu);
s32 cpu;
+ validate_local_idle_state();
+
/*
* We could just use __COMPAT_scx_bpf_pick_any_cpu_node() here,
* since it already tries to pick an idle CPU within the node
@@ -48,8 +66,15 @@ s32 BPF_STRUCT_OPS(numa_select_cpu,
cpu = __COMPAT_scx_bpf_pick_any_cpu_node(p->cpus_ptr, node,
__COMPAT_SCX_PICK_IDLE_IN_NODE);
- if (is_cpu_idle(cpu, node))
- scx_bpf_error("CPU %d should be marked as busy", cpu);
+ /*
+ * @task_cpu may be outside of p->cpus_ptr if @p's affinity
+ * changed while it was sleeping. This means it's possible for
+ * p->cpus_ptr to not include any CPUs from @node.
+ * If we failed to find a cpu in @node, check if @task_cpu
+ * is outside of p->cpus_ptr and just return @prev_cpu if it is.
+ */
+ if (cpu < 0 && !bpf_cpumask_test_cpu(task_cpu, p->cpus_ptr))
+ return prev_cpu;
if (__COMPAT_scx_bpf_cpu_node(cpu) != node)
scx_bpf_error("CPU %d should be in node %d", cpu, node);
@@ -68,7 +93,7 @@ void BPF_STRUCT_OPS(numa_dispatch, s32 cpu, struct task_struct *prev)
{
int node = __COMPAT_scx_bpf_cpu_node(cpu);
- scx_bpf_dsq_move_to_local(node);
+ scx_bpf_dsq_move_to_local(node, 0);
}
s32 BPF_STRUCT_OPS_SLEEPABLE(numa_init)
diff --git a/tools/testing/selftests/sched_ext/peek_dsq.bpf.c b/tools/testing/selftests/sched_ext/peek_dsq.bpf.c
index a3faf5bb49d6..9e802b52b29e 100644
--- a/tools/testing/selftests/sched_ext/peek_dsq.bpf.c
+++ b/tools/testing/selftests/sched_ext/peek_dsq.bpf.c
@@ -58,14 +58,14 @@ static void record_peek_result(long pid)
{
u32 slot_key;
long *slot_pid_ptr;
- int ix;
+ u32 ix;
if (pid <= 0)
return;
/* Find an empty slot or one with the same PID */
bpf_for(ix, 0, 10) {
- slot_key = (pid + ix) % MAX_SAMPLES;
+ slot_key = ((u64)pid + ix) % MAX_SAMPLES;
slot_pid_ptr = bpf_map_lookup_elem(&peek_results, &slot_key);
if (!slot_pid_ptr)
continue;
@@ -95,7 +95,7 @@ static int scan_dsq_pool(void)
record_peek_result(task->pid);
/* Try to move this task to local */
- if (!moved && scx_bpf_dsq_move_to_local(dsq_id) == 0) {
+ if (!moved && scx_bpf_dsq_move_to_local(dsq_id, 0)) {
moved = 1;
break;
}
@@ -156,19 +156,19 @@ void BPF_STRUCT_OPS(peek_dsq_dispatch, s32 cpu, struct task_struct *prev)
dsq_peek_result2_pid = peek_result ? peek_result->pid : -1;
/* Now consume the task since we've peeked at it */
- scx_bpf_dsq_move_to_local(test_dsq_id);
+ scx_bpf_dsq_move_to_local(test_dsq_id, 0);
/* Mark phase 1 as complete */
phase1_complete = 1;
bpf_printk("Phase 1 complete, starting phase 2 stress testing");
} else if (!phase1_complete) {
/* Still in phase 1, use real DSQ */
- scx_bpf_dsq_move_to_local(real_dsq_id);
+ scx_bpf_dsq_move_to_local(real_dsq_id, 0);
} else {
/* Phase 2: Scan all DSQs in the pool and try to move a task */
if (!scan_dsq_pool()) {
/* No tasks found in DSQ pool, fall back to real DSQ */
- scx_bpf_dsq_move_to_local(real_dsq_id);
+ scx_bpf_dsq_move_to_local(real_dsq_id, 0);
}
}
}
@@ -197,7 +197,7 @@ s32 BPF_STRUCT_OPS_SLEEPABLE(peek_dsq_init)
}
err = scx_bpf_create_dsq(real_dsq_id, -1);
if (err) {
- scx_bpf_error("Failed to create DSQ %d: %d", test_dsq_id, err);
+ scx_bpf_error("Failed to create DSQ %d: %d", real_dsq_id, err);
return err;
}
diff --git a/tools/testing/selftests/sched_ext/prog_run.c b/tools/testing/selftests/sched_ext/prog_run.c
index 05974820ca69..1129ec2aaddc 100644
--- a/tools/testing/selftests/sched_ext/prog_run.c
+++ b/tools/testing/selftests/sched_ext/prog_run.c
@@ -28,7 +28,8 @@ static enum scx_test_status setup(void **ctx)
static enum scx_test_status run(void *ctx)
{
struct prog_run *skel = ctx;
- struct bpf_link *link;
+ struct bpf_link *link = NULL;
+ enum scx_test_status status = SCX_TEST_PASS;
int prog_fd, err = 0;
prog_fd = bpf_program__fd(skel->progs.prog_run_syscall);
@@ -42,23 +43,40 @@ static enum scx_test_status run(void *ctx)
link = bpf_map__attach_struct_ops(skel->maps.prog_run_ops);
if (!link) {
SCX_ERR("Failed to attach scheduler");
- close(prog_fd);
- return SCX_TEST_FAIL;
+ status = SCX_TEST_FAIL;
+ goto out;
}
err = bpf_prog_test_run_opts(prog_fd, &topts);
- SCX_EQ(err, 0);
+ if (err) {
+ SCX_ERR("BPF_PROG_RUN failed (%d)", err);
+ status = SCX_TEST_FAIL;
+ goto out;
+ }
/* Assumes uei.kind is written last */
while (skel->data->uei.kind == EXIT_KIND(SCX_EXIT_NONE))
sched_yield();
- SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG_BPF));
- SCX_EQ(skel->data->uei.exit_code, 0xdeadbeef);
+ if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_UNREG_BPF)) {
+ SCX_ERR("Unexpected exit kind: %llu",
+ (unsigned long long)skel->data->uei.kind);
+ status = SCX_TEST_FAIL;
+ goto out;
+ }
+ if (skel->data->uei.exit_code != 0xdeadbeef) {
+ SCX_ERR("Unexpected exit code: %lld",
+ (long long)skel->data->uei.exit_code);
+ status = SCX_TEST_FAIL;
+ goto out;
+ }
+
+out:
close(prog_fd);
- bpf_link__destroy(link);
+ if (link)
+ bpf_link__destroy(link);
- return SCX_TEST_PASS;
+ return status;
}
static void cleanup(void *ctx)
diff --git a/tools/testing/selftests/sched_ext/reload_loop.c b/tools/testing/selftests/sched_ext/reload_loop.c
index 308211d80436..49297b83d748 100644
--- a/tools/testing/selftests/sched_ext/reload_loop.c
+++ b/tools/testing/selftests/sched_ext/reload_loop.c
@@ -23,6 +23,9 @@ static enum scx_test_status setup(void **ctx)
SCX_ENUM_INIT(skel);
SCX_FAIL_IF(maximal__load(skel), "Failed to load skel");
+ bpf_map__set_autoattach(skel->maps.maximal_ops, false);
+ SCX_FAIL_IF(maximal__attach(skel), "Failed to attach skel");
+
return SCX_TEST_PASS;
}
diff --git a/tools/testing/selftests/sched_ext/rt_stall.bpf.c b/tools/testing/selftests/sched_ext/rt_stall.bpf.c
new file mode 100644
index 000000000000..80086779dd1e
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/rt_stall.bpf.c
@@ -0,0 +1,23 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A scheduler that verified if RT tasks can stall SCHED_EXT tasks.
+ *
+ * Copyright (c) 2025 NVIDIA Corporation.
+ */
+
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+UEI_DEFINE(uei);
+
+void BPF_STRUCT_OPS(rt_stall_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops rt_stall_ops = {
+ .exit = (void *)rt_stall_exit,
+ .name = "rt_stall",
+};
diff --git a/tools/testing/selftests/sched_ext/rt_stall.c b/tools/testing/selftests/sched_ext/rt_stall.c
new file mode 100644
index 000000000000..a5041fc2e44f
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/rt_stall.c
@@ -0,0 +1,293 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2025 NVIDIA Corporation.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <unistd.h>
+#include <sched.h>
+#include <sys/prctl.h>
+#include <sys/types.h>
+#include <sys/wait.h>
+#include <time.h>
+#include <linux/sched.h>
+#include <signal.h>
+#include <bpf/bpf.h>
+#include <scx/common.h>
+#include "rt_stall.bpf.skel.h"
+#include "scx_test.h"
+#include "../kselftest.h"
+
+#define CORE_ID 0 /* CPU to pin tasks to */
+#define RUN_TIME 5 /* How long to run the test in seconds */
+
+/* Signal the parent that setup is complete by writing to a pipe */
+static void signal_ready(int fd)
+{
+ char c = 1;
+
+ if (write(fd, &c, 1) != 1) {
+ perror("write to ready pipe");
+ exit(EXIT_FAILURE);
+ }
+ close(fd);
+}
+
+/* Wait for a child to signal readiness via a pipe */
+static void wait_ready(int fd)
+{
+ char c;
+
+ if (read(fd, &c, 1) != 1) {
+ perror("read from ready pipe");
+ exit(EXIT_FAILURE);
+ }
+ close(fd);
+}
+
+/* Simple busy-wait function for test tasks */
+static void process_func(void)
+{
+ while (1) {
+ /* Busy wait */
+ for (volatile unsigned long i = 0; i < 10000000UL; i++)
+ ;
+ }
+}
+
+/* Set CPU affinity to a specific core */
+static void set_affinity(int cpu)
+{
+ cpu_set_t mask;
+
+ CPU_ZERO(&mask);
+ CPU_SET(cpu, &mask);
+ if (sched_setaffinity(0, sizeof(mask), &mask) != 0) {
+ perror("sched_setaffinity");
+ exit(EXIT_FAILURE);
+ }
+}
+
+/* Set task scheduling policy and priority */
+static void set_sched(int policy, int priority)
+{
+ struct sched_param param;
+
+ param.sched_priority = priority;
+ if (sched_setscheduler(0, policy, &param) != 0) {
+ perror("sched_setscheduler");
+ exit(EXIT_FAILURE);
+ }
+}
+
+/* Get process runtime from /proc/<pid>/stat */
+static float get_process_runtime(int pid)
+{
+ char path[256];
+ FILE *file;
+ long utime, stime;
+ int fields;
+
+ snprintf(path, sizeof(path), "/proc/%d/stat", pid);
+ file = fopen(path, "r");
+ if (file == NULL) {
+ perror("Failed to open stat file");
+ return -1;
+ }
+
+ /* Skip the first 13 fields and read the 14th and 15th */
+ fields = fscanf(file,
+ "%*d %*s %*c %*d %*d %*d %*d %*d %*u %*u %*u %*u %*u %lu %lu",
+ &utime, &stime);
+ fclose(file);
+
+ if (fields != 2) {
+ fprintf(stderr, "Failed to read stat file\n");
+ return -1;
+ }
+
+ /* Calculate the total time spent in the process */
+ long total_time = utime + stime;
+ long ticks_per_second = sysconf(_SC_CLK_TCK);
+ float runtime_seconds = total_time * 1.0 / ticks_per_second;
+
+ return runtime_seconds;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+ struct rt_stall *skel;
+
+ if (!__COMPAT_struct_has_field("rq", "ext_server")) {
+ fprintf(stderr, "SKIP: ext DL server not supported\n");
+ return SCX_TEST_SKIP;
+ }
+
+ skel = rt_stall__open();
+ SCX_FAIL_IF(!skel, "Failed to open");
+ SCX_ENUM_INIT(skel);
+ SCX_FAIL_IF(rt_stall__load(skel), "Failed to load skel");
+
+ *ctx = skel;
+
+ return SCX_TEST_PASS;
+}
+
+static bool sched_stress_test(bool is_ext)
+{
+ /*
+ * We're expecting the EXT task to get around 5% of CPU time when
+ * competing with the RT task (small 1% fluctuations are expected).
+ *
+ * However, the EXT task should get at least 4% of the CPU to prove
+ * that the EXT deadline server is working correctly. A percentage
+ * less than 4% indicates a bug where RT tasks can potentially
+ * stall SCHED_EXT tasks, causing the test to fail.
+ */
+ const float expected_min_ratio = 0.04; /* 4% */
+ const char *class_str = is_ext ? "EXT" : "FAIR";
+
+ float ext_runtime, rt_runtime, actual_ratio;
+ int ext_pid, rt_pid;
+ int ext_ready[2], rt_ready[2];
+
+ ksft_print_header();
+ ksft_set_plan(1);
+
+ if (pipe(ext_ready) || pipe(rt_ready)) {
+ perror("pipe");
+ ksft_exit_fail();
+ }
+
+ /* Create and set up a EXT task */
+ ext_pid = fork();
+ if (ext_pid == 0) {
+ close(ext_ready[0]);
+ close(rt_ready[0]);
+ close(rt_ready[1]);
+ set_affinity(CORE_ID);
+ signal_ready(ext_ready[1]);
+ process_func();
+ exit(0);
+ } else if (ext_pid < 0) {
+ perror("fork task");
+ ksft_exit_fail();
+ }
+
+ /* Create an RT task */
+ rt_pid = fork();
+ if (rt_pid == 0) {
+ close(ext_ready[0]);
+ close(ext_ready[1]);
+ close(rt_ready[0]);
+ set_affinity(CORE_ID);
+ set_sched(SCHED_FIFO, 50);
+ signal_ready(rt_ready[1]);
+ process_func();
+ exit(0);
+ } else if (rt_pid < 0) {
+ perror("fork for RT task");
+ ksft_exit_fail();
+ }
+
+ /*
+ * Wait for both children to complete their setup (affinity and
+ * scheduling policy) before starting the measurement window.
+ * This prevents flaky failures caused by the RT child's setup
+ * time eating into the measurement period.
+ */
+ close(ext_ready[1]);
+ close(rt_ready[1]);
+ wait_ready(ext_ready[0]);
+ wait_ready(rt_ready[0]);
+
+ /* Let the processes run for the specified time */
+ sleep(RUN_TIME);
+
+ /* Get runtime for the EXT task */
+ ext_runtime = get_process_runtime(ext_pid);
+ if (ext_runtime == -1)
+ ksft_exit_fail_msg("Error getting runtime for %s task (PID %d)\n",
+ class_str, ext_pid);
+ ksft_print_msg("Runtime of %s task (PID %d) is %f seconds\n",
+ class_str, ext_pid, ext_runtime);
+
+ /* Get runtime for the RT task */
+ rt_runtime = get_process_runtime(rt_pid);
+ if (rt_runtime == -1)
+ ksft_exit_fail_msg("Error getting runtime for RT task (PID %d)\n", rt_pid);
+ ksft_print_msg("Runtime of RT task (PID %d) is %f seconds\n", rt_pid, rt_runtime);
+
+ /* Kill the processes */
+ kill(ext_pid, SIGKILL);
+ kill(rt_pid, SIGKILL);
+ waitpid(ext_pid, NULL, 0);
+ waitpid(rt_pid, NULL, 0);
+
+ /* Verify that the scx task got enough runtime */
+ actual_ratio = ext_runtime / (ext_runtime + rt_runtime);
+ ksft_print_msg("%s task got %.2f%% of total runtime\n",
+ class_str, actual_ratio * 100);
+
+ if (actual_ratio >= expected_min_ratio) {
+ ksft_test_result_pass("PASS: %s task got more than %.2f%% of runtime\n",
+ class_str, expected_min_ratio * 100);
+ return true;
+ }
+ ksft_test_result_fail("FAIL: %s task got less than %.2f%% of runtime\n",
+ class_str, expected_min_ratio * 100);
+ return false;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct rt_stall *skel = ctx;
+ struct bpf_link *link = NULL;
+ bool res;
+ int i;
+
+ /*
+ * Test if the dl_server is working both with and without the
+ * sched_ext scheduler attached.
+ *
+ * This ensures all the scenarios are covered:
+ * - fair_server stop -> ext_server start
+ * - ext_server stop -> fair_server stop
+ */
+ for (i = 0; i < 4; i++) {
+ bool is_ext = i % 2;
+
+ if (is_ext) {
+ memset(&skel->data->uei, 0, sizeof(skel->data->uei));
+ link = bpf_map__attach_struct_ops(skel->maps.rt_stall_ops);
+ SCX_FAIL_IF(!link, "Failed to attach scheduler");
+ }
+ res = sched_stress_test(is_ext);
+ if (is_ext) {
+ SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_NONE));
+ bpf_link__destroy(link);
+ }
+
+ if (!res)
+ ksft_exit_fail();
+ }
+
+ return SCX_TEST_PASS;
+}
+
+static void cleanup(void *ctx)
+{
+ struct rt_stall *skel = ctx;
+
+ rt_stall__destroy(skel);
+}
+
+struct scx_test rt_stall = {
+ .name = "rt_stall",
+ .description = "Verify that RT tasks cannot stall SCHED_EXT tasks",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&rt_stall)
diff --git a/tools/testing/selftests/sched_ext/runner.c b/tools/testing/selftests/sched_ext/runner.c
index aa2d7d32dda9..c264807caa91 100644
--- a/tools/testing/selftests/sched_ext/runner.c
+++ b/tools/testing/selftests/sched_ext/runner.c
@@ -18,7 +18,7 @@ const char help_fmt[] =
"It's required for the testcases to be serial, as only a single host-wide sched_ext\n"
"scheduler may be loaded at any given time."
"\n"
-"Usage: %s [-t TEST] [-h]\n"
+"Usage: %s [-t TEST] [-s] [-l] [-q]\n"
"\n"
" -t TEST Only run tests whose name includes this string\n"
" -s Include print output for skipped tests\n"
@@ -46,6 +46,14 @@ static void print_test_preamble(const struct scx_test *test, bool quiet)
if (!quiet)
printf("DESCRIPTION: %s\n", test->description);
printf("OUTPUT:\n");
+
+ /*
+ * The tests may fork with the preamble buffered
+ * in the children's stdout. Flush before the test
+ * to avoid printing the message multiple times.
+ */
+ fflush(stdout);
+ fflush(stderr);
}
static const char *status_to_result(enum scx_test_status status)
@@ -125,6 +133,8 @@ static bool test_valid(const struct scx_test *test)
int main(int argc, char **argv)
{
const char *filter = NULL;
+ const char *failed_tests[MAX_SCX_TESTS];
+ const char *skipped_tests[MAX_SCX_TESTS];
unsigned testnum = 0, i;
unsigned passed = 0, skipped = 0, failed = 0;
int opt;
@@ -154,10 +164,33 @@ int main(int argc, char **argv)
}
}
+ if (optind < argc) {
+ fprintf(stderr, "Unexpected argument '%s'. Use -t to filter tests.\n",
+ argv[optind]);
+ return 1;
+ }
+
+ if (filter) {
+ for (i = 0; i < __scx_num_tests; i++) {
+ if (!should_skip_test(&__scx_tests[i], filter))
+ break;
+ }
+ if (i == __scx_num_tests) {
+ fprintf(stderr, "No tests matched filter '%s'\n", filter);
+ fprintf(stderr, "Available tests (use -l to list):\n");
+ for (i = 0; i < __scx_num_tests; i++)
+ fprintf(stderr, " %s\n", __scx_tests[i].name);
+ return 1;
+ }
+ }
+
for (i = 0; i < __scx_num_tests; i++) {
enum scx_test_status status;
struct scx_test *test = &__scx_tests[i];
+ if (exit_req)
+ break;
+
if (list) {
printf("%s\n", test->name);
if (i == (__scx_num_tests - 1))
@@ -187,10 +220,10 @@ int main(int argc, char **argv)
passed++;
break;
case SCX_TEST_SKIP:
- skipped++;
+ skipped_tests[skipped++] = test->name;
break;
case SCX_TEST_FAIL:
- failed++;
+ failed_tests[failed++] = test->name;
break;
}
}
@@ -199,8 +232,18 @@ int main(int argc, char **argv)
printf("PASSED: %u\n", passed);
printf("SKIPPED: %u\n", skipped);
printf("FAILED: %u\n", failed);
+ if (skipped > 0) {
+ printf("\nSkipped tests:\n");
+ for (i = 0; i < skipped; i++)
+ printf(" - %s\n", skipped_tests[i]);
+ }
+ if (failed > 0) {
+ printf("\nFailed tests:\n");
+ for (i = 0; i < failed; i++)
+ printf(" - %s\n", failed_tests[i]);
+ }
- return 0;
+ return failed > 0 ? 1 : 0;
}
void scx_test_register(struct scx_test *test)
diff --git a/tools/testing/selftests/sched_ext/select_cpu_dfl.c b/tools/testing/selftests/sched_ext/select_cpu_dfl.c
index 5b6e045e1109..7e342c0cec65 100644
--- a/tools/testing/selftests/sched_ext/select_cpu_dfl.c
+++ b/tools/testing/selftests/sched_ext/select_cpu_dfl.c
@@ -6,6 +6,7 @@
*/
#include <bpf/bpf.h>
#include <scx/common.h>
+#include <stdlib.h>
#include <sys/wait.h>
#include <unistd.h>
#include "select_cpu_dfl.bpf.skel.h"
@@ -13,29 +14,44 @@
#define NUM_CHILDREN 1028
+struct select_cpu_dfl_ctx {
+ struct select_cpu_dfl *skel;
+ struct bpf_link *link;
+};
+
static enum scx_test_status setup(void **ctx)
{
- struct select_cpu_dfl *skel;
+ struct select_cpu_dfl_ctx *tctx;
+
+ tctx = malloc(sizeof(*tctx));
+ SCX_FAIL_IF(!tctx, "Failed to allocate test context");
+ tctx->link = NULL;
- skel = select_cpu_dfl__open();
- SCX_FAIL_IF(!skel, "Failed to open");
- SCX_ENUM_INIT(skel);
- SCX_FAIL_IF(select_cpu_dfl__load(skel), "Failed to load skel");
+ tctx->skel = select_cpu_dfl__open();
+ if (!tctx->skel) {
+ free(tctx);
+ SCX_FAIL("Failed to open");
+ }
+ SCX_ENUM_INIT(tctx->skel);
+ if (select_cpu_dfl__load(tctx->skel)) {
+ select_cpu_dfl__destroy(tctx->skel);
+ free(tctx);
+ SCX_FAIL("Failed to load skel");
+ }
- *ctx = skel;
+ *ctx = tctx;
return SCX_TEST_PASS;
}
static enum scx_test_status run(void *ctx)
{
- struct select_cpu_dfl *skel = ctx;
- struct bpf_link *link;
+ struct select_cpu_dfl_ctx *tctx = ctx;
pid_t pids[NUM_CHILDREN];
- int i, status;
+ int i, status, nforked = 0;
- link = bpf_map__attach_struct_ops(skel->maps.select_cpu_dfl_ops);
- SCX_FAIL_IF(!link, "Failed to attach scheduler");
+ tctx->link = bpf_map__attach_struct_ops(tctx->skel->maps.select_cpu_dfl_ops);
+ SCX_FAIL_IF(!tctx->link, "Failed to attach scheduler");
for (i = 0; i < NUM_CHILDREN; i++) {
pids[i] = fork();
@@ -43,25 +59,31 @@ static enum scx_test_status run(void *ctx)
sleep(1);
exit(0);
}
+ if (pids[i] > 0)
+ nforked++;
}
for (i = 0; i < NUM_CHILDREN; i++) {
+ if (pids[i] <= 0)
+ continue;
SCX_EQ(waitpid(pids[i], &status, 0), pids[i]);
SCX_EQ(status, 0);
}
- SCX_ASSERT(!skel->bss->saw_local);
-
- bpf_link__destroy(link);
+ SCX_GT(nforked, 0);
+ SCX_ASSERT(!tctx->skel->bss->saw_local);
return SCX_TEST_PASS;
}
static void cleanup(void *ctx)
{
- struct select_cpu_dfl *skel = ctx;
+ struct select_cpu_dfl_ctx *tctx = ctx;
- select_cpu_dfl__destroy(skel);
+ if (tctx->link)
+ bpf_link__destroy(tctx->link);
+ select_cpu_dfl__destroy(tctx->skel);
+ free(tctx);
}
struct scx_test select_cpu_dfl = {
diff --git a/tools/testing/selftests/sched_ext/select_cpu_vtime.bpf.c b/tools/testing/selftests/sched_ext/select_cpu_vtime.bpf.c
index bfcb96cd4954..eec70d388cbf 100644
--- a/tools/testing/selftests/sched_ext/select_cpu_vtime.bpf.c
+++ b/tools/testing/selftests/sched_ext/select_cpu_vtime.bpf.c
@@ -53,7 +53,7 @@ ddsp:
void BPF_STRUCT_OPS(select_cpu_vtime_dispatch, s32 cpu, struct task_struct *p)
{
- if (scx_bpf_dsq_move_to_local(VTIME_DSQ))
+ if (scx_bpf_dsq_move_to_local(VTIME_DSQ, 0))
consumed = true;
}
@@ -66,12 +66,14 @@ void BPF_STRUCT_OPS(select_cpu_vtime_running, struct task_struct *p)
void BPF_STRUCT_OPS(select_cpu_vtime_stopping, struct task_struct *p,
bool runnable)
{
- p->scx.dsq_vtime += (SCX_SLICE_DFL - p->scx.slice) * 100 / p->scx.weight;
+ u64 delta = scale_by_task_weight_inverse(p, SCX_SLICE_DFL - p->scx.slice);
+
+ scx_bpf_task_set_dsq_vtime(p, p->scx.dsq_vtime + delta);
}
void BPF_STRUCT_OPS(select_cpu_vtime_enable, struct task_struct *p)
{
- p->scx.dsq_vtime = vtime_now;
+ scx_bpf_task_set_dsq_vtime(p, vtime_now);
}
s32 BPF_STRUCT_OPS_SLEEPABLE(select_cpu_vtime_init)
diff --git a/tools/testing/selftests/sched_ext/total_bw.c b/tools/testing/selftests/sched_ext/total_bw.c
new file mode 100644
index 000000000000..2af01cee90cc
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/total_bw.c
@@ -0,0 +1,480 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test to verify that total_bw value remains consistent across all CPUs
+ * in different BPF program states.
+ *
+ * Copyright (C) 2025 NVIDIA Corporation.
+ */
+#include <bpf/bpf.h>
+#include <errno.h>
+#include <pthread.h>
+#include <scx/common.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/wait.h>
+#include <unistd.h>
+#include "minimal.bpf.skel.h"
+#include "scx_test.h"
+
+#define MAX_CPUS 512
+#define STRESS_DURATION_SEC 5
+
+struct total_bw_ctx {
+ struct minimal *skel;
+ long baseline_bw[MAX_CPUS];
+ int nr_cpus;
+};
+
+static void *cpu_stress_thread(void *arg)
+{
+ volatile int i;
+ time_t end_time = time(NULL) + STRESS_DURATION_SEC;
+
+ while (time(NULL) < end_time)
+ for (i = 0; i < 1000000; i++)
+ ;
+
+ return NULL;
+}
+
+/*
+ * The first enqueue on a CPU causes the DL server to start, for that
+ * reason run stressor threads in the hopes it schedules on all CPUs.
+ */
+static int run_cpu_stress(int nr_cpus)
+{
+ pthread_t *threads;
+ int i, ret = 0;
+
+ threads = calloc(nr_cpus, sizeof(pthread_t));
+ if (!threads)
+ return -ENOMEM;
+
+ /* Create threads to run on each CPU */
+ for (i = 0; i < nr_cpus; i++) {
+ if (pthread_create(&threads[i], NULL, cpu_stress_thread, NULL)) {
+ ret = -errno;
+ fprintf(stderr, "Failed to create thread %d: %s\n", i, strerror(-ret));
+ break;
+ }
+ }
+
+ /* Wait for all threads to complete */
+ for (i = 0; i < nr_cpus; i++) {
+ if (threads[i])
+ pthread_join(threads[i], NULL);
+ }
+
+ free(threads);
+ return ret;
+}
+
+static int read_total_bw_values(long *bw_values, int max_cpus)
+{
+ FILE *fp;
+ char line[256];
+ int cpu_count = 0;
+
+ fp = fopen("/sys/kernel/debug/sched/debug", "r");
+ if (!fp) {
+ SCX_ERR("Failed to open debug file");
+ return -1;
+ }
+
+ while (fgets(line, sizeof(line), fp)) {
+ char *bw_str = strstr(line, "total_bw");
+
+ if (bw_str) {
+ bw_str = strchr(bw_str, ':');
+ if (bw_str) {
+ /* Only store up to max_cpus values */
+ if (cpu_count < max_cpus)
+ bw_values[cpu_count] = atol(bw_str + 1);
+ cpu_count++;
+ }
+ }
+ }
+
+ fclose(fp);
+ return cpu_count;
+}
+
+/*
+ * Read a per-CPU dl_server param (runtime or period) from debugfs.
+ * Returns the value in nanoseconds, or -1 on failure.
+ */
+static long read_server_param(const char *server, const char *param, int cpu)
+{
+ char path[128];
+ long value = -1;
+ FILE *fp;
+
+ snprintf(path, sizeof(path),
+ "/sys/kernel/debug/sched/%s_server/cpu%d/%s",
+ server, cpu, param);
+ fp = fopen(path, "r");
+ if (!fp)
+ return -1;
+ if (fscanf(fp, "%ld", &value) != 1)
+ value = -1;
+ fclose(fp);
+
+ return value;
+}
+
+/*
+ * Write a per-CPU dl_server param to debugfs. Returns 0 on success.
+ */
+static int write_server_param(const char *server, const char *param,
+ int cpu, long value)
+{
+ char path[128];
+ FILE *fp;
+ int ret = 0;
+
+ snprintf(path, sizeof(path),
+ "/sys/kernel/debug/sched/%s_server/cpu%d/%s",
+ server, cpu, param);
+ fp = fopen(path, "w");
+ if (!fp)
+ return -1;
+ if (fprintf(fp, "%ld", value) < 0)
+ ret = -1;
+ if (fclose(fp) != 0)
+ ret = -1;
+
+ return ret;
+}
+
+static int read_fair_runtime_all(int nr_cpus, long *runtimes)
+{
+ int i;
+
+ for (i = 0; i < nr_cpus; i++) {
+ runtimes[i] = read_server_param("fair", "runtime", i);
+ if (runtimes[i] <= 0)
+ return -1;
+ }
+
+ return 0;
+}
+
+static int write_fair_runtime_all(int nr_cpus, long value)
+{
+ int i;
+
+ for (i = 0; i < nr_cpus; i++) {
+ if (write_server_param("fair", "runtime", i, value) < 0) {
+ SCX_ERR("Failed to write fair_server runtime on CPU %d", i);
+ return -1;
+ }
+ }
+
+ return 0;
+}
+
+/*
+ * Restore per-CPU fair_server runtimes.
+ */
+static int restore_fair_runtime_all(int nr_cpus, const long *runtimes)
+{
+ int ret = 0;
+ int i;
+
+ for (i = 0; i < nr_cpus; i++) {
+ if (write_server_param("fair", "runtime", i, runtimes[i]) < 0) {
+ SCX_ERR("Failed to restore fair_server runtime on CPU %d", i);
+ ret = -1;
+ }
+ }
+
+ return ret;
+}
+
+static bool verify_total_bw_consistency(long *bw_values, int count)
+{
+ int i;
+ long first_value;
+
+ if (count <= 0)
+ return false;
+
+ first_value = bw_values[0];
+
+ for (i = 1; i < count; i++) {
+ if (bw_values[i] != first_value) {
+ SCX_ERR("Inconsistent total_bw: CPU0=%ld, CPU%d=%ld",
+ first_value, i, bw_values[i]);
+ return false;
+ }
+ }
+
+ return true;
+}
+
+static int fetch_verify_total_bw(long *bw_values, int nr_cpus)
+{
+ int attempts = 0;
+ int max_attempts = 10;
+ int count;
+
+ /*
+ * The first enqueue on a CPU causes the DL server to start, for that
+ * reason run stressor threads in the hopes it schedules on all CPUs.
+ */
+ if (run_cpu_stress(nr_cpus) < 0) {
+ SCX_ERR("Failed to run CPU stress");
+ return -1;
+ }
+
+ /* Try multiple times to get stable values */
+ while (attempts < max_attempts) {
+ count = read_total_bw_values(bw_values, nr_cpus);
+ fprintf(stderr, "Read %d total_bw values (testing %d CPUs)\n", count, nr_cpus);
+ /* If system has more CPUs than we're testing, that's OK */
+ if (count < nr_cpus) {
+ SCX_ERR("Expected at least %d CPUs, got %d", nr_cpus, count);
+ attempts++;
+ sleep(1);
+ continue;
+ }
+
+ /* Only verify the CPUs we're testing */
+ if (verify_total_bw_consistency(bw_values, nr_cpus)) {
+ fprintf(stderr, "Values are consistent: %ld\n", bw_values[0]);
+ return 0;
+ }
+
+ attempts++;
+ sleep(1);
+ }
+
+ return -1;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+ struct total_bw_ctx *test_ctx;
+
+ if (access("/sys/kernel/debug/sched/debug", R_OK) != 0) {
+ fprintf(stderr, "Skipping test: debugfs sched/debug not accessible\n");
+ return SCX_TEST_SKIP;
+ }
+
+ test_ctx = calloc(1, sizeof(*test_ctx));
+ if (!test_ctx)
+ return SCX_TEST_FAIL;
+
+ test_ctx->nr_cpus = sysconf(_SC_NPROCESSORS_ONLN);
+ if (test_ctx->nr_cpus <= 0) {
+ free(test_ctx);
+ return SCX_TEST_FAIL;
+ }
+
+ /* If system has more CPUs than MAX_CPUS, just test the first MAX_CPUS */
+ if (test_ctx->nr_cpus > MAX_CPUS)
+ test_ctx->nr_cpus = MAX_CPUS;
+
+ /* Test scenario 1: BPF program not loaded */
+ /* Read and verify baseline total_bw before loading BPF program */
+ fprintf(stderr, "BPF prog initially not loaded, reading total_bw values\n");
+ if (fetch_verify_total_bw(test_ctx->baseline_bw, test_ctx->nr_cpus) < 0) {
+ SCX_ERR("Failed to get stable baseline values");
+ free(test_ctx);
+ return SCX_TEST_FAIL;
+ }
+
+ /* Load the BPF skeleton */
+ test_ctx->skel = minimal__open();
+ if (!test_ctx->skel) {
+ free(test_ctx);
+ return SCX_TEST_FAIL;
+ }
+
+ SCX_ENUM_INIT(test_ctx->skel);
+ if (minimal__load(test_ctx->skel)) {
+ minimal__destroy(test_ctx->skel);
+ free(test_ctx);
+ return SCX_TEST_FAIL;
+ }
+
+ *ctx = test_ctx;
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct total_bw_ctx *test_ctx = ctx;
+ struct bpf_link *link;
+ long loaded_bw[MAX_CPUS];
+ long unloaded_bw[MAX_CPUS];
+ long doubled_bw[MAX_CPUS];
+ long original_runtime[MAX_CPUS], doubled_runtime;
+ enum scx_test_status ret;
+ int i;
+
+ /* Test scenario 2: BPF program loaded */
+ link = bpf_map__attach_struct_ops(test_ctx->skel->maps.minimal_ops);
+ if (!link) {
+ SCX_ERR("Failed to attach scheduler");
+ return SCX_TEST_FAIL;
+ }
+
+ fprintf(stderr, "BPF program loaded, reading total_bw values\n");
+ if (fetch_verify_total_bw(loaded_bw, test_ctx->nr_cpus) < 0) {
+ SCX_ERR("Failed to get stable values with BPF loaded");
+ bpf_link__destroy(link);
+ return SCX_TEST_FAIL;
+ }
+ bpf_link__destroy(link);
+
+ /* Test scenario 3: BPF program unloaded */
+ fprintf(stderr, "BPF program unloaded, reading total_bw values\n");
+ if (fetch_verify_total_bw(unloaded_bw, test_ctx->nr_cpus) < 0) {
+ SCX_ERR("Failed to get stable values after BPF unload");
+ return SCX_TEST_FAIL;
+ }
+
+ /* Verify all three scenarios have the same total_bw values */
+ for (i = 0; i < test_ctx->nr_cpus; i++) {
+ if (test_ctx->baseline_bw[i] != loaded_bw[i]) {
+ SCX_ERR("CPU%d: baseline_bw=%ld != loaded_bw=%ld",
+ i, test_ctx->baseline_bw[i], loaded_bw[i]);
+ return SCX_TEST_FAIL;
+ }
+
+ if (test_ctx->baseline_bw[i] != unloaded_bw[i]) {
+ SCX_ERR("CPU%d: baseline_bw=%ld != unloaded_bw=%ld",
+ i, test_ctx->baseline_bw[i], unloaded_bw[i]);
+ return SCX_TEST_FAIL;
+ }
+ }
+
+ fprintf(stderr, "All total_bw values are consistent across all scenarios\n");
+
+ /*
+ * Validate auto-register/unregister of dl_server bandwidth reservations.
+ *
+ * Doubling fair_server's runtime doubles its bw contribution. With a
+ * full-mode BPF scheduler (minimal_ops), the kernel should detach
+ * fair_server and attach ext_server, dropping total_bw back to its
+ * pre-customization (default ext_server-only) value. On unload, the
+ * fair_server reservation should come back with its customized runtime
+ * preserved, so total_bw doubles again.
+ */
+ if (read_fair_runtime_all(test_ctx->nr_cpus, original_runtime) < 0) {
+ fprintf(stderr, "Skipping attach/detach validation: debugfs not accessible\n");
+ return SCX_TEST_PASS;
+ }
+ doubled_runtime = original_runtime[0] * 2;
+
+ fprintf(stderr,
+ "Setting fair_server runtime to %ld ns on all CPUs (orig %ld)\n",
+ doubled_runtime, original_runtime[0]);
+
+ if (write_fair_runtime_all(test_ctx->nr_cpus, doubled_runtime) < 0) {
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+
+ if (fetch_verify_total_bw(doubled_bw, test_ctx->nr_cpus) < 0) {
+ SCX_ERR("Failed to get stable values after doubling fair runtime");
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+
+ /*
+ * After doubling the runtime, fair_server's bw contribution must grow.
+ * We don't assert exactly 2x, because the kernel's to_ratio() truncates
+ * the value, so 2 * to_ratio(period, runtime) and
+ * to_ratio(period, 2 * runtime) can differ.
+ */
+ for (i = 0; i < test_ctx->nr_cpus; i++) {
+ if (doubled_bw[i] <= test_ctx->baseline_bw[i]) {
+ SCX_ERR("CPU%d: fair did not increase total_bw (baseline=%ld, doubled=%ld)",
+ i, test_ctx->baseline_bw[i], doubled_bw[i]);
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+ }
+
+ link = bpf_map__attach_struct_ops(test_ctx->skel->maps.minimal_ops);
+ if (!link) {
+ SCX_ERR("Failed to attach scheduler for detach test");
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+
+ if (fetch_verify_total_bw(loaded_bw, test_ctx->nr_cpus) < 0) {
+ SCX_ERR("Failed to get stable values with BPF loaded (detach test)");
+ bpf_link__destroy(link);
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+
+ /*
+ * In full mode the customized fair_server is detached and ext_server is
+ * attached at its default runtime, total_bw must match baseline.
+ */
+ for (i = 0; i < test_ctx->nr_cpus; i++) {
+ if (loaded_bw[i] != test_ctx->baseline_bw[i]) {
+ SCX_ERR("CPU%d: expected bw %ld (fair detached, ext default), got %ld",
+ i, test_ctx->baseline_bw[i], loaded_bw[i]);
+ bpf_link__destroy(link);
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+ }
+
+ bpf_link__destroy(link);
+
+ if (fetch_verify_total_bw(unloaded_bw, test_ctx->nr_cpus) < 0) {
+ SCX_ERR("Failed to get stable values after BPF unload (detach test)");
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+
+ /*
+ * After unload, fair_server is re-attached with its preserved 2x
+ * runtime, so total_bw should return to the doubled value.
+ */
+ for (i = 0; i < test_ctx->nr_cpus; i++) {
+ if (unloaded_bw[i] != doubled_bw[i]) {
+ SCX_ERR("CPU%d: BPF unloaded: expected %ld (fair restored at 2x), got %ld",
+ i, doubled_bw[i], unloaded_bw[i]);
+ ret = SCX_TEST_FAIL;
+ goto restore;
+ }
+ }
+
+ fprintf(stderr,
+ "dl_server attach/detach with customized fair runtime verified\n");
+ ret = SCX_TEST_PASS;
+
+restore:
+ if (restore_fair_runtime_all(test_ctx->nr_cpus, original_runtime) < 0)
+ SCX_ERR("Failed to fully restore per-CPU fair_server runtimes");
+
+ return ret;
+}
+
+static void cleanup(void *ctx)
+{
+ struct total_bw_ctx *test_ctx = ctx;
+
+ if (test_ctx) {
+ if (test_ctx->skel)
+ minimal__destroy(test_ctx->skel);
+ free(test_ctx);
+ }
+}
+
+struct scx_test total_bw = {
+ .name = "total_bw",
+ .description = "Verify total_bw consistency across BPF program states",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&total_bw)
diff --git a/tools/testing/selftests/sched_ext/util.c b/tools/testing/selftests/sched_ext/util.c
index e47769c91918..2111329ed289 100644
--- a/tools/testing/selftests/sched_ext/util.c
+++ b/tools/testing/selftests/sched_ext/util.c
@@ -60,11 +60,11 @@ int file_write_long(const char *path, long val)
char buf[64];
int ret;
- ret = sprintf(buf, "%lu", val);
+ ret = sprintf(buf, "%ld", val);
if (ret < 0)
return ret;
- if (write_text(path, buf, sizeof(buf)) <= 0)
+ if (write_text(path, buf, ret) <= 0)
return -1;
return 0;
diff --git a/tools/testing/selftests/sched_ext/util.h b/tools/testing/selftests/sched_ext/util.h
index bc13dfec1267..681cec04b439 100644
--- a/tools/testing/selftests/sched_ext/util.h
+++ b/tools/testing/selftests/sched_ext/util.h
@@ -10,4 +10,4 @@
long file_read_long(const char *path);
int file_write_long(const char *path, long val);
-#endif // __SCX_TEST_H__
+#endif // __SCX_TEST_UTIL_H__