mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Kuba Piecuch <jpiecuch@google.com>
To: Tejun Heo <tj@kernel.org>, David Vernet <void@manifault.com>,
	Andrea Righi <arighi@nvidia.com>,
	 Changwoo Min <changwoo@igalia.com>
Cc: Kuba Piecuch <jpiecuch@google.com>,
	Emil Tsalapatis <emil@etsalapatis.com>,
	sched-ext@lists.linux.dev,  linux-kernel@vger.kernel.org
Subject: [PATCH v2 2/2] selftests/sched_ext: Add a test for ops.dequeue() on remote local DSQ moves
Date: Wed, 30 Sep 2026 11:47:23 +0000	[thread overview]
Message-ID: <20260930114725.331370-3-jpiecuch@google.com> (raw)
In-Reply-To: <20260930114725.331370-1-jpiecuch@google.com>

Add a dequeue_remote test that makes moves of tasks in the BPF
scheduler's custody to another CPU's local DSQ the common case, both via
SCX_DSQ_LOCAL_ON dispatch and via scx_bpf_dsq_move_to_local(). The BPF
scheduler tracks each task's custody state and triggers scx_bpf_error()
if a custody period doesn't end with exactly one ops.dequeue() before
the task runs, or if it ends with an SCX_DEQ_CORE_SCHED_EXEC dequeue of
a task without a core cookie.

Without the previous patch, the test fails with:

  sched_ext: dequeue_remote: dequeue_remote.bpf.c:141: 15 (rcu_preempt): late ops.dequeue() with SCX_DEQ_CORE_SCHED_EXEC (enq_cpu=3 cpu=2 seq=1)
     ...
     ops_dequeue+0x114/0x170
     set_next_task_scx+0x104/0x1e0
     __pick_next_task+0xc7/0x180
     __schedule+0x154/0x1870

Assisted-by: Claude:claude-opus-5.5
Signed-off-by: Kuba Piecuch <jpiecuch@google.com>
---
 tools/testing/selftests/sched_ext/Makefile    |   1 +
 .../selftests/sched_ext/dequeue_remote.bpf.c  | 265 ++++++++++++++++++
 .../selftests/sched_ext/dequeue_remote.c      | 202 +++++++++++++
 3 files changed, 468 insertions(+)
 create mode 100644 tools/testing/selftests/sched_ext/dequeue_remote.bpf.c
 create mode 100644 tools/testing/selftests/sched_ext/dequeue_remote.c

diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile
index 4e06d0baaeec..af919a56a4f4 100644
--- a/tools/testing/selftests/sched_ext/Makefile
+++ b/tools/testing/selftests/sched_ext/Makefile
@@ -165,6 +165,7 @@ auto-test-targets :=			\
 	create_dsq			\
 	dequeue				\
 	dequeue_iter			\
+	dequeue_remote			\
 	enq_last_no_enq_fails		\
 	ddsp_bogus_dsq_fail		\
 	ddsp_vtimelocal_fail		\
diff --git a/tools/testing/selftests/sched_ext/dequeue_remote.bpf.c b/tools/testing/selftests/sched_ext/dequeue_remote.bpf.c
new file mode 100644
index 000000000000..55fc77a0de0b
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/dequeue_remote.bpf.c
@@ -0,0 +1,265 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Verify that ops.dequeue() is called when a task leaves the BPF scheduler's
+ * custody by being moved to the local DSQ of a CPU other than the one whose
+ * rq it's on (move_remote_task_to_local_dsq()).
+ *
+ * ops.enqueue() puts every task into custody and ops.dispatch() moves it to
+ * the dispatching CPU's local DSQ, so most moves cross CPUs. With
+ * @test_use_move_to_local, tasks are queued on a user DSQ and consumed with
+ * scx_bpf_dsq_move_to_local(). Otherwise, they are queued in a BPF queue and
+ * dispatched with SCX_DSQ_LOCAL_ON.
+ *
+ * Copyright (c) 2026 Google LLC.
+ */
+
+#include <scx/common.bpf.h>
+
+#define SHARED_DSQ		0
+#define MAX_DISPATCH_POPS	8
+
+char _license[] SEC("license") = "GPL";
+
+UEI_DEFINE(uei);
+
+struct {
+	__uint(type, BPF_MAP_TYPE_QUEUE);
+	__uint(max_entries, 32768);
+	__type(value, s32);
+} global_queue SEC(".maps");
+
+enum task_state {
+	TASK_NONE = 0,
+	TASK_ENQUEUED,		/* in BPF custody, waiting for ops.dequeue() */
+	TASK_DISPATCHED,	/* left custody */
+};
+
+struct task_ctx {
+	enum task_state state;
+	s32 enq_cpu;		/* scx_bpf_task_cpu() at ops.enqueue() */
+	u64 enqueue_seq;
+};
+
+struct {
+	__uint(type, BPF_MAP_TYPE_TASK_STORAGE);
+	__uint(map_flags, BPF_F_NO_PREALLOC);
+	__type(key, int);
+	__type(value, struct task_ctx);
+} task_ctx_stor SEC(".maps");
+
+/* core_cookie only exists with CONFIG_SCHED_CORE */
+struct task_struct___core_sched {
+	unsigned long core_cookie;
+} __attribute__((preserve_access_index));
+
+bool test_use_move_to_local;
+
+u64 enqueue_cnt, dequeue_cnt, dispatch_dequeue_cnt, change_dequeue_cnt;
+u64 remote_dispatch_cnt, remote_running_cnt, missed_dequeue_cnt;
+u64 core_sched_exec_dequeue_cnt;
+
+static struct task_ctx *lookup_task_ctx(struct task_struct *p)
+{
+	return bpf_task_storage_get(&task_ctx_stor, p, 0, 0);
+}
+
+static bool task_has_core_cookie(struct task_struct *p)
+{
+	struct task_struct___core_sched *t = (void *)p;
+
+	if (!bpf_core_field_exists(t->core_cookie))
+		return false;
+	return BPF_CORE_READ(t, core_cookie);
+}
+
+s32 BPF_STRUCT_OPS(dequeue_remote_select_cpu, struct task_struct *p,
+		   s32 prev_cpu, u64 wake_flags)
+{
+	/* no direct dispatch, always go through ops.enqueue() */
+	return prev_cpu;
+}
+
+void BPF_STRUCT_OPS(dequeue_remote_enqueue, struct task_struct *p, u64 enq_flags)
+{
+	struct task_ctx *tctx;
+	s32 pid = p->pid;
+
+	tctx = lookup_task_ctx(p);
+	if (!tctx) {
+		scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+		return;
+	}
+
+	/* the previous custody period must have ended with ops.dequeue() */
+	if (tctx->state == TASK_ENQUEUED)
+		scx_bpf_error("%d (%s): enqueue while in ENQUEUED state seq=%llu",
+			      p->pid, p->comm, tctx->enqueue_seq);
+
+	/*
+	 * Mark @p as enqueued before making it visible to ops.dispatch() on
+	 * other CPUs, which skips queue entries of tasks not in ENQUEUED
+	 * state as stale.
+	 */
+	tctx->state = TASK_ENQUEUED;
+	tctx->enq_cpu = scx_bpf_task_cpu(p);
+	tctx->enqueue_seq++;
+
+	if (test_use_move_to_local) {
+		scx_bpf_dsq_insert(p, SHARED_DSQ, SCX_SLICE_DFL, enq_flags);
+	} else if (bpf_map_push_elem(&global_queue, &pid, 0)) {
+		scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, enq_flags);
+		tctx->state = TASK_DISPATCHED;
+		tctx->enq_cpu = -1;
+		goto out;
+	}
+
+	__sync_fetch_and_add(&enqueue_cnt, 1);
+out:
+	scx_bpf_kick_cpu(scx_bpf_task_cpu(p), SCX_KICK_IDLE);
+}
+
+void BPF_STRUCT_OPS(dequeue_remote_dequeue, struct task_struct *p, u64 deq_flags)
+{
+	struct task_ctx *tctx;
+
+	__sync_fetch_and_add(&dequeue_cnt, 1);
+
+	tctx = lookup_task_ctx(p);
+	if (!tctx)
+		return;
+
+	/*
+	 * Only core scheduling can pick a task straight out of custody, and
+	 * only if the task has a core cookie. Otherwise, the custody exit was
+	 * missed when @p was inserted into a local DSQ and got deferred until
+	 * @p was picked.
+	 */
+	if ((deq_flags & SCX_DEQ_CORE_SCHED_EXEC) && !task_has_core_cookie(p)) {
+		__sync_fetch_and_add(&core_sched_exec_dequeue_cnt, 1);
+		scx_bpf_error("%d (%s): late ops.dequeue() with SCX_DEQ_CORE_SCHED_EXEC (enq_cpu=%d cpu=%d seq=%llu)",
+			      p->pid, p->comm, tctx->enq_cpu,
+			      scx_bpf_task_cpu(p), tctx->enqueue_seq);
+	}
+
+	/* ops.dequeue() ends the custody period started by ops.enqueue() */
+	if (tctx->state != TASK_ENQUEUED)
+		scx_bpf_error("%d (%s): dequeue outside custody deq_flags=0x%llx state=%d seq=%llu",
+			      p->pid, p->comm, deq_flags, tctx->state,
+			      tctx->enqueue_seq);
+
+	if (deq_flags & SCX_DEQ_SCHED_CHANGE) {
+		__sync_fetch_and_add(&change_dequeue_cnt, 1);
+		tctx->state = TASK_NONE;
+	} else {
+		__sync_fetch_and_add(&dispatch_dequeue_cnt, 1);
+		tctx->state = TASK_DISPATCHED;
+	}
+}
+
+void BPF_STRUCT_OPS(dequeue_remote_dispatch, s32 cpu, struct task_struct *prev)
+{
+	struct task_ctx *tctx;
+	struct task_struct *p;
+	s32 pid;
+	int i;
+
+	if (test_use_move_to_local) {
+		scx_bpf_dsq_move_to_local(SHARED_DSQ, 0);
+		return;
+	}
+
+	/* pop past stale entries so that they don't leave this CPU idle */
+	bpf_for(i, 0, MAX_DISPATCH_POPS) {
+		if (bpf_map_pop_elem(&global_queue, &pid))
+			return;
+
+		p = bpf_task_from_pid(pid);
+		if (!p)
+			continue;
+
+		/*
+		 * Entries are stale if @p left custody through a property
+		 * change dequeue or was dispatched from a duplicate entry.
+		 */
+		tctx = lookup_task_ctx(p);
+		if (!tctx || tctx->state != TASK_ENQUEUED) {
+			bpf_task_release(p);
+			continue;
+		}
+
+		if (bpf_cpumask_test_cpu(cpu, p->cpus_ptr)) {
+			if (scx_bpf_task_cpu(p) != cpu)
+				__sync_fetch_and_add(&remote_dispatch_cnt, 1);
+			scx_bpf_dsq_insert(p, SCX_DSQ_LOCAL_ON | cpu,
+					   SCX_SLICE_DFL, 0);
+		} else {
+			scx_bpf_dsq_insert(p, SCX_DSQ_GLOBAL, SCX_SLICE_DFL, 0);
+		}
+
+		bpf_task_release(p);
+		return;
+	}
+}
+
+void BPF_STRUCT_OPS(dequeue_remote_running, struct task_struct *p)
+{
+	struct task_ctx *tctx;
+
+	tctx = lookup_task_ctx(p);
+	if (!tctx)
+		return;
+
+	/* tasks can only run from a local DSQ, i.e. after leaving custody */
+	if (tctx->state == TASK_ENQUEUED) {
+		__sync_fetch_and_add(&missed_dequeue_cnt, 1);
+		scx_bpf_error("%d (%s): running without ops.dequeue() (enq_cpu=%d cpu=%d seq=%llu)",
+			      p->pid, p->comm, tctx->enq_cpu,
+			      scx_bpf_task_cpu(p), tctx->enqueue_seq);
+		return;
+	}
+
+	if (tctx->enq_cpu >= 0 && tctx->enq_cpu != scx_bpf_task_cpu(p))
+		__sync_fetch_and_add(&remote_running_cnt, 1);
+	tctx->enq_cpu = -1;
+}
+
+s32 BPF_STRUCT_OPS(dequeue_remote_init_task, struct task_struct *p,
+		   struct scx_init_task_args *args)
+{
+	struct task_ctx *tctx;
+
+	tctx = bpf_task_storage_get(&task_ctx_stor, p, 0,
+				    BPF_LOCAL_STORAGE_GET_F_CREATE);
+	if (!tctx)
+		return -ENOMEM;
+
+	/* task storage persists across attachments, start from scratch */
+	tctx->state = TASK_NONE;
+	tctx->enq_cpu = -1;
+
+	return 0;
+}
+
+s32 BPF_STRUCT_OPS_SLEEPABLE(dequeue_remote_init)
+{
+	return scx_bpf_create_dsq(SHARED_DSQ, -1);
+}
+
+void BPF_STRUCT_OPS(dequeue_remote_exit, struct scx_exit_info *ei)
+{
+	UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops dequeue_remote_ops = {
+	.select_cpu		= (void *)dequeue_remote_select_cpu,
+	.enqueue		= (void *)dequeue_remote_enqueue,
+	.dequeue		= (void *)dequeue_remote_dequeue,
+	.dispatch		= (void *)dequeue_remote_dispatch,
+	.running		= (void *)dequeue_remote_running,
+	.init_task		= (void *)dequeue_remote_init_task,
+	.init			= (void *)dequeue_remote_init,
+	.exit			= (void *)dequeue_remote_exit,
+	.flags			= SCX_OPS_ENQ_LAST,
+	.name			= "dequeue_remote",
+};
diff --git a/tools/testing/selftests/sched_ext/dequeue_remote.c b/tools/testing/selftests/sched_ext/dequeue_remote.c
new file mode 100644
index 000000000000..0f03044a5bc6
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/dequeue_remote.c
@@ -0,0 +1,202 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Verify that ops.dequeue() is called for tasks leaving BPF custody through
+ * an SCX-internal cross-CPU migration (move_remote_task_to_local_dsq()).
+ *
+ * Copyright (c) 2026 Google LLC.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <unistd.h>
+#include <signal.h>
+#include <time.h>
+#include <sched.h>
+#include <bpf/bpf.h>
+#include <scx/common.h>
+#include <sys/wait.h>
+#include "scx_test.h"
+#include "dequeue_remote.bpf.skel.h"
+
+#define MAX_WORKERS	64
+#define RUN_MS		2000
+
+static int nr_cpus;
+
+static long long now_ms(void)
+{
+	struct timespec ts;
+
+	clock_gettime(CLOCK_MONOTONIC, &ts);
+	return ts.tv_sec * 1000LL + ts.tv_nsec / 1000000;
+}
+
+/* mix of short bursts and sleeps to generate lots of enqueues and wakeups */
+static void worker_fn(int id)
+{
+	long long end = now_ms() + RUN_MS;
+	volatile unsigned long sum = 0;
+
+	while (now_ms() < end) {
+		unsigned long j;
+
+		for (j = 0; j < 20000 + id * 1000; j++)
+			sum += j;
+		if (id & 1)
+			usleep(100);
+		else
+			sched_yield();
+	}
+
+	exit(0);
+}
+
+static enum scx_test_status run_scenario(struct dequeue_remote *skel,
+					 bool use_move_to_local,
+					 const char *name)
+{
+	enum scx_test_status ret = SCX_TEST_PASS;
+	struct bpf_link *link;
+	pid_t pids[MAX_WORKERS];
+	int nr_workers, nr_forked, i;
+
+	nr_workers = 2 * nr_cpus;
+	if (nr_workers < 4)
+		nr_workers = 4;
+	if (nr_workers > MAX_WORKERS)
+		nr_workers = MAX_WORKERS;
+
+	skel->bss->test_use_move_to_local = use_move_to_local;
+	skel->bss->enqueue_cnt = 0;
+	skel->bss->dequeue_cnt = 0;
+	skel->bss->dispatch_dequeue_cnt = 0;
+	skel->bss->change_dequeue_cnt = 0;
+	skel->bss->remote_dispatch_cnt = 0;
+	skel->bss->remote_running_cnt = 0;
+	skel->bss->missed_dequeue_cnt = 0;
+	skel->bss->core_sched_exec_dequeue_cnt = 0;
+
+	link = bpf_map__attach_struct_ops(skel->maps.dequeue_remote_ops);
+	SCX_FAIL_IF(!link, "Failed to attach struct_ops for %s", name);
+
+	fflush(stdout);
+	fflush(stderr);
+
+	for (nr_forked = 0; nr_forked < nr_workers; nr_forked++) {
+		pid_t pid = fork();
+
+		if (pid < 0) {
+			SCX_ERR("Failed to fork worker %d", nr_forked);
+			ret = SCX_TEST_FAIL;
+			break;
+		}
+		if (pid == 0)
+			worker_fn(nr_forked);
+		pids[nr_forked] = pid;
+	}
+
+	/* on failure, kill the remaining workers but still reap them */
+	for (i = 0; i < nr_forked; i++) {
+		int status;
+
+		if (ret != SCX_TEST_PASS)
+			kill(pids[i], SIGKILL);
+
+		if (waitpid(pids[i], &status, 0) != pids[i]) {
+			SCX_ERR("Failed to wait for worker %d", i);
+			ret = SCX_TEST_FAIL;
+		} else if (ret == SCX_TEST_PASS && status != 0) {
+			SCX_ERR("Worker %d exited with status %d", i, status);
+			ret = SCX_TEST_FAIL;
+		}
+	}
+
+	bpf_link__destroy(link);
+
+	if (ret != SCX_TEST_PASS)
+		return ret;
+
+	printf("%s:\n", name);
+	printf("  workers: %d\n", nr_workers);
+	printf("  enqueues: %lu\n", (unsigned long)skel->bss->enqueue_cnt);
+	printf("  dequeues: %lu (dispatch: %lu, property_change: %lu)\n",
+	       (unsigned long)skel->bss->dequeue_cnt,
+	       (unsigned long)skel->bss->dispatch_dequeue_cnt,
+	       (unsigned long)skel->bss->change_dequeue_cnt);
+	if (!use_move_to_local)
+		printf("  remote SCX_DSQ_LOCAL_ON dispatches: %lu\n",
+		       (unsigned long)skel->bss->remote_dispatch_cnt);
+	printf("  ran on a CPU other than the enqueue CPU: %lu\n",
+	       (unsigned long)skel->bss->remote_running_cnt);
+	printf("  ran without ops.dequeue(): %lu\n",
+	       (unsigned long)skel->bss->missed_dequeue_cnt);
+	printf("  late SCX_DEQ_CORE_SCHED_EXEC dequeues: %lu\n",
+	       (unsigned long)skel->bss->core_sched_exec_dequeue_cnt);
+
+	if (skel->data->uei.kind != EXIT_KIND(SCX_EXIT_UNREG))
+		SCX_ERR("Scheduler exited with kind=%lld: %s",
+			(long long)skel->data->uei.kind, skel->data->uei.msg);
+	SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG));
+
+	/* the test is meaningless if no task was moved across CPUs */
+	SCX_GT(skel->bss->remote_running_cnt, 0);
+	SCX_EQ(skel->bss->enqueue_cnt, skel->bss->dequeue_cnt);
+
+	return SCX_TEST_PASS;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+	struct dequeue_remote *skel;
+	cpu_set_t cpus;
+
+	SCX_FAIL_IF(sched_getaffinity(0, sizeof(cpus), &cpus),
+		    "Failed to get CPU affinity");
+	nr_cpus = CPU_COUNT(&cpus);
+	if (nr_cpus < 2) {
+		fprintf(stderr, "Skipping: requires at least 2 usable CPUs\n");
+		return SCX_TEST_SKIP;
+	}
+
+	skel = SCX_OPS_OPEN(dequeue_remote_ops, dequeue_remote);
+	SCX_OPS_LOAD(skel, dequeue_remote_ops, dequeue_remote, uei);
+
+	*ctx = skel;
+
+	return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+	struct dequeue_remote *skel = ctx;
+	enum scx_test_status status;
+
+	status = run_scenario(skel, false,
+			      "BPF queue -> SCX_DSQ_LOCAL_ON | cpu");
+	if (status != SCX_TEST_PASS)
+		return status;
+
+	status = run_scenario(skel, true,
+			      "user DSQ -> scx_bpf_dsq_move_to_local()");
+	if (status != SCX_TEST_PASS)
+		return status;
+
+	return SCX_TEST_PASS;
+}
+
+static void cleanup(void *ctx)
+{
+	struct dequeue_remote *skel = ctx;
+
+	dequeue_remote__destroy(skel);
+}
+
+struct scx_test dequeue_remote_test = {
+	.name = "dequeue_remote",
+	.description = "Verify ops.dequeue() on SCX-internal cross-CPU migrations",
+	.setup = setup,
+	.run = run,
+	.cleanup = cleanup,
+};
+
+REGISTER_SCX_TEST(&dequeue_remote_test)
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


  parent reply	other threads:[~2026-09-30 11:47 UTC|newest]

Thread overview: 5+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-30 11:47 [PATCHSET v2 sched_ext/for-7.3-fixes] sched_ext: Fix missing " Kuba Piecuch
2026-09-30 11:47 ` [PATCH v2 1/2] sched_ext: Call ops.dequeue() when a task arrives on a remote local DSQ Kuba Piecuch
2026-09-30 11:47 ` Kuba Piecuch [this message]
2026-09-30 13:53   ` [PATCH v2 2/2] selftests/sched_ext: Add a test for ops.dequeue() on remote local DSQ moves Andrea Righi
2026-09-30 14:27     ` Kuba Piecuch

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260930114725.331370-3-jpiecuch@google.com \
    --to=jpiecuch@google.com \
    --cc=arighi@nvidia.com \
    --cc=changwoo@igalia.com \
    --cc=emil@etsalapatis.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=sched-ext@lists.linux.dev \
    --cc=tj@kernel.org \
    --cc=void@manifault.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®