From: Andrea Righi <arighi@nvidia.com>
To: Tejun Heo <tj@kernel.org>, David Vernet <void@manifault.com>,
Changwoo Min <changwoo@igalia.com>
Cc: Ingo Molnar <mingo@redhat.com>,
Peter Zijlstra <peterz@infradead.org>,
Juri Lelli <juri.lelli@redhat.com>,
Vincent Guittot <vincent.guittot@linaro.org>,
Dietmar Eggemann <dietmar.eggemann@arm.com>,
Steven Rostedt <rostedt@goodmis.org>,
Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
Valentin Schneider <vschneid@redhat.com>,
K Prateek Nayak <kprateek.nayak@amd.com>,
Vladimir Vdovin <deliran@verdict.gg>,
Emil Tsalapatis <etsal@meta.com>,
Christian Loehle <christian.loehle@arm.com>,
Balbir Singh <balbirs@nvidia.com>,
Lee Trager <ltrager@nvidia.com>,
sched-ext@lists.linux.dev, linux-kernel@vger.kernel.org
Subject: [PATCH 5/5] selftests/sched_ext: Add a test for scx_bpf_task_numa_nid()
Date: Sun, 4 Oct 2026 09:27:11 +0200 [thread overview]
Message-ID: <20261004072901.3579967-6-arighi@nvidia.com> (raw)
In-Reply-To: <20261004072901.3579967-1-arighi@nvidia.com>
Add a scheduler which opts into NUMA hinting-fault scanning and reads
every running task's preferred node. Check that each result is either
NUMA_NO_NODE or a node reported by the kernel.
The test runs two threads touching memory for a few seconds, which on a
NUMA machine with numa_balancing enabled is enough for its tasks to be
scanned and acquire a preferred node. Two threads are needed because
the scan skips the pages of a single-threaded process that are already
on the node it runs on. Whether a node is acquired depends on the
machine, so the test only reports the counts it observed and does not
require one.
Assisted-by: LLM
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
tools/testing/selftests/sched_ext/Makefile | 1 +
.../selftests/sched_ext/numa_nid.bpf.c | 52 ++++++++
tools/testing/selftests/sched_ext/numa_nid.c | 111 ++++++++++++++++++
3 files changed, 164 insertions(+)
create mode 100644 tools/testing/selftests/sched_ext/numa_nid.bpf.c
create mode 100644 tools/testing/selftests/sched_ext/numa_nid.c
diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile
index c5d3a2eaea7da..707ab511acb51 100644
--- a/tools/testing/selftests/sched_ext/Makefile
+++ b/tools/testing/selftests/sched_ext/Makefile
@@ -186,6 +186,7 @@ auto-test-targets := \
non_scx_kfunc_deny \
nohz_tick \
numa \
+ numa_nid \
allowed_cpus \
peek_dsq \
prog_run \
diff --git a/tools/testing/selftests/sched_ext/numa_nid.bpf.c b/tools/testing/selftests/sched_ext/numa_nid.bpf.c
new file mode 100644
index 0000000000000..54f7bdbc0a3d0
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/numa_nid.bpf.c
@@ -0,0 +1,52 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A scheduler that reads the NUMA node a task's hinting faults point at.
+ *
+ * Every task that runs is asked for its preferred node. The answer is either
+ * NUMA_NO_NODE, which a task keeps until its address space has been scanned a
+ * few times, or a node id the kernel knows about. Anything else means the
+ * value reaching BPF is not the one the placement code maintains.
+ *
+ * Copyright (c) 2026 NVIDIA Corporation.
+ */
+
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+UEI_DEFINE(uei);
+
+/* Tasks seen with and without a preferred node. */
+u64 nr_with_nid;
+u64 nr_without_nid;
+
+void BPF_STRUCT_OPS(numa_nid_running, struct task_struct *p)
+{
+ s32 nid = scx_bpf_task_numa_nid(p);
+
+ if (nid == NUMA_NO_NODE) {
+ __sync_fetch_and_add(&nr_without_nid, 1);
+ return;
+ }
+
+ if (nid < 0 || nid >= scx_bpf_nr_node_ids()) {
+ scx_bpf_error("task %d reported node %d, kernel has %d node ids",
+ p->pid, nid, scx_bpf_nr_node_ids());
+ return;
+ }
+
+ __sync_fetch_and_add(&nr_with_nid, 1);
+}
+
+void BPF_STRUCT_OPS(numa_nid_exit, struct scx_exit_info *ei)
+{
+ UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops numa_nid_ops = {
+ .running = (void *)numa_nid_running,
+ .exit = (void *)numa_nid_exit,
+ .flags = SCX_OPS_NUMA_BALANCING,
+ .name = "numa_nid",
+};
diff --git a/tools/testing/selftests/sched_ext/numa_nid.c b/tools/testing/selftests/sched_ext/numa_nid.c
new file mode 100644
index 0000000000000..3aa1630ff6844
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/numa_nid.c
@@ -0,0 +1,111 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 NVIDIA Corporation.
+ */
+#define _GNU_SOURCE
+#include <bpf/bpf.h>
+#include <pthread.h>
+#include <scx/common.h>
+#include <stdlib.h>
+#include <time.h>
+#include <unistd.h>
+#include "numa_nid.bpf.skel.h"
+#include "scx_test.h"
+
+#define WORKLOAD_SIZE (64 << 20)
+#define WORKLOAD_THREADS 2
+#define WORKLOAD_SECONDS 3
+
+/*
+ * Touch a private buffer for a while. The hinting-fault scan skips the pages
+ * of a single-threaded process that are already on the node it runs on, so
+ * one such thread alone would never take a hinting fault: run two of them.
+ */
+static void *touch_memory(void *arg)
+{
+ struct timespec start, now;
+ volatile char *buffer;
+ size_t offset;
+
+ buffer = malloc(WORKLOAD_SIZE);
+ if (!buffer)
+ return NULL;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+ do {
+ for (offset = 0; offset < WORKLOAD_SIZE; offset += 4096)
+ buffer[offset]++;
+ clock_gettime(CLOCK_MONOTONIC, &now);
+ } while (now.tv_sec - start.tv_sec < WORKLOAD_SECONDS);
+
+ free((void *)buffer);
+ return NULL;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+ struct numa_nid *skel;
+
+ skel = numa_nid__open();
+ SCX_FAIL_IF(!skel, "Failed to open");
+ SCX_ENUM_INIT(skel);
+ SCX_FAIL_IF(numa_nid__load(skel), "Failed to load skel");
+
+ *ctx = skel;
+
+ return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+ struct numa_nid *skel = ctx;
+ pthread_t threads[WORKLOAD_THREADS - 1];
+ struct bpf_link *link;
+ int i;
+
+ link = bpf_map__attach_struct_ops(skel->maps.numa_nid_ops);
+ SCX_FAIL_IF(!link, "Failed to attach scheduler");
+
+ /*
+ * Give the scan something to work on, so that the tasks of this test
+ * can acquire a preferred node on a NUMA machine with numa_balancing
+ * enabled. The test does not require one, as it depends on the
+ * machine: the scheduler reports an error through UEI if it ever sees
+ * a node id that is not valid and the counts below show what it saw.
+ */
+ for (i = 0; i < WORKLOAD_THREADS - 1; i++) {
+ if (pthread_create(&threads[i], NULL, touch_memory, NULL)) {
+ SCX_ERR("Failed to create a worker thread");
+ bpf_link__destroy(link);
+ return SCX_TEST_FAIL;
+ }
+ }
+ touch_memory(NULL);
+ for (i = 0; i < WORKLOAD_THREADS - 1; i++)
+ pthread_join(threads[i], NULL);
+
+ fprintf(stderr, "tasks with a preferred node: %lu, without: %lu\n",
+ (unsigned long)skel->bss->nr_with_nid,
+ (unsigned long)skel->bss->nr_without_nid);
+
+ bpf_link__destroy(link);
+ SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG));
+
+ return SCX_TEST_PASS;
+}
+
+static void cleanup(void *ctx)
+{
+ struct numa_nid *skel = ctx;
+
+ numa_nid__destroy(skel);
+}
+
+struct scx_test numa_nid = {
+ .name = "numa_nid",
+ .description = "Read a task's preferred NUMA node from BPF",
+ .setup = setup,
+ .run = run,
+ .cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&numa_nid)
--
2.55.0
prev parent reply other threads:[~2026-10-04 7:29 UTC|newest]
Thread overview: 6+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-04 7:27 [PATCHSET sched_ext/for-7.4] sched_ext: Add NUMA balancing support Andrea Righi
2026-10-04 7:27 ` [PATCH 1/5] sched/numa: Let other scheduling classes drive NUMA scanning Andrea Righi
2026-10-04 7:27 ` [PATCH 2/5] sched/numa: Leave the placement of a BPF-scheduled task to its scheduler Andrea Righi
2026-10-04 7:27 ` [PATCH 3/5] sched_ext: Scan NUMA hinting faults for opted-in BPF schedulers Andrea Righi
2026-10-04 7:27 ` [PATCH 4/5] sched_ext: Add scx_bpf_task_numa_nid() Andrea Righi
2026-10-04 7:27 ` Andrea Righi [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261004072901.3579967-6-arighi@nvidia.com \
--to=arighi@nvidia.com \
--cc=balbirs@nvidia.com \
--cc=bsegall@google.com \
--cc=changwoo@igalia.com \
--cc=christian.loehle@arm.com \
--cc=deliran@verdict.gg \
--cc=dietmar.eggemann@arm.com \
--cc=etsal@meta.com \
--cc=juri.lelli@redhat.com \
--cc=kprateek.nayak@amd.com \
--cc=linux-kernel@vger.kernel.org \
--cc=ltrager@nvidia.com \
--cc=mgorman@suse.de \
--cc=mingo@redhat.com \
--cc=peterz@infradead.org \
--cc=rostedt@goodmis.org \
--cc=sched-ext@lists.linux.dev \
--cc=tj@kernel.org \
--cc=vincent.guittot@linaro.org \
--cc=void@manifault.com \
--cc=vschneid@redhat.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®