mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Andrea Righi <arighi@nvidia.com>
To: Tejun Heo <tj@kernel.org>, David Vernet <void@manifault.com>,
	Changwoo Min <changwoo@igalia.com>
Cc: Ingo Molnar <mingo@redhat.com>,
	Peter Zijlstra <peterz@infradead.org>,
	Juri Lelli <juri.lelli@redhat.com>,
	Vincent Guittot <vincent.guittot@linaro.org>,
	Dietmar Eggemann <dietmar.eggemann@arm.com>,
	Steven Rostedt <rostedt@goodmis.org>,
	Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
	Valentin Schneider <vschneid@redhat.com>,
	K Prateek Nayak <kprateek.nayak@amd.com>,
	Vladimir Vdovin <deliran@verdict.gg>,
	Emil Tsalapatis <etsal@meta.com>,
	Christian Loehle <christian.loehle@arm.com>,
	Balbir Singh <balbirs@nvidia.com>,
	Lee Trager <ltrager@nvidia.com>,
	sched-ext@lists.linux.dev, linux-kernel@vger.kernel.org
Subject: [PATCH 5/5] selftests/sched_ext: Add a test for scx_bpf_task_numa_nid()
Date: Sun,  4 Oct 2026 09:27:11 +0200	[thread overview]
Message-ID: <20261004072901.3579967-6-arighi@nvidia.com> (raw)
In-Reply-To: <20261004072901.3579967-1-arighi@nvidia.com>

Add a scheduler which opts into NUMA hinting-fault scanning and reads
every running task's preferred node. Check that each result is either
NUMA_NO_NODE or a node reported by the kernel.

The test runs two threads touching memory for a few seconds, which on a
NUMA machine with numa_balancing enabled is enough for its tasks to be
scanned and acquire a preferred node. Two threads are needed because
the scan skips the pages of a single-threaded process that are already
on the node it runs on. Whether a node is acquired depends on the
machine, so the test only reports the counts it observed and does not
require one.

Assisted-by: LLM
Signed-off-by: Andrea Righi <arighi@nvidia.com>
---
 tools/testing/selftests/sched_ext/Makefile    |   1 +
 .../selftests/sched_ext/numa_nid.bpf.c        |  52 ++++++++
 tools/testing/selftests/sched_ext/numa_nid.c  | 111 ++++++++++++++++++
 3 files changed, 164 insertions(+)
 create mode 100644 tools/testing/selftests/sched_ext/numa_nid.bpf.c
 create mode 100644 tools/testing/selftests/sched_ext/numa_nid.c

diff --git a/tools/testing/selftests/sched_ext/Makefile b/tools/testing/selftests/sched_ext/Makefile
index c5d3a2eaea7da..707ab511acb51 100644
--- a/tools/testing/selftests/sched_ext/Makefile
+++ b/tools/testing/selftests/sched_ext/Makefile
@@ -186,6 +186,7 @@ auto-test-targets :=			\
 	non_scx_kfunc_deny		\
 	nohz_tick			\
 	numa				\
+	numa_nid			\
 	allowed_cpus			\
 	peek_dsq			\
 	prog_run			\
diff --git a/tools/testing/selftests/sched_ext/numa_nid.bpf.c b/tools/testing/selftests/sched_ext/numa_nid.bpf.c
new file mode 100644
index 0000000000000..54f7bdbc0a3d0
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/numa_nid.bpf.c
@@ -0,0 +1,52 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A scheduler that reads the NUMA node a task's hinting faults point at.
+ *
+ * Every task that runs is asked for its preferred node. The answer is either
+ * NUMA_NO_NODE, which a task keeps until its address space has been scanned a
+ * few times, or a node id the kernel knows about. Anything else means the
+ * value reaching BPF is not the one the placement code maintains.
+ *
+ * Copyright (c) 2026 NVIDIA Corporation.
+ */
+
+#include <scx/common.bpf.h>
+
+char _license[] SEC("license") = "GPL";
+
+UEI_DEFINE(uei);
+
+/* Tasks seen with and without a preferred node. */
+u64 nr_with_nid;
+u64 nr_without_nid;
+
+void BPF_STRUCT_OPS(numa_nid_running, struct task_struct *p)
+{
+	s32 nid = scx_bpf_task_numa_nid(p);
+
+	if (nid == NUMA_NO_NODE) {
+		__sync_fetch_and_add(&nr_without_nid, 1);
+		return;
+	}
+
+	if (nid < 0 || nid >= scx_bpf_nr_node_ids()) {
+		scx_bpf_error("task %d reported node %d, kernel has %d node ids",
+			      p->pid, nid, scx_bpf_nr_node_ids());
+		return;
+	}
+
+	__sync_fetch_and_add(&nr_with_nid, 1);
+}
+
+void BPF_STRUCT_OPS(numa_nid_exit, struct scx_exit_info *ei)
+{
+	UEI_RECORD(uei, ei);
+}
+
+SEC(".struct_ops.link")
+struct sched_ext_ops numa_nid_ops = {
+	.running			= (void *)numa_nid_running,
+	.exit				= (void *)numa_nid_exit,
+	.flags				= SCX_OPS_NUMA_BALANCING,
+	.name				= "numa_nid",
+};
diff --git a/tools/testing/selftests/sched_ext/numa_nid.c b/tools/testing/selftests/sched_ext/numa_nid.c
new file mode 100644
index 0000000000000..3aa1630ff6844
--- /dev/null
+++ b/tools/testing/selftests/sched_ext/numa_nid.c
@@ -0,0 +1,111 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2026 NVIDIA Corporation.
+ */
+#define _GNU_SOURCE
+#include <bpf/bpf.h>
+#include <pthread.h>
+#include <scx/common.h>
+#include <stdlib.h>
+#include <time.h>
+#include <unistd.h>
+#include "numa_nid.bpf.skel.h"
+#include "scx_test.h"
+
+#define WORKLOAD_SIZE		(64 << 20)
+#define WORKLOAD_THREADS	2
+#define WORKLOAD_SECONDS	3
+
+/*
+ * Touch a private buffer for a while. The hinting-fault scan skips the pages
+ * of a single-threaded process that are already on the node it runs on, so
+ * one such thread alone would never take a hinting fault: run two of them.
+ */
+static void *touch_memory(void *arg)
+{
+	struct timespec start, now;
+	volatile char *buffer;
+	size_t offset;
+
+	buffer = malloc(WORKLOAD_SIZE);
+	if (!buffer)
+		return NULL;
+
+	clock_gettime(CLOCK_MONOTONIC, &start);
+	do {
+		for (offset = 0; offset < WORKLOAD_SIZE; offset += 4096)
+			buffer[offset]++;
+		clock_gettime(CLOCK_MONOTONIC, &now);
+	} while (now.tv_sec - start.tv_sec < WORKLOAD_SECONDS);
+
+	free((void *)buffer);
+	return NULL;
+}
+
+static enum scx_test_status setup(void **ctx)
+{
+	struct numa_nid *skel;
+
+	skel = numa_nid__open();
+	SCX_FAIL_IF(!skel, "Failed to open");
+	SCX_ENUM_INIT(skel);
+	SCX_FAIL_IF(numa_nid__load(skel), "Failed to load skel");
+
+	*ctx = skel;
+
+	return SCX_TEST_PASS;
+}
+
+static enum scx_test_status run(void *ctx)
+{
+	struct numa_nid *skel = ctx;
+	pthread_t threads[WORKLOAD_THREADS - 1];
+	struct bpf_link *link;
+	int i;
+
+	link = bpf_map__attach_struct_ops(skel->maps.numa_nid_ops);
+	SCX_FAIL_IF(!link, "Failed to attach scheduler");
+
+	/*
+	 * Give the scan something to work on, so that the tasks of this test
+	 * can acquire a preferred node on a NUMA machine with numa_balancing
+	 * enabled. The test does not require one, as it depends on the
+	 * machine: the scheduler reports an error through UEI if it ever sees
+	 * a node id that is not valid and the counts below show what it saw.
+	 */
+	for (i = 0; i < WORKLOAD_THREADS - 1; i++) {
+		if (pthread_create(&threads[i], NULL, touch_memory, NULL)) {
+			SCX_ERR("Failed to create a worker thread");
+			bpf_link__destroy(link);
+			return SCX_TEST_FAIL;
+		}
+	}
+	touch_memory(NULL);
+	for (i = 0; i < WORKLOAD_THREADS - 1; i++)
+		pthread_join(threads[i], NULL);
+
+	fprintf(stderr, "tasks with a preferred node: %lu, without: %lu\n",
+		(unsigned long)skel->bss->nr_with_nid,
+		(unsigned long)skel->bss->nr_without_nid);
+
+	bpf_link__destroy(link);
+	SCX_EQ(skel->data->uei.kind, EXIT_KIND(SCX_EXIT_UNREG));
+
+	return SCX_TEST_PASS;
+}
+
+static void cleanup(void *ctx)
+{
+	struct numa_nid *skel = ctx;
+
+	numa_nid__destroy(skel);
+}
+
+struct scx_test numa_nid = {
+	.name = "numa_nid",
+	.description = "Read a task's preferred NUMA node from BPF",
+	.setup = setup,
+	.run = run,
+	.cleanup = cleanup,
+};
+REGISTER_SCX_TEST(&numa_nid)
-- 
2.55.0


      parent reply	other threads:[~2026-10-04  7:29 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-04  7:27 [PATCHSET sched_ext/for-7.4] sched_ext: Add NUMA balancing support Andrea Righi
2026-10-04  7:27 ` [PATCH 1/5] sched/numa: Let other scheduling classes drive NUMA scanning Andrea Righi
2026-10-04  7:27 ` [PATCH 2/5] sched/numa: Leave the placement of a BPF-scheduled task to its scheduler Andrea Righi
2026-10-04  7:27 ` [PATCH 3/5] sched_ext: Scan NUMA hinting faults for opted-in BPF schedulers Andrea Righi
2026-10-04  7:27 ` [PATCH 4/5] sched_ext: Add scx_bpf_task_numa_nid() Andrea Righi
2026-10-04  7:27 ` Andrea Righi [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261004072901.3579967-6-arighi@nvidia.com \
    --to=arighi@nvidia.com \
    --cc=balbirs@nvidia.com \
    --cc=bsegall@google.com \
    --cc=changwoo@igalia.com \
    --cc=christian.loehle@arm.com \
    --cc=deliran@verdict.gg \
    --cc=dietmar.eggemann@arm.com \
    --cc=etsal@meta.com \
    --cc=juri.lelli@redhat.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=ltrager@nvidia.com \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=peterz@infradead.org \
    --cc=rostedt@goodmis.org \
    --cc=sched-ext@lists.linux.dev \
    --cc=tj@kernel.org \
    --cc=vincent.guittot@linaro.org \
    --cc=void@manifault.com \
    --cc=vschneid@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®