mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Suleiman Souhlal <suleiman@google.com>
To: linux-kernel@vger.kernel.org
Cc: "Suleiman Souhlal" <suleiman@google.com>,
	"Thomas Gleixner" <tglx@kernel.org>,
	"Ingo Molnar" <mingo@redhat.com>,
	"Peter Zijlstra" <peterz@infradead.org>,
	"Darren Hart" <dvhart@infradead.org>,
	"Davidlohr Bueso" <dave@stgolabs.net>,
	"André Almeida" <andrealmeid@igalia.com>,
	"Juri Lelli" <juri.lelli@redhat.com>,
	"Vincent Guittot" <vincent.guittot@linaro.org>,
	"Dietmar Eggemann" <dietmar.eggemann@arm.com>,
	"Steven Rostedt" <rostedt@goodmis.org>,
	"Ben Segall" <bsegall@google.com>, "Mel Gorman" <mgorman@suse.de>,
	"Valentin Schneider" <vschneid@redhat.com>,
	"K Prateek Nayak" <kprateek.nayak@amd.com>,
	"zhidao su" <soolaugust@gmail.com>,
	"John Stultz" <jstultz@google.com>,
	"Qais Yousef" <qyousef@google.com>,
	ssouhlal@FreeBSD.org
Subject: [RFC PATCH 12/12] tools/testing/futex: Add ping_bench, a tool for benchmarking futexes.
Date: Thu, 17 Sep 2026 04:33:36 +0000	[thread overview]
Message-ID: <20260917043339.2093426-13-suleiman@google.com> (raw)
In-Reply-To: <20260917043339.2093426-1-suleiman@google.com>

It can generate metrics across a number of configurations, measuring
how long locking takes.

It prints out lock call durations for the foreground thread to
stdout (unless -q is passed), for analysis with other tools such as
ministat or turning into histograms.

Some parameters:
    -a:  Print out durations for all threads instead of only for the
         foreground thread.
    -q:  Don't print out locking durations.
    -f:  Type of futex ("f" FUTEX_WAIT, "p" FUTEX_LOCK_PI,
         "n" FUTEX_LOCK_PING, "N" FUTEX_LOCK_PING with userspace stealing,
         "m" pthread_mutex_t).
    -t:  Number of threads acquiring/releasing the lock.
    -n:  Number of iterations the threads will take the lock.
    -s:  How frequently a thread tries to take the lock.
    -S:  Duration in usec to sleep instead of spinning after unlocking
         (use instead of -s).
    -w:  Lock hold time.
    -b:  Number of “busy” cpu spinner threads.
    -r:  Number of threads using RT prio.
    -p:  Use nice() biasing (foreground thread gets more cpu time,
         background threads get less - allows for priority inversions).

Co-developed-by: John Stultz <jstultz@google.com>
Signed-off-by: John Stultz <jstultz@google.com>
Signed-off-by: Suleiman Souhlal <suleiman@google.com>
---
 tools/testing/futex/Makefile     |  13 +
 tools/testing/futex/ping_bench.c | 428 +++++++++++++++++++++++++++++++
 2 files changed, 441 insertions(+)
 create mode 100644 tools/testing/futex/Makefile
 create mode 100644 tools/testing/futex/ping_bench.c

diff --git a/tools/testing/futex/Makefile b/tools/testing/futex/Makefile
new file mode 100644
index 000000000000..cbb2deb30923
--- /dev/null
+++ b/tools/testing/futex/Makefile
@@ -0,0 +1,13 @@
+# SPDX-License-Identifier: GPL-2.0
+
+.PHONY: clean
+
+TARGETS = ping_bench
+CFLAGS = -O -Wall -g
+OFILES = ping_bench.o
+TARGETS = ping_bench
+
+ping_bench: $(OFILES)
+
+clean:
+	$(RM) $(TARGETS) $(OFILES)
diff --git a/tools/testing/futex/ping_bench.c b/tools/testing/futex/ping_bench.c
new file mode 100644
index 000000000000..418e147174d0
--- /dev/null
+++ b/tools/testing/futex/ping_bench.c
@@ -0,0 +1,428 @@
+// SPDX-License-Identifier: GPL-2.0-only
+
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <stdatomic.h>
+#include <string.h>
+#include <stdbool.h>
+#include <err.h>
+#include <unistd.h>
+#include <stdint.h>
+#include <errno.h>
+#include <fcntl.h>
+#include <string.h>
+#include <time.h>
+#include <stdatomic.h>
+#include <stdarg.h>
+#include <pthread.h>
+#include <limits.h>
+#include <linux/futex.h>
+#include <sys/prctl.h>
+#include <sys/syscall.h>
+#include <sys/time.h>
+
+#define	MAX_THR 512
+
+#define	FUTEX_LOCK_PING		14
+#define	FUTEX_UNLOCK_PING	15
+
+#define	READ_ONCE(x) (*(volatile typeof(x) *)&(x))
+
+struct thread {
+	pthread_t pthr;
+	uint64_t *dur;
+	int id;
+};
+
+static struct thread bthr[MAX_THR];
+static struct thread thr[MAX_THR];
+static pthread_barrier_t bar;
+
+uint32_t _lock;
+pthread_mutex_t mtx = PTHREAD_MUTEX_INITIALIZER;
+void *lock = &_lock;
+
+static long counter;
+static int num_thr;
+static int num_rt;
+static int busy_thr;
+static uint64_t num_iter = 10000;
+static  __thread pid_t tid;
+static int work_times = 100000;
+static uint64_t sleep_dur = 1000;
+static bool do_sleep;
+static bool print_all;
+static bool quiet;
+static bool bias;
+
+enum futex_type {
+	FUTEX,
+	FUTEX_PI,
+	FUTEX_PING,
+	FUTEX_PING_USTEAL,
+	MUTEX,
+};
+static enum futex_type futex_type = FUTEX_PING;
+
+extern char *optarg;
+extern int optind;
+
+int trace_marker_fd;
+
+static void
+init_trace_marker(void)
+{
+	trace_marker_fd = open("/sys/kernel/tracing/trace_marker", O_WRONLY);
+	if (trace_marker_fd < 0)
+		perror("Failed to open trace_marker");
+}
+
+static int
+write_trace_marker(const char *format, ...)
+{
+	char buffer[256];
+	va_list args;
+	int len;
+
+	if (trace_marker_fd <= 0)
+		return -1;
+
+	va_start(args, format);
+	len = vsnprintf(buffer, sizeof(buffer), format, args);
+	va_end(args);
+
+	if (len > 0)
+		write(trace_marker_fd, buffer, len);
+
+	return (0);
+}
+
+static inline uint64_t
+now_ns(void)
+{
+	struct timespec ts;
+
+	if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0)
+		err(1, "clock_gettime");
+
+	return (ts.tv_sec * 1000000000UL + ts.tv_nsec);
+}
+
+static int
+futex(uint32_t *uaddr, int op, uint32_t val, struct timespec *to)
+{
+	return (syscall(SYS_futex, uaddr, op, val, to));
+}
+
+static void
+futex_lock(void *lock)
+{
+	pthread_mutex_t *mu;
+	uint32_t *fu, old;
+	int ret;
+
+	fu = lock;
+	mu = lock;
+	switch (futex_type) {
+	case FUTEX:
+		while (1) {
+			old = 0;
+			if (atomic_compare_exchange_strong(fu, &old, 1))
+				return;
+			if (futex(fu, FUTEX_WAIT, 1, NULL) != 0 && errno !=
+			    EAGAIN)
+				err(1, "FUTEX_WAIT");
+		}
+		break;
+	case FUTEX_PI:
+		old = 0;
+		if (atomic_compare_exchange_strong(fu, &old, tid))
+			return;
+		if ((ret = futex(fu, FUTEX_LOCK_PI, 0, NULL)) != 0)
+			errx(1, "FUTEX_LOCK_PI %s", strerror(ret));
+		break;
+	case FUTEX_PING:
+	case FUTEX_PING_USTEAL:
+		old = 0;
+		if (atomic_compare_exchange_strong(fu, &old, tid))
+			return;
+		old = FUTEX_WAITERS;
+		if (futex_type == FUTEX_PING_USTEAL &&
+		    atomic_compare_exchange_strong(fu, &old, tid |
+		    FUTEX_WAITERS))
+			return;
+		if ((ret = futex(fu, FUTEX_LOCK_PING, 0, NULL)) != 0)
+			err(1, "FUTEX_LOCK_PING");
+		break;
+	case MUTEX:
+		if (pthread_mutex_lock(mu) < 0)
+			err(1, "pthread_mutex_lock");
+		break;
+	}
+}
+
+static void
+futex_unlock(void *lock)
+{
+	pthread_mutex_t *mu;
+	uint32_t *fu, old;
+
+	fu = lock;
+	mu = lock;
+	switch (futex_type) {
+	case FUTEX:
+		old = 1;
+		if (atomic_compare_exchange_strong(fu, &old, 0))
+			if (futex(fu, FUTEX_WAKE, 1, NULL) < 0)
+				err(1, "FUTEX_WAKE");
+		break;
+	case FUTEX_PI:
+		old = tid;
+		if (atomic_compare_exchange_strong(fu, &old, 0))
+			return;
+		if (futex(fu, FUTEX_UNLOCK_PI, 0, NULL) != 0)
+			err(1, "FUTEX_UNLOCK_PI t %x old %x", tid,
+			    READ_ONCE(*fu));
+		break;
+	case FUTEX_PING:
+	case FUTEX_PING_USTEAL:
+		old = tid;
+		if (atomic_compare_exchange_strong(fu, &old, 0))
+			return;
+		if (futex(fu, FUTEX_UNLOCK_PING, 0, NULL) != 0)
+			err(1, "FUTEX_UNLOCK_PING t %x old %x", tid,
+			    READ_ONCE(*fu));
+		break;
+	case MUTEX:
+		if (pthread_mutex_unlock(mu) < 0)
+			err(1, "pthread_mutex_unlock");
+		break;
+	}
+}
+
+atomic_int stop_spinners = 0;
+
+static void *
+func(void *p)
+{
+	struct thread *thr;
+	uint64_t end, start;
+	uint64_t i, j, _num_iter;
+	uint64_t my_sleep_dur = sleep_dur;
+	uint64_t my_work_times = work_times;
+	struct timespec ts;
+
+	tid = gettid();
+
+	thr = p;
+	thr->dur = malloc(num_iter * sizeof(uint64_t));
+	_num_iter = num_iter;
+
+	if (thr->id == 0) {
+		prctl(PR_SET_NAME, "foreground", 0, 0, 0);
+		if (bias) {
+			nice(-5);
+			my_sleep_dur *= 10;
+			if (my_work_times)
+				my_work_times /= 10;
+		}
+	} else {
+		prctl(PR_SET_NAME, "background", 0, 0, 0);
+		if (bias) {
+			if (my_sleep_dur)
+				my_sleep_dur /= 10;
+			my_work_times *= 10;
+			nice(19);
+		}
+	}
+
+	ts.tv_sec  = my_sleep_dur / 1000000;
+	ts.tv_nsec = (my_sleep_dur % 1000000) * 1000;
+
+
+	if (thr->id < num_rt) {
+		struct sched_param param = { .sched_priority = 10 };
+
+		if (sched_setscheduler(0, SCHED_FIFO, &param) != 0)
+			err(1, "sched_setscheduler");
+	}
+
+	pthread_barrier_wait(&bar);
+
+	for (i = 0; i < _num_iter; i++) {
+		if (thr->id == 0)
+			write_trace_marker("B|%lu|Locking", (unsigned long)tid);
+
+		start = now_ns();
+		futex_lock(lock);
+		end = now_ns();
+
+		if (thr->id == 0)
+			write_trace_marker("E|%lu|Locking", (unsigned long)tid);
+		thr->dur[i] = end - start;
+
+		for (j = 0; j < my_work_times; j++)
+			__asm __volatile("" ::: "memory");
+
+		counter++;
+		futex_unlock(lock);
+
+		if (atomic_load(&stop_spinners))
+			break;
+
+		if (sleep_dur) {
+			if (do_sleep)
+				clock_nanosleep(CLOCK_MONOTONIC, 0, &ts, 0);
+			else
+				for (j = 0; j < my_sleep_dur; j++)
+					__asm __volatile("" ::: "memory");
+		}
+	}
+
+	if (thr->id == 0)
+		atomic_store(&stop_spinners, 1);
+	return (NULL);
+}
+
+static void *
+busy(void *p)
+{
+	prctl(PR_SET_NAME, "spinner", 0, 0, 0);
+	if (0 &&  num_rt) {
+		struct sched_param param = { .sched_priority = 1 };
+
+		if (sched_setscheduler(0, SCHED_FIFO, &param) != 0)
+			err(1, "sched_setscheduler");
+	}
+
+	pthread_barrier_wait(&bar);
+
+	while (!atomic_load(&stop_spinners))
+		__asm __volatile("" ::: "memory");
+
+	return (NULL);
+}
+
+
+static void
+usage(char *a)
+{
+	fprintf(stderr, "Usage: %s [-a] [-b n] [-c] [-f f/m/n/N/p] [-n n] [-q]"
+	    " [-r n] [-s n] [-S n] [-t n] [-w n]\n", a);
+	exit(1);
+}
+
+int
+main(int argc, char **argv)
+{
+	int c, i, j, ret;
+
+	while ((c = getopt(argc, argv, "ab:f:n:pqr:S:s:t:w:")) != -1) {
+		switch (c) {
+		case 'a':
+			print_all = 1;
+			break;
+		case 'f':
+			switch (*optarg) {
+			case 'f':
+				futex_type = FUTEX;
+				break;
+			case 'm':
+				futex_type = MUTEX;
+				lock = &mtx;
+				break;
+			case 'n':
+				futex_type = FUTEX_PING;
+				break;
+			case 'N':
+				futex_type = FUTEX_PING_USTEAL;
+				break;
+			case 'p':
+				futex_type = FUTEX_PI;
+				break;
+			default:
+				usage(argv[0]);
+			}
+			break;
+		case 'n':
+			num_iter = atoi(optarg);
+			break;
+		case 'q':
+			quiet = 1;
+			break;
+		case 'p':
+			bias = 1;
+			break;
+		case 'r':
+			num_rt = atoi(optarg);
+			break;
+		case 'S':
+			do_sleep = 1;
+			/* Fallthrough */
+		case 's':
+			sleep_dur = atoi(optarg);
+			break;
+		case 't':
+			num_thr = atoi(optarg);
+			break;
+		case 'b':
+			busy_thr = atoi(optarg);
+			break;
+		case 'w':
+			work_times = atoi(optarg);
+			break;
+		default:
+			usage(argv[0]);
+		}
+	}
+
+	init_trace_marker();
+
+	if (num_thr < num_rt)
+		num_thr = num_rt;
+
+	num_thr -= num_rt;
+	if (num_thr > MAX_THR)
+		num_thr = MAX_THR;
+
+	tid = gettid();
+
+	ret = pthread_barrier_init(&bar, NULL, num_thr + num_rt + busy_thr);
+	if (ret != 0)
+		errx(1, "pthread_barrier_init %s", strerror(ret));
+
+	for (i = 0; i < num_thr + num_rt; i++) {
+		thr[i].id = i;
+		if ((ret = pthread_create(&thr[i].pthr, NULL, func, &thr[i]))
+		    != 0)
+			errx(1, "pthread_create %d: %s", i, strerror(ret));
+	}
+	
+	for (i = 0; i < busy_thr; i++) {
+		if ((ret = pthread_create(&bthr[i].pthr, NULL, busy, &bthr[i]))
+		    != 0)
+			errx(1, "pthread_create %d: %s", i, strerror(ret));
+	}
+
+	for (i = 0; i < busy_thr; i++)
+		if ((ret = pthread_join(bthr[i].pthr, NULL)) != 0)
+			errx(1, "pthread_join: %s\n", strerror(ret));
+
+	for (i = 0; i < num_thr + num_rt; i++)
+		if ((ret = pthread_join(thr[i].pthr, NULL)) != 0)
+			errx(1, "pthread_join: %s\n", strerror(ret));
+
+	if (!quiet) {
+		/* skip the first run */
+		if (print_all)
+			for (i = 0; i < num_thr; i++)
+				for (j = 1; j < num_iter; j++)
+					printf("%lu\n", thr[0].dur[j]);
+		else
+			for (j = 1; j < num_iter; j++)
+				printf("%lu\n", thr[0].dur[j]);
+	}
+
+	return (0);
+}
-- 
2.55.0.1082.g2b9226bbc0-goog


  parent reply	other threads:[~2026-09-17  4:34 UTC|newest]

Thread overview: 20+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-17  4:33 [RFC PATCH 00/12] FUTEX_PING: A stealable futex using Proxy Execution Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 01/12] sched: Abstract task_struct->blocked_on by locking primitive Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 02/12] futex: Switch PI futex to use p->pi_futex_lock instead of p->pi_lock Suleiman Souhlal
2026-09-17 15:38   ` Peter Zijlstra
2026-09-17  4:33 ` [RFC PATCH 03/12] futex: Add "ping" parameter to pi_state management functions and export them Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 04/12] futex: Introduce stealable PI futex, FUTEX_*_PING Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 05/12] futex: Implement exit_ping_state_list() Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 06/12] futex: Address aborting from futex_lock_ping() while owning ping_state Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 07/12] futex: Make FUTEX_*_PING use Proxy Execution Suleiman Souhlal
2026-09-17 13:18   ` Jihan LIN
2026-09-17 14:39     ` K Prateek Nayak
2026-09-17 15:36       ` Peter Zijlstra
2026-09-17  4:33 ` [RFC PATCH 08/12] futex: Implement PING futex handoff Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 09/12] futex: Wake up donor in PING futex unlock Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 10/12] futex: Optimistic spinning for PING futexes Suleiman Souhlal
2026-09-17  4:33 ` [RFC PATCH 11/12] futex: Allow userspace stealing " Suleiman Souhlal
2026-09-17  4:33 ` Suleiman Souhlal [this message]
2026-09-17  8:58 ` [RFC PATCH 00/12] FUTEX_PING: A stealable futex using Proxy Execution Peter Zijlstra
2026-09-17 17:53   ` John Stultz
2026-09-17 18:51     ` Steven Rostedt

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260917043339.2093426-13-suleiman@google.com \
    --to=suleiman@google.com \
    --cc=andrealmeid@igalia.com \
    --cc=bsegall@google.com \
    --cc=dave@stgolabs.net \
    --cc=dietmar.eggemann@arm.com \
    --cc=dvhart@infradead.org \
    --cc=jstultz@google.com \
    --cc=juri.lelli@redhat.com \
    --cc=kprateek.nayak@amd.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=peterz@infradead.org \
    --cc=qyousef@google.com \
    --cc=rostedt@goodmis.org \
    --cc=soolaugust@gmail.com \
    --cc=ssouhlal@FreeBSD.org \
    --cc=tglx@kernel.org \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®