From: Kunwu Chan <kunwu.chan@gmail.com>
To: paulmck@kernel.org, corbet@lwn.net, mingo@redhat.com,
frederic@kernel.org, neeraj.upadhyay@kernel.org,
josh@joshtriplett.org, urezki@gmail.com, dave@stgolabs.net,
lianux.mm@gmail.com
Cc: stern@rowland.harvard.edu, parri.andrea@gmail.com,
will@kernel.org, peterz@infradead.org, boqun@kernel.org,
npiggin@gmail.com, dhowells@redhat.com, j.alglave@ucl.ac.uk,
luc.maranget@inria.fr, akiyks@gmail.com, dlustig@nvidia.com,
joelagnelf@nvidia.com, skhan@linuxfoundation.org,
rdunlap@infradead.org, longman@redhat.com, rostedt@goodmis.org,
mathieu.desnoyers@efficios.com, jiangshanlai@gmail.com,
qiang.zhang@linux.dev, kunwu.chan@gmail.com,
brads@mainlining.org, linux-kernel@vger.kernel.org,
linux-arch@vger.kernel.org, lkmm@lists.linux.dev,
linux-doc@vger.kernel.org, rcu@vger.kernel.org,
linux-kselftest@vger.kernel.org
Subject: [PATCH RFC v2 01/15] hazptr: add shared scan kthread
Date: Sat, 3 Oct 2026 01:08:33 +0800 [thread overview]
Message-ID: <20261002170847.3653663-2-kunwu.chan@gmail.com> (raw)
In-Reply-To: <20261002170847.3653663-1-kunwu.chan@gmail.com>
Batch concurrent hazptr_synchronize() callers into a shared scan
cycle, avoiding redundant wildcard flips and slot scans.
Queue waiters to a kthread and perform a two-phase wildcard scan
once for all queued waiters. After both phases, complete waiters
whose address is no longer held by any slot; waiters that remain
blocked are retried in a later cycle.
Waiters are embedded in the caller's stack frame, so no dynamic
allocation is needed in the synchronize path.
Fall back to the direct scan if the scan kthread is unavailable.
Adapted from the scan-kthread approach in Boqun Feng's shazptr
implementation.
Link: https://lore.kernel.org/lkml/20250625031101.12555-1-boqun.feng@gmail.com/
Signed-off-by: Kunwu Chan <kunwu.chan@gmail.com>
---
kernel/hazptr.c | 201 ++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 201 insertions(+)
diff --git a/kernel/hazptr.c b/kernel/hazptr.c
index d3d1050d92cf..9e274a691af5 100644
--- a/kernel/hazptr.c
+++ b/kernel/hazptr.c
@@ -12,6 +12,10 @@
#include <linux/mutex.h>
#include <linux/list.h>
#include <linux/export.h>
+#include <linux/completion.h>
+#include <linux/kthread.h>
+#include <linux/swait.h>
+#include <linux/sched.h>
/*
* The current hazard pointer wildcard. Flips between 1UL and 2UL to guarantee
@@ -209,12 +213,179 @@ void hazptr_scan_period(void *addr, void *scan_wildcard)
}
}
+struct hazptr_waiter {
+ struct list_head node;
+ void *addr;
+ struct completion done;
+};
+
+struct hazptr_scan_state {
+ struct task_struct *kthread;
+ struct swait_queue_head wq;
+ bool wakeup;
+ struct mutex lock;
+ struct list_head pending;
+ struct list_head scanning; /* kthread only */
+};
+static struct hazptr_scan_state hazptr_scan;
+
+/*
+ * Check per-CPU slots before overflow-list slots to match the
+ * acquisition ordering of promoted slots.
+ */
+static bool hazptr_value_present(void *val)
+{
+ int cpu;
+
+ for_each_possible_cpu(cpu) {
+ struct hazptr_percpu_slots *percpu_slots = per_cpu_ptr(&hazptr_percpu_slots, cpu);
+ struct hazptr_overflow_list_flip *overflow_list_flip = per_cpu_ptr(&percpu_overflow_list_flip, cpu);
+ unsigned int idx;
+
+ for (idx = 0; idx < NR_HAZPTR_PERCPU_SLOTS; idx++) {
+ struct hazptr_slot_item *item = &percpu_slots->items[idx];
+
+ /* Pairs with smp_store_release in hazptr_release(). */
+ if (smp_load_acquire(&item->slot.addr) == val)
+ return true;
+ }
+ for (int i = 0; i < 2; i++) {
+ struct hazptr_overflow_list *list = &overflow_list_flip->array[i];
+ struct hazptr_backup_slot *b;
+ unsigned long flags;
+
+ raw_spin_lock_irqsave(&list->lock, flags);
+ hlist_for_each_entry(b, &list->head, overflow_node) {
+ /* Pairs with smp_store_release in hazptr_release(). */
+ if (smp_load_acquire(&b->slot.addr) == val) {
+ raw_spin_unlock_irqrestore(&list->lock, flags);
+ return true;
+ }
+ }
+ raw_spin_unlock_irqrestore(&list->lock, flags);
+ }
+ }
+ return false;
+}
+
+/*
+ * Wait until no per-CPU slot or overflow-list slot holds @wc.
+ * Callers must ensure that the wildcard value in use by new acquires
+ * differs from @wc, so that the set of slots holding @wc only
+ * shrinks, which guarantees forward progress.
+ */
+static void hazptr_drain_wildcard(void *wc)
+{
+ while (hazptr_value_present(wc))
+ cond_resched();
+}
+
+/*
+ * Move pending waiters to ->scanning and perform a two-phase
+ * wildcard scan shared by all waiters.
+ */
+static void hazptr_scan_do_cycle(void)
+{
+ void *scan_wildcard, *old_wildcard;
+ struct hazptr_waiter *w, *n;
+ LIST_HEAD(done);
+
+ mutex_lock(&hazptr_wildcard_lock);
+
+ mutex_lock(&hazptr_scan.lock);
+ list_splice_tail_init(&hazptr_scan.pending, &hazptr_scan.scanning);
+ mutex_unlock(&hazptr_scan.lock);
+
+ if (list_empty(&hazptr_scan.scanning)) {
+ mutex_unlock(&hazptr_wildcard_lock);
+ return;
+ }
+
+ /* Pass 1: drain the unpublished wildcard. */
+ scan_wildcard = flip_wildcard(READ_ONCE(hazptr_wildcard));
+ hazptr_drain_wildcard(scan_wildcard);
+
+ /* Flip so new acquires use the new generation. */
+ WRITE_ONCE(hazptr_wildcard, scan_wildcard);
+ old_wildcard = flip_wildcard(scan_wildcard);
+
+ /* Pass 2: drain the old wildcard. */
+ hazptr_drain_wildcard(old_wildcard);
+
+ /* Complete waiters whose address is no longer held by any slot. */
+ list_for_each_entry_safe(w, n, &hazptr_scan.scanning, node) {
+ if (!hazptr_value_present(w->addr))
+ list_move(&w->node, &done);
+ }
+
+ mutex_unlock(&hazptr_wildcard_lock);
+
+ list_for_each_entry_safe(w, n, &done, node) {
+ list_del_init(&w->node);
+ complete(&w->done);
+ }
+}
+
+static int hazptr_scan_kthread(void *unused)
+{
+ for (;;) {
+ bool idle;
+
+ swait_event_idle_exclusive(hazptr_scan.wq,
+ READ_ONCE(hazptr_scan.wakeup));
+
+ hazptr_scan_do_cycle();
+
+ mutex_lock(&hazptr_scan.lock);
+ idle = list_empty(&hazptr_scan.pending) &&
+ list_empty(&hazptr_scan.scanning);
+ if (idle)
+ WRITE_ONCE(hazptr_scan.wakeup, false);
+ mutex_unlock(&hazptr_scan.lock);
+
+ if (idle)
+ continue;
+ /* Waiters still blocked: retry after a short delay. */
+ schedule_timeout_idle(1);
+ }
+ return 0;
+}
+
+/*
+ * Queue @addr for the shared scan. The waiter lives on the
+ * caller's stack, so no allocation is needed.
+ */
+static void hazptr_synchronize_queued(void *addr)
+{
+ struct hazptr_waiter waiter = {
+ .addr = addr,
+ };
+
+ init_completion(&waiter.done);
+ INIT_LIST_HEAD(&waiter.node);
+
+ /* Enqueue and wake the scan kthread. */
+ mutex_lock(&hazptr_scan.lock);
+ list_add_tail(&waiter.node, &hazptr_scan.pending);
+ if (!READ_ONCE(hazptr_scan.wakeup)) {
+ WRITE_ONCE(hazptr_scan.wakeup, true);
+ swake_up_one(&hazptr_scan.wq);
+ }
+ mutex_unlock(&hazptr_scan.lock);
+
+ /* Sleep until the scan kthread completes this waiter. */
+ wait_for_completion(&waiter.done);
+}
+
/*
* hazptr_synchronize: Wait until @addr is released from all slots.
*
* Wait to observe that each slot contains a value that differs from
* @addr before returning.
* Should be called from preemptible context.
+ *
+ * If available, queue the caller for a shared scan; otherwise use
+ * the direct scan path.
*/
void hazptr_synchronize(void *addr)
{
@@ -235,6 +406,13 @@ void hazptr_synchronize(void *addr)
/* Memory ordering: Store A before Load B. */
smp_mb();
+ /* Pairs with smp_store_release in hazptr_scan_init(). */
+ if (smp_load_acquire(&hazptr_scan.kthread)) {
+ hazptr_synchronize_queued(addr);
+ return;
+ }
+
+ /* Fallback: use the direct scan path. */
guard(mutex)(&hazptr_wildcard_lock);
scan_wildcard = flip_wildcard(hazptr_wildcard);
hazptr_scan_period(addr, scan_wildcard);
@@ -282,3 +460,26 @@ void __init hazptr_init(void)
}
}
}
+
+/*
+ * Initialize the scan kthread. Failure falls back to the
+ * direct scan.
+ */
+static int __init hazptr_scan_init(void)
+{
+ struct task_struct *t;
+
+ init_swait_queue_head(&hazptr_scan.wq);
+ mutex_init(&hazptr_scan.lock);
+ INIT_LIST_HEAD(&hazptr_scan.pending);
+ INIT_LIST_HEAD(&hazptr_scan.scanning);
+
+ t = kthread_run(hazptr_scan_kthread, NULL, "hazptr_scan");
+ if (!IS_ERR(t))
+ /* Pairs with smp_load_acquire in hazptr_synchronize(). */
+ smp_store_release(&hazptr_scan.kthread, t);
+ else
+ pr_warn("hazptr: scan kthread failed, using direct scan\n");
+ return 0;
+}
+core_initcall(hazptr_scan_init);
--
2.43.0
next prev parent reply other threads:[~2026-10-02 17:09 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-02 17:08 [PATCH RFC v2 00/15] hazptr: batch synchronize operations through a shared scan Kunwu Chan
2026-10-02 17:08 ` Kunwu Chan [this message]
2026-10-02 17:08 ` [PATCH RFC v2 02/15] hazptr: use Bloom filter for shared scan waiters Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 03/15] hazptr: scan all per-CPU slots before overflow lists Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 04/15] hazptr: add scoped_guard() support Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 05/15] hazptr: add debug option to force the acquire slow path Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 06/15] hazptr: elide redundant first drain pass Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 07/15] Documentation/litmus-tests: add hazptr wildcard-flip escape test Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 08/15] locking/lockdep: use hazptr to wait for dynamic key lookups Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 09/15] rcuscale: add hazptr scale type Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 10/15] hazptr: fix kernel-doc of hazptr_release() Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 11/15] Documentation/litmus-tests: add hazptr acquire-before-scan test Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 12/15] hazptrtorture: add slowpath and lockdep scenarios Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 13/15] hazptrtorture: add READERS4 and READERS0 torture configs Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 14/15] hazptrtorture: add 128- and 256-CPU configs Kunwu Chan
2026-10-02 17:08 ` [PATCH RFC v2 15/15] selftests/rcutorture: add hazptr torture test script Kunwu Chan
2026-10-02 17:12 ` [PATCH RFC v2 00/15] hazptr: batch synchronize operations through a shared scan Bradley Morgan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261002170847.3653663-2-kunwu.chan@gmail.com \
--to=kunwu.chan@gmail.com \
--cc=akiyks@gmail.com \
--cc=boqun@kernel.org \
--cc=brads@mainlining.org \
--cc=corbet@lwn.net \
--cc=dave@stgolabs.net \
--cc=dhowells@redhat.com \
--cc=dlustig@nvidia.com \
--cc=frederic@kernel.org \
--cc=j.alglave@ucl.ac.uk \
--cc=jiangshanlai@gmail.com \
--cc=joelagnelf@nvidia.com \
--cc=josh@joshtriplett.org \
--cc=lianux.mm@gmail.com \
--cc=linux-arch@vger.kernel.org \
--cc=linux-doc@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=lkmm@lists.linux.dev \
--cc=longman@redhat.com \
--cc=luc.maranget@inria.fr \
--cc=mathieu.desnoyers@efficios.com \
--cc=mingo@redhat.com \
--cc=neeraj.upadhyay@kernel.org \
--cc=npiggin@gmail.com \
--cc=parri.andrea@gmail.com \
--cc=paulmck@kernel.org \
--cc=peterz@infradead.org \
--cc=qiang.zhang@linux.dev \
--cc=rcu@vger.kernel.org \
--cc=rdunlap@infradead.org \
--cc=rostedt@goodmis.org \
--cc=skhan@linuxfoundation.org \
--cc=stern@rowland.harvard.edu \
--cc=urezki@gmail.com \
--cc=will@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®