From: "Masami Hiramatsu (Google)" <mhiramat@kernel.org>
To: Steven Rostedt <rostedt@goodmis.org>,
Peter Zijlstra <peterz@infradead.org>,
Ingo Molnar <mingo@kernel.org>,
x86@kernel.org
Cc: Jinchao Wang <wangjinchao600@gmail.com>,
Mathieu Desnoyers <mathieu.desnoyers@efficios.com>,
Masami Hiramatsu <mhiramat@kernel.org>,
Thomas Gleixner <tglx@linutronix.de>,
Borislav Petkov <bp@alien8.de>,
Dave Hansen <dave.hansen@linux.intel.com>,
"H . Peter Anvin" <hpa@zytor.com>,
Alexander Shishkin <alexander.shishkin@linux.intel.com>,
Ian Rogers <irogers@google.com>,
linux-kernel@vger.kernel.org, linux-trace-kernel@vger.kernel.org,
linux-doc@vger.kernel.org, linux-perf-users@vger.kernel.org
Subject: [PATCH v14 04/14] x86/hw_breakpoints: Make DR7 updates NMI safe
Date: Sun, 30 Aug 2026 23:27:34 +0900 [thread overview]
Message-ID: <178810005476.64882.15902865504188093281.stgit@devnote2> (raw)
In-Reply-To: <178810001186.64882.2161016469449127450.stgit@devnote2>
From: Jinchao Wang <wangjinchao600@gmail.com>
Hardware breakpoint installation and removal run with IRQs disabled, but
an NMI can still enter the same code through KGDB. The interrupted
operation and the NMI can consequently claim the same slot or overwrite
each other's DR7 state.
Claim and release per-CPU slots with cmpxchg. Update cpu_dr7 with
single-instruction per-CPU operations, and preserve hardware-first
disable and hardware-last enable ordering. Add a per-CPU sequence number
so interrupted DR7 writers and restore paths detect an NMI update and
retry from the latest shadow state.
Link: https://lore.kernel.org/all/4ee0a2efc9e8387af83286b8495b7d490247e165.1785067572.git.wangjinchao600@gmail.com/
Signed-off-by: Jinchao Wang <wangjinchao600@gmail.com>
Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>
---
Changes in v14:
- In local_db_restore(), return immediately if saved dr7 is 0 to avoid
overwriting intentionally cleared hardware DR7 state (e.g. KVM guest
entry) with cpu_dr7 when an NMI increments cpu_dr7_seq.
Changes in v13:
- Add compiler barriers (barrier()) before checking cpu_dr7_seq to
prevent the compiler from hoisting the sequence check before debug
register updates or shadow state modifications.
Changes in v12:
- Use stack variables dr7 and dr7_seq directly in exc_nmi() and remove
unnecessary per-CPU nmi_dr7 and nmi_dr7_seq variables.
---
arch/x86/include/asm/debugreg.h | 37 ++++++++++----
arch/x86/kernel/cpu/mce/core.c | 10 ++--
arch/x86/kernel/hw_breakpoint.c | 101 ++++++++++++++++++++-------------------
arch/x86/kernel/nmi.c | 7 ++-
arch/x86/kernel/traps.c | 10 +++-
5 files changed, 95 insertions(+), 70 deletions(-)
diff --git a/arch/x86/include/asm/debugreg.h b/arch/x86/include/asm/debugreg.h
index a2c1f2d24b64..65e80cd62193 100644
--- a/arch/x86/include/asm/debugreg.h
+++ b/arch/x86/include/asm/debugreg.h
@@ -18,6 +18,7 @@
#define DR7_FIXED_1 0x00000400
DECLARE_PER_CPU(unsigned long, cpu_dr7);
+DECLARE_PER_CPU(unsigned int, cpu_dr7_seq);
#ifndef CONFIG_PARAVIRT_XXL
/*
@@ -125,18 +126,20 @@ static __always_inline bool hw_breakpoint_active(void)
extern void hw_breakpoint_restore(void);
-static __always_inline unsigned long local_db_save(void)
+static __always_inline void local_db_save(unsigned long *dr7,
+ unsigned int *dr7_seq)
{
- unsigned long dr7;
+ *dr7 = 0;
+ *dr7_seq = this_cpu_read(cpu_dr7_seq);
if (static_cpu_has(X86_FEATURE_HYPERVISOR) && !hw_breakpoint_active())
- return 0;
+ return;
- get_debugreg(dr7, 7);
+ get_debugreg(*dr7, 7);
/* Architecturally set bit */
- dr7 &= ~DR7_FIXED_1;
- if (dr7)
+ *dr7 &= ~DR7_FIXED_1;
+ if (*dr7)
set_debugreg(DR7_FIXED_1, 7);
/*
@@ -145,20 +148,34 @@ static __always_inline unsigned long local_db_save(void)
* be good.
*/
barrier();
-
- return dr7;
}
-static __always_inline void local_db_restore(unsigned long dr7)
+static __always_inline void local_db_restore(unsigned long dr7,
+ unsigned int dr7_seq)
{
+ unsigned int seq;
+
/*
* Ensure the compiler doesn't raise this statement into
* the critical section; enabling breakpoints early would
* not be good.
*/
barrier();
- if (dr7)
+
+ if (!dr7)
+ return;
+
+ do {
+ seq = this_cpu_read(cpu_dr7_seq);
+ if (seq != dr7_seq) {
+ dr7 = this_cpu_read(cpu_dr7);
+ if (!dr7)
+ dr7 = DR7_FIXED_1;
+ }
+
set_debugreg(dr7, 7);
+ barrier();
+ } while (unlikely(seq != this_cpu_read(cpu_dr7_seq)));
}
#ifdef CONFIG_CPU_SUP_AMD
diff --git a/arch/x86/kernel/cpu/mce/core.c b/arch/x86/kernel/cpu/mce/core.c
index 61ade8b5c9d7..d45e8b780deb 100644
--- a/arch/x86/kernel/cpu/mce/core.c
+++ b/arch/x86/kernel/cpu/mce/core.c
@@ -2109,6 +2109,7 @@ static __always_inline void exc_machine_check_kernel(struct pt_regs *regs)
{
irqentry_state_t irq_state;
unsigned long dr7;
+ unsigned int dr7_seq;
WARN_ON_ONCE(user_mode(regs));
@@ -2119,24 +2120,25 @@ static __always_inline void exc_machine_check_kernel(struct pt_regs *regs)
if (mca_cfg.initialized && mce_check_crashing_cpu())
return;
- dr7 = local_db_save();
+ local_db_save(&dr7, &dr7_seq);
irq_state = irqentry_nmi_enter(regs);
do_machine_check(regs);
irqentry_nmi_exit(regs, irq_state);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
static __always_inline void exc_machine_check_user(struct pt_regs *regs)
{
unsigned long dr7;
+ unsigned int dr7_seq;
irqentry_enter_from_user_mode(regs);
- dr7 = local_db_save();
+ local_db_save(&dr7, &dr7_seq);
do_machine_check(regs);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
irqentry_exit_to_user_mode(regs);
}
diff --git a/arch/x86/kernel/hw_breakpoint.c b/arch/x86/kernel/hw_breakpoint.c
index f846c15f21ca..3204e75dc454 100644
--- a/arch/x86/kernel/hw_breakpoint.c
+++ b/arch/x86/kernel/hw_breakpoint.c
@@ -40,6 +40,9 @@
DEFINE_PER_CPU(unsigned long, cpu_dr7);
EXPORT_PER_CPU_SYMBOL(cpu_dr7);
+/* Sequence number of the per-CPU DR7 state. */
+DEFINE_PER_CPU(unsigned int, cpu_dr7_seq);
+
/* Per cpu debug address registers values */
static DEFINE_PER_CPU(unsigned long, cpu_debugreg[HBP_NUM]);
@@ -97,38 +100,31 @@ int decode_dr7(unsigned long dr7, int bpnum, unsigned *len, unsigned *type)
int arch_install_hw_breakpoint(struct perf_event *bp)
{
struct arch_hw_breakpoint *info = counter_arch_bp(bp);
- unsigned long *dr7;
+ unsigned int seq;
int i;
lockdep_assert_irqs_disabled();
for (i = 0; i < HBP_NUM; i++) {
- struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
-
- if (!*slot) {
- *slot = bp;
+ if (!this_cpu_cmpxchg(bp_per_reg[i], NULL, bp))
break;
- }
}
if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
return -EBUSY;
- set_debugreg(info->address, i);
- __this_cpu_write(cpu_debugreg[i], info->address);
-
- dr7 = this_cpu_ptr(&cpu_dr7);
- *dr7 |= encode_dr7(i, info->len, info->type);
-
- /*
- * Ensure we first write cpu_dr7 before we set the DR7 register.
- * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
- */
- barrier();
-
- set_debugreg(*dr7, 7);
- if (info->mask)
- amd_set_dr_addr_mask(info->mask, i);
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ this_cpu_write(cpu_debugreg[i], info->address);
+ barrier();
+ set_debugreg(info->address, i);
+ if (info->mask)
+ amd_set_dr_addr_mask(info->mask, i);
+ this_cpu_or(cpu_dr7, encode_dr7(i, info->len, info->type));
+ barrier();
+ set_debugreg(this_cpu_read(cpu_dr7), 7);
+ barrier();
+ } while (seq != this_cpu_read(cpu_dr7_seq));
return 0;
}
@@ -146,36 +142,34 @@ void arch_uninstall_hw_breakpoint(struct perf_event *bp)
{
struct arch_hw_breakpoint *info = counter_arch_bp(bp);
unsigned long dr7;
+ unsigned int seq;
int i;
lockdep_assert_irqs_disabled();
for (i = 0; i < HBP_NUM; i++) {
- struct perf_event **slot = this_cpu_ptr(&bp_per_reg[i]);
-
- if (*slot == bp) {
- *slot = NULL;
+ if (this_cpu_read(bp_per_reg[i]) == bp)
break;
- }
}
if (WARN_ONCE(i == HBP_NUM, "Can't find any breakpoint slot"))
return;
- dr7 = this_cpu_read(cpu_dr7);
- dr7 &= ~__encode_dr7(i, info->len, info->type);
-
- set_debugreg(dr7, 7);
- if (info->mask)
- amd_set_dr_addr_mask(0, i);
-
- /*
- * Ensure the write to cpu_dr7 is after we've set the DR7 register.
- * This ensures an NMI never see cpu_dr7 0 when DR7 is not.
- */
- barrier();
-
- this_cpu_write(cpu_dr7, dr7);
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ dr7 = this_cpu_read(cpu_dr7);
+ dr7 &= ~__encode_dr7(i, info->len, info->type);
+ set_debugreg(dr7, 7);
+ if (info->mask)
+ amd_set_dr_addr_mask(0, i);
+ barrier();
+ this_cpu_and(cpu_dr7,
+ ~__encode_dr7(i, info->len, info->type));
+ barrier();
+ } while (seq != this_cpu_read(cpu_dr7_seq));
+
+ WARN_ONCE(this_cpu_cmpxchg(bp_per_reg[i], bp, NULL) != bp,
+ "Can't release breakpoint slot");
}
static int arch_bp_generic_len(int x86_len)
@@ -309,13 +303,14 @@ static inline bool within_cpu_entry(unsigned long addr, unsigned long end)
sizeof(struct tlb_state)))
return true;
- /*
- * When in guest (X86_FEATURE_HYPERVISOR), local_db_save()
- * will read per-cpu cpu_dr7 before clear dr7 register.
- */
+ /* local_db_save() reads this state before clearing DR7. */
if (within_area(addr, end, (unsigned long)&per_cpu(cpu_dr7, cpu),
sizeof(cpu_dr7)))
return true;
+ if (within_area(addr, end,
+ (unsigned long)&per_cpu(cpu_dr7_seq, cpu),
+ sizeof(cpu_dr7_seq)))
+ return true;
}
return false;
@@ -483,12 +478,18 @@ void flush_ptrace_hw_breakpoint(struct task_struct *tsk)
void hw_breakpoint_restore(void)
{
- set_debugreg(__this_cpu_read(cpu_debugreg[0]), 0);
- set_debugreg(__this_cpu_read(cpu_debugreg[1]), 1);
- set_debugreg(__this_cpu_read(cpu_debugreg[2]), 2);
- set_debugreg(__this_cpu_read(cpu_debugreg[3]), 3);
- set_debugreg(DR6_RESERVED, 6);
- set_debugreg(__this_cpu_read(cpu_dr7), 7);
+ unsigned int seq;
+
+ do {
+ seq = this_cpu_inc_return(cpu_dr7_seq);
+ set_debugreg(this_cpu_read(cpu_debugreg[0]), 0);
+ set_debugreg(this_cpu_read(cpu_debugreg[1]), 1);
+ set_debugreg(this_cpu_read(cpu_debugreg[2]), 2);
+ set_debugreg(this_cpu_read(cpu_debugreg[3]), 3);
+ set_debugreg(DR6_RESERVED, 6);
+ set_debugreg(this_cpu_read(cpu_dr7), 7);
+ barrier();
+ } while (seq != this_cpu_read(cpu_dr7_seq));
}
EXPORT_SYMBOL_FOR_KVM(hw_breakpoint_restore);
diff --git a/arch/x86/kernel/nmi.c b/arch/x86/kernel/nmi.c
index 3c9f60d6ca5a..c3806f7bd3e8 100644
--- a/arch/x86/kernel/nmi.c
+++ b/arch/x86/kernel/nmi.c
@@ -531,11 +531,12 @@ enum nmi_states {
};
static DEFINE_PER_CPU(enum nmi_states, nmi_state);
static DEFINE_PER_CPU(unsigned long, nmi_cr2);
-static DEFINE_PER_CPU(unsigned long, nmi_dr7);
DEFINE_IDTENTRY_RAW(exc_nmi)
{
irqentry_state_t irq_state;
+ unsigned long dr7;
+ unsigned int dr7_seq;
struct nmi_stats *nsp = this_cpu_ptr(&nmi_stats);
/*
@@ -572,7 +573,7 @@ DEFINE_IDTENTRY_RAW(exc_nmi)
*/
sev_es_ist_enter(regs);
- this_cpu_write(nmi_dr7, local_db_save());
+ local_db_save(&dr7, &dr7_seq);
irq_state = irqentry_nmi_enter(regs);
@@ -594,7 +595,7 @@ DEFINE_IDTENTRY_RAW(exc_nmi)
irqentry_nmi_exit(regs, irq_state);
- local_db_restore(this_cpu_read(nmi_dr7));
+ local_db_restore(dr7, dr7_seq);
sev_es_ist_exit();
diff --git a/arch/x86/kernel/traps.c b/arch/x86/kernel/traps.c
index 30aa8369957e..018abe736285 100644
--- a/arch/x86/kernel/traps.c
+++ b/arch/x86/kernel/traps.c
@@ -1231,8 +1231,12 @@ static noinstr void exc_debug_kernel(struct pt_regs *regs, unsigned long dr6)
* it results in an endless recursion and stack overflow. Thus we stay
* with the IDT approach, i.e., save DR7 and disable #DB.
*/
- unsigned long dr7 = local_db_save();
- irqentry_state_t irq_state = irqentry_nmi_enter(regs);
+ unsigned long dr7;
+ unsigned int dr7_seq;
+ irqentry_state_t irq_state;
+
+ local_db_save(&dr7, &dr7_seq);
+ irq_state = irqentry_nmi_enter(regs);
instrumentation_begin();
/*
@@ -1289,7 +1293,7 @@ static noinstr void exc_debug_kernel(struct pt_regs *regs, unsigned long dr6)
instrumentation_end();
irqentry_nmi_exit(regs, irq_state);
- local_db_restore(dr7);
+ local_db_restore(dr7, dr7_seq);
}
static noinstr void exc_debug_user(struct pt_regs *regs, unsigned long dr6)
next prev parent reply other threads:[~2026-08-30 14:27 UTC|newest]
Thread overview: 15+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-30 14:26 [PATCH v14 00/14] tracing: wprobe: x86: Add wprobe for watchpoint Masami Hiramatsu (Google)
2026-08-30 14:27 ` [PATCH v14 01/14] x86/mce: Fix hardware debug register corruption on task migration Masami Hiramatsu (Google)
2026-08-30 14:27 ` [PATCH v14 02/14] tracing/probes: Fix BTF kflag check for anonymous struct member access Masami Hiramatsu (Google)
2026-08-30 14:27 ` [PATCH v14 03/14] kprobes: Protect kprobe_blacklist with RCU Masami Hiramatsu (Google)
2026-08-30 14:27 ` Masami Hiramatsu (Google) [this message]
2026-08-30 14:27 ` [PATCH v14 05/14] x86/hw_breakpoints: Add arch_modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
2026-08-30 14:27 ` [PATCH v14 06/14] HWBP: Add modify_local_hw_breakpoint_addr() API Masami Hiramatsu (Google)
2026-08-30 14:28 ` [PATCH v14 07/14] tracing/wprobe: Add wprobe (watchpoint probe) trace event support Masami Hiramatsu (Google)
2026-08-30 14:28 ` [PATCH v14 08/14] x86: hw_breakpoint: Add a kconfig to clarify when a breakpoint fires Masami Hiramatsu (Google)
2026-08-30 14:28 ` [PATCH v14 09/14] selftests: tracing: Add a basic testcase for wprobe Masami Hiramatsu (Google)
2026-08-30 14:28 ` [PATCH v14 10/14] selftests: tracing: Add syntax " Masami Hiramatsu (Google)
2026-08-30 14:28 ` [PATCH v14 11/14] tracing/wprobe: Add set_wprobe and clear_wprobe event triggers Masami Hiramatsu (Google)
2026-08-30 14:29 ` [PATCH v14 12/14] selftests: ftrace: Add wprobe trigger testcase Masami Hiramatsu (Google)
2026-08-30 14:29 ` [PATCH v14 13/14] tracing/wprobe: Support BTF typecast in fetchargs Masami Hiramatsu (Google)
2026-08-30 14:29 ` [PATCH v14 14/14] tracing/wprobe: Support BTF struct offset resolution in set_wprobe trigger Masami Hiramatsu (Google)
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=178810005476.64882.15902865504188093281.stgit@devnote2 \
--to=mhiramat@kernel.org \
--cc=alexander.shishkin@linux.intel.com \
--cc=bp@alien8.de \
--cc=dave.hansen@linux.intel.com \
--cc=hpa@zytor.com \
--cc=irogers@google.com \
--cc=linux-doc@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-perf-users@vger.kernel.org \
--cc=linux-trace-kernel@vger.kernel.org \
--cc=mathieu.desnoyers@efficios.com \
--cc=mingo@kernel.org \
--cc=peterz@infradead.org \
--cc=rostedt@goodmis.org \
--cc=tglx@linutronix.de \
--cc=wangjinchao600@gmail.com \
--cc=x86@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®