* [PATCH v3 01/31] gpu: nova-core: gsp: pass boot context through setup helpers
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 02/31] gpu: nova-core: gsp: decouple boot context from VgpuManager Zhi Wang
` (29 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The GSP boot paths pass the device, chipset, BAR and vGPU state to the
VRAM layout and GSP_INIT helpers as separate arguments. These values
are already available in GspBootContext, and some are forwarded again
to the WPR2 heap and VF topology helpers.
Pass the boot context through these helpers so each can read the fields
it needs. This simplifies the helper signatures and their call sites.
No functional change is intended.
Cc: Alexandre Courbot <acourbot@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/fb.rs | 26 +++++++++++---------------
drivers/gpu/nova-core/gsp/boot.rs | 2 +-
drivers/gpu/nova-core/gsp/commands.rs | 26 +++++++++++---------------
drivers/gpu/nova-core/gsp/hal/gh100.rs | 2 +-
drivers/gpu/nova-core/gsp/hal/tu102.rs | 2 +-
5 files changed, 25 insertions(+), 33 deletions(-)
diff --git a/drivers/gpu/nova-core/fb.rs b/drivers/gpu/nova-core/fb.rs
index 5ceb7760c78e..26e94d908b77 100644
--- a/drivers/gpu/nova-core/fb.rs
+++ b/drivers/gpu/nova-core/fb.rs
@@ -174,13 +174,9 @@ pub(crate) struct FbRanges {
impl FbRanges {
/// Computes concrete framebuffer ranges required on non-FSP booting architectures.
- pub(crate) fn new(
- chipset: Chipset,
- bar: Bar0<'_>,
- gsp_fw: &GspFirmware<'_>,
- vgpu_state: VgpuState,
- ) -> Result<Self> {
- let hal = hal::fb_hal(chipset);
+ pub(crate) fn new(ctx: &gsp::GspBootContext<'_, '_>, gsp_fw: &GspFirmware<'_>) -> Result<Self> {
+ let bar = ctx.bar;
+ let hal = hal::fb_hal(ctx.chipset);
let fb = {
let fb_size = hal.vidmem_size(bar);
@@ -242,7 +238,7 @@ pub(crate) fn new(
FbRange(fw_image_addr..fw_image_addr + fw_image_size)
};
- let (vf_partition_count, wpr2_heap_size) = wpr2_heap_params(chipset, vgpu_state, fb.end)?;
+ let (vf_partition_count, wpr2_heap_size) = wpr2_heap_params(ctx, fb.end)?;
let wpr2_heap = {
const WPR2_HEAP_DOWN_ALIGN: Alignment = Alignment::SZ_1M;
@@ -298,11 +294,11 @@ pub(crate) fn wpr2_range(bar: Bar0<'_>) -> Option<Range<u64>> {
}
/// Computes the number of VF partitions and the WPR2 heap size from the vGPU state.
-fn wpr2_heap_params(chipset: Chipset, vgpu_state: VgpuState, fb_size: u64) -> Result<(u8, u64)> {
- Ok(match vgpu_state {
+fn wpr2_heap_params(ctx: &gsp::GspBootContext<'_, '_>, fb_size: u64) -> Result<(u8, u64)> {
+ Ok(match ctx.vgpu.state() {
VgpuState::Disabled => (
0,
- gsp::LibosParams::from_chipset(chipset).wpr_heap_size(chipset, fb_size)?,
+ gsp::LibosParams::from_chipset(ctx.chipset).wpr_heap_size(ctx.chipset, fb_size)?,
),
VgpuState::Enabled { total_vfs } => (
u8::try_from(total_vfs.get()).map_err(|_| EINVAL)?,
@@ -328,10 +324,10 @@ pub(crate) struct FbSizes {
impl FbSizes {
/// Computes the framebuffer region sizes for GSP-FMC boot.
- pub(crate) fn new(chipset: Chipset, bar: Bar0<'_>, vgpu_state: VgpuState) -> Result<Self> {
- let hal = hal::fb_hal(chipset);
- let fb_size = hal.vidmem_size(bar);
- let (vf_partition_count, wpr2_heap_size) = wpr2_heap_params(chipset, vgpu_state, fb_size)?;
+ pub(crate) fn new(ctx: &gsp::GspBootContext<'_, '_>) -> Result<Self> {
+ let hal = hal::fb_hal(ctx.chipset);
+ let fb_size = hal.vidmem_size(ctx.bar);
+ let (vf_partition_count, wpr2_heap_size) = wpr2_heap_params(ctx, fb_size)?;
Ok(Self {
frts_size: hal.frts_size(),
diff --git a/drivers/gpu/nova-core/gsp/boot.rs b/drivers/gpu/nova-core/gsp/boot.rs
index 8848feebb9db..510d32212d41 100644
--- a/drivers/gpu/nova-core/gsp/boot.rs
+++ b/drivers/gpu/nova-core/gsp/boot.rs
@@ -401,7 +401,7 @@ pub(crate) fn boot(
dev_dbg!(pdev, "RISC-V active? {}\n", gsp_falcon.is_riscv_active(),);
- let init_payload = commands::build_gsp_init_payload(pdev, chipset, ctx.vgpu.state())?;
+ let init_payload = commands::build_gsp_init_payload(ctx)?;
let load_exec = LoadExecContext {
bootloader: generic_bootloader.as_ref(),
gsp_falcon,
diff --git a/drivers/gpu/nova-core/gsp/commands.rs b/drivers/gpu/nova-core/gsp/commands.rs
index d12f90d94eb6..d6895dc5207e 100644
--- a/drivers/gpu/nova-core/gsp/commands.rs
+++ b/drivers/gpu/nova-core/gsp/commands.rs
@@ -2,14 +2,12 @@
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
use kernel::{
- device,
pci,
prelude::*,
transmute::AsBytes, //
};
use crate::{
- gpu::Chipset,
gsp::{
cmdq::Cmdq,
fw::{
@@ -30,6 +28,7 @@
Encoder,
UnknownKeyPolicy, //
},
+ GspBootContext, //
},
sbuffer::SBufferIter,
vgpu::VgpuState, //
@@ -42,28 +41,25 @@
/// # Errors
///
/// - `ENOMEM` if the request or the encoder buffer cannot be allocated.
-pub(crate) fn build_gsp_init_payload(
- pdev: &pci::Device<device::Bound>,
- chipset: Chipset,
- vgpu_state: VgpuState,
-) -> Result<EncodedStream> {
+/// - `ENODEV` if vGPU mode is enabled but the SR-IOV capability is missing.
+///
+/// Errors reading the PCI configuration or decoding the VF BAR layout are propagated as-is.
+pub(super) fn build_gsp_init_payload(ctx: &GspBootContext<'_, '_>) -> Result<EncodedStream> {
let mut encoder = Encoder::new();
- let vf_info = build_vf_info(pdev, vgpu_state)?;
- GspInitRequest::new(pdev, chipset, vgpu_state, vf_info)?.encode(&mut encoder)?;
+ let vf_info = build_vf_info(ctx)?;
+ GspInitRequest::new(ctx.pdev, ctx.chipset, ctx.vgpu.state(), vf_info)?.encode(&mut encoder)?;
Ok(encoder.finish())
}
/// Builds the optional VF topology portion of the `GSP_INIT` request.
-fn build_vf_info(
- pdev: &pci::Device<device::Bound>,
- vgpu_state: VgpuState,
-) -> Result<Option<VfInfo>> {
- let VgpuState::Enabled { total_vfs } = vgpu_state else {
+fn build_vf_info(ctx: &GspBootContext<'_, '_>) -> Result<Option<VfInfo>> {
+ let VgpuState::Enabled { total_vfs } = ctx.vgpu.state() else {
return Ok(None);
};
- let sriov = pdev
+ let sriov = ctx
+ .pdev
.config_space_extended()?
.find_ext_capability::<pci::ExtSriovRegs>()?
.ok_or(ENODEV)?;
diff --git a/drivers/gpu/nova-core/gsp/hal/gh100.rs b/drivers/gpu/nova-core/gsp/hal/gh100.rs
index 91201b51030e..f91cdd061706 100644
--- a/drivers/gpu/nova-core/gsp/hal/gh100.rs
+++ b/drivers/gpu/nova-core/gsp/hal/gh100.rs
@@ -151,7 +151,7 @@ fn boot<'gpu>(
let chipset = ctx.chipset;
let gsp_falcon = ctx.gsp_falcon;
- let fb_sizes = FbSizes::new(chipset, ctx.bar, ctx.vgpu.state())?;
+ let fb_sizes = FbSizes::new(ctx)?;
dev_dbg!(dev, "{:#x?}\n", fb_sizes);
let wpr_meta =
diff --git a/drivers/gpu/nova-core/gsp/hal/tu102.rs b/drivers/gpu/nova-core/gsp/hal/tu102.rs
index 265039b8ff55..c221797d6f3c 100644
--- a/drivers/gpu/nova-core/gsp/hal/tu102.rs
+++ b/drivers/gpu/nova-core/gsp/hal/tu102.rs
@@ -260,7 +260,7 @@ fn boot<'gpu>(
let gsp_falcon = ctx.gsp_falcon;
let sec2_falcon = ctx.sec2_falcon;
- let fb_ranges = FbRanges::new(chipset, bar, gsp_fw, ctx.vgpu.state())?;
+ let fb_ranges = FbRanges::new(ctx, gsp_fw)?;
dev_dbg!(dev, "{:#x?}\n", fb_ranges);
// Declared before the unload guard so that if Booter fails while running, SEC2 is reset
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 02/31] gpu: nova-core: gsp: decouple boot context from VgpuManager
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
2026-09-28 10:28 ` [PATCH v3 01/31] gpu: nova-core: gsp: pass boot context through setup helpers Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 03/31] gpu: nova-core: vgpu: detect boot state independently Zhi Wang
` (28 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
GSP boot uses the detected vGPU mode to size firmware resources and
populate GSP_INIT. Carrying the whole manager in the boot and unload
context couples these operations to its runtime interface.
The boot code only needs read access to the mode, and unloading does
not use the manager. A short state borrow is sufficient for both
context construction sites.
Make GspBootContext borrow VgpuState directly and read it in the
VRAM and GSP_INIT helpers. Return a state reference from the
manager without changing mode detection or the values passed to firmware.
No functional change is intended.
Cc: Alexandre Courbot <acourbot@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/fb.rs | 2 +-
drivers/gpu/nova-core/gpu.rs | 4 ++--
drivers/gpu/nova-core/gsp.rs | 4 ++--
drivers/gpu/nova-core/gsp/commands.rs | 4 ++--
drivers/gpu/nova-core/vgpu.rs | 4 ++--
5 files changed, 9 insertions(+), 9 deletions(-)
diff --git a/drivers/gpu/nova-core/fb.rs b/drivers/gpu/nova-core/fb.rs
index 26e94d908b77..882827bcfd3a 100644
--- a/drivers/gpu/nova-core/fb.rs
+++ b/drivers/gpu/nova-core/fb.rs
@@ -295,7 +295,7 @@ pub(crate) fn wpr2_range(bar: Bar0<'_>) -> Option<Range<u64>> {
/// Computes the number of VF partitions and the WPR2 heap size from the vGPU state.
fn wpr2_heap_params(ctx: &gsp::GspBootContext<'_, '_>, fb_size: u64) -> Result<(u8, u64)> {
- Ok(match ctx.vgpu.state() {
+ Ok(match *ctx.vgpu_state {
VgpuState::Disabled => (
0,
gsp::LibosParams::from_chipset(ctx.chipset).wpr_heap_size(ctx.chipset, fb_size)?,
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index be212bbf724f..3272afdb7d1e 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -372,7 +372,7 @@ fn drop(self: Pin<&mut Self>) {
gsp_falcon: &*this.gsp_falcon,
sec2_falcon: &*this.sec2_falcon,
fsp: this.fsp.as_mut(),
- vgpu: &*this.vgpu,
+ vgpu_state: this.vgpu.state(),
},
bundle,
)
@@ -445,7 +445,7 @@ pub(crate) fn new<'a>(
gsp_falcon,
sec2_falcon,
fsp: fsp.as_mut(),
- vgpu,
+ vgpu_state: vgpu.state(),
})?,
}),
diff --git a/drivers/gpu/nova-core/gsp.rs b/drivers/gpu/nova-core/gsp.rs
index 5b57be7f2d42..3dbc37a3b35b 100644
--- a/drivers/gpu/nova-core/gsp.rs
+++ b/drivers/gpu/nova-core/gsp.rs
@@ -66,7 +66,7 @@
fw::GspArgumentsPadded, //
},
num,
- vgpu::VgpuManager, //
+ vgpu::VgpuState, //
};
pub(crate) const GSP_PAGE_SHIFT: usize = 12;
@@ -85,7 +85,7 @@ pub(crate) struct GspBootContext<'ctx, 'gpu> {
pub(crate) gsp_falcon: &'ctx Falcon<'gpu, GspFalcon>,
pub(crate) sec2_falcon: &'ctx Falcon<'gpu, Sec2Falcon>,
pub(crate) fsp: Option<&'ctx mut Fsp<'gpu>>,
- pub(crate) vgpu: &'ctx VgpuManager,
+ pub(crate) vgpu_state: &'ctx VgpuState,
}
impl<'ctx, 'gpu> GspBootContext<'ctx, 'gpu> {
diff --git a/drivers/gpu/nova-core/gsp/commands.rs b/drivers/gpu/nova-core/gsp/commands.rs
index d6895dc5207e..9473004a0921 100644
--- a/drivers/gpu/nova-core/gsp/commands.rs
+++ b/drivers/gpu/nova-core/gsp/commands.rs
@@ -47,14 +47,14 @@
pub(super) fn build_gsp_init_payload(ctx: &GspBootContext<'_, '_>) -> Result<EncodedStream> {
let mut encoder = Encoder::new();
let vf_info = build_vf_info(ctx)?;
- GspInitRequest::new(ctx.pdev, ctx.chipset, ctx.vgpu.state(), vf_info)?.encode(&mut encoder)?;
+ GspInitRequest::new(ctx.pdev, ctx.chipset, *ctx.vgpu_state, vf_info)?.encode(&mut encoder)?;
Ok(encoder.finish())
}
/// Builds the optional VF topology portion of the `GSP_INIT` request.
fn build_vf_info(ctx: &GspBootContext<'_, '_>) -> Result<Option<VfInfo>> {
- let VgpuState::Enabled { total_vfs } = ctx.vgpu.state() else {
+ let VgpuState::Enabled { total_vfs } = *ctx.vgpu_state else {
return Ok(None);
};
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 6b7e045acea8..905b3a7331fd 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -85,7 +85,7 @@ fn detect_state(
}
/// Returns the detected vGPU state for this boot.
- pub(crate) fn state(&self) -> VgpuState {
- self.state
+ pub(crate) fn state(&self) -> &VgpuState {
+ &self.state
}
}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 03/31] gpu: nova-core: vgpu: detect boot state independently
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
2026-09-28 10:28 ` [PATCH v3 01/31] gpu: nova-core: gsp: pass boot context through setup helpers Zhi Wang
2026-09-28 10:28 ` [PATCH v3 02/31] gpu: nova-core: gsp: decouple boot context from VgpuManager Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 04/31] gpu: nova-core: gpu: add a channel ID pool for vGPU Zhi Wang
` (27 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The vGPU manager currently contains only the detected boot mode. Its
construction couples mode detection to an object with no runtime
management responsibilities.
Firmware setup needs the mode before allocating GSP resources, even
when vGPU is disabled. Keeping that value with GspResources also makes
it available throughout the firmware lifecycle without a manager.
Move detection to VgpuState and store its result in GspResources.
Borrow it directly when constructing the boot and unload contexts, and
remove the state-only manager wrapper. Preserve the FSP query order,
existing diagnostics and disabled fallback.
No functional change is intended.
Cc: Alexandre Courbot <acourbot@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 11 +++++------
drivers/gpu/nova-core/vgpu.rs | 25 ++++++++-----------------
2 files changed, 13 insertions(+), 23 deletions(-)
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index 3272afdb7d1e..650bef8b5f44 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -48,7 +48,7 @@
GpuMm,
VramAddress, //
},
- vgpu::VgpuManager, //
+ vgpu::VgpuState, //
};
#[cfg_attr(not(CONFIG_KUNIT = "y"), expect(dead_code))]
@@ -299,8 +299,7 @@ struct GspResources<'gpu> {
// TODO: use different resource types for each boot method, and make the relevant Gsp methods
// generic against them.
fsp: Option<Fsp<'gpu>>,
- /// vGPU state detected before GSP boot.
- vgpu: VgpuManager,
+ vgpu_state: VgpuState,
/// GSP runtime data.
#[pin]
gsp: Gsp<'gpu>,
@@ -372,7 +371,7 @@ fn drop(self: Pin<&mut Self>) {
gsp_falcon: &*this.gsp_falcon,
sec2_falcon: &*this.sec2_falcon,
fsp: this.fsp.as_mut(),
- vgpu_state: this.vgpu.state(),
+ vgpu_state: this.vgpu_state,
},
bundle,
)
@@ -431,7 +430,7 @@ pub(crate) fn new<'a>(
fsp: Fsp::try_new(dev, bar, spec.chipset)?,
- vgpu: VgpuManager::new(pdev, spec.chipset, fsp.as_mut()),
+ vgpu_state: VgpuState::detect(pdev, spec.chipset, fsp.as_mut()),
gsp <- Gsp::new(pdev, *spec, bar),
@@ -445,7 +444,7 @@ pub(crate) fn new<'a>(
gsp_falcon,
sec2_falcon,
fsp: fsp.as_mut(),
- vgpu_state: vgpu.state(),
+ vgpu_state,
})?,
}),
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 905b3a7331fd..b405ba49c490 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -30,19 +30,16 @@ pub(crate) enum VgpuState {
},
}
-/// vGPU state manager.
-pub(crate) struct VgpuManager {
- state: VgpuState,
-}
-
-impl VgpuManager {
- /// Creates a vGPU manager by querying SR-IOV and the FSP PRC vGPU knob.
- pub(crate) fn new(
+impl VgpuState {
+ /// Detects the boot mode, falling back to disabled if querying the device fails.
+ ///
+ /// Call after creating the FSP and before allocating GSP firmware resources.
+ pub(crate) fn detect(
pdev: &pci::Device<device::Core<'_>>,
chipset: Chipset,
fsp: Option<&mut Fsp<'_>>,
) -> Self {
- let state = Self::detect_state(pdev, chipset, fsp).unwrap_or_else(|e| {
+ let state = Self::query_state(pdev, chipset, fsp).unwrap_or_else(|e| {
dev_warn!(
pdev,
"vGPU state detection failed: {:?}; disabling vGPU\n",
@@ -51,12 +48,11 @@ pub(crate) fn new(
VgpuState::Disabled
});
dev_dbg!(pdev, "vGPU state: {:?}\n", state);
-
- Self { state }
+ state
}
/// Detects the vGPU state from the chipset, SR-IOV capability and FSP PRC knob.
- fn detect_state(
+ fn query_state(
pdev: &pci::Device<device::Core<'_>>,
chipset: Chipset,
fsp: Option<&mut Fsp<'_>>,
@@ -83,9 +79,4 @@ fn detect_state(
VgpuMode::Disabled => Ok(VgpuState::Disabled),
}
}
-
- /// Returns the detected vGPU state for this boot.
- pub(crate) fn state(&self) -> &VgpuState {
- &self.state
- }
}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 04/31] gpu: nova-core: gpu: add a channel ID pool for vGPU
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (2 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 03/31] gpu: nova-core: vgpu: detect boot state independently Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 05/31] gpu: nova-core: gsp: decode the FIFO engine table Zhi Wang
` (26 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Channel ID reservations need a GPU-owned pool shared by vGPU instances.
The pool must remain at a stable address while its reservations exist.
Its storage also needs to remain available throughout GPU cleanup.
Create a private pinned ChannelIdPool with a single 2048-channel
capacity constant.
Cc: Alexandre Courbot <acourbot@nvidia.com>
Suggested-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 9 +++++++++
drivers/gpu/nova-core/gpu/channel.rs | 4 ++++
2 files changed, 13 insertions(+)
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index 650bef8b5f44..c239b6614d44 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -48,6 +48,7 @@
GpuMm,
VramAddress, //
},
+ num,
vgpu::VgpuState, //
};
@@ -56,6 +57,10 @@
mod hal;
mod regs;
+use self::channel::TOTAL_CHANNELS;
+
+pub(crate) use self::channel::ChannelIdPool;
+
macro_rules! define_chipset {
({ $($variant:ident = $value:expr),* $(,)* }) =>
{
@@ -335,6 +340,8 @@ pub(crate) struct Gpu<'gpu> {
/// GSP and its resources.
#[pin]
gsp_resources: GspResources<'gpu>,
+ #[pin]
+ chid_pool: ChannelIdPool,
/// System memory page required for flushing all pending GPU-side memory writes done through
/// PCIE into system memory, via sysmembar (A GPU-initiated HW memory-barrier operation).
///
@@ -417,6 +424,8 @@ pub(crate) fn new<'a>(
// Initialize this early because `gsp_resources` depends on it.
sysmem_flush: SysmemFlush::register(dev, bar, spec.chipset)?,
+ chid_pool <- ChannelIdPool::new(cv!(num::u32_as_usize(TOTAL_CHANNELS))),
+
gsp_resources <- try_pin_init!(GspResources {
device: pdev,
diff --git a/drivers/gpu/nova-core/gpu/channel.rs b/drivers/gpu/nova-core/gpu/channel.rs
index 485efaba059d..75666c6768e6 100644
--- a/drivers/gpu/nova-core/gpu/channel.rs
+++ b/drivers/gpu/nova-core/gpu/channel.rs
@@ -21,6 +21,10 @@
}, //
};
+// TODO: Query the channel capacity through GMCAPI once an equivalent of
+// NV2080_CTRL_CMD_INTERNAL_FIFO_GET_NUM_CHANNELS is available.
+pub(super) const TOTAL_CHANNELS: u32 = 2048;
+
/// Pool for tracking reservations of channel IDs.
#[pin_data]
pub(crate) struct ChannelIdPool {
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 05/31] gpu: nova-core: gsp: decode the FIFO engine table
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (3 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 04/31] gpu: nova-core: gpu: add a channel ID pool for vGPU Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 06/31] gpu: nova-core: vgpu: initialize runtime parameters after GSP boot Zhi Wang
` (25 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
From: Alok Kumar <alkumar@nvidia.com>
GSP_INIT reports the FIFO engine topology used to build the channel map
for vGPU instances.
Decode the count, GMC IDs and flags into GspStaticInfo and expose
host-driven engines in firmware FIFO order, retaining repeated IDs.
Reject counts and indexed entries beyond the driver's 64-entry table
capacity rather than silently accepting a prefix. An omitted count
remains zero.
Cc: Alexandre Courbot <acourbot@nvidia.com>
Signed-off-by: Alok Kumar <alkumar@nvidia.com>
Co-developed-by: Zhi Wang <zhiw@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gsp/commands.rs | 2 +-
drivers/gpu/nova-core/gsp/fw/commands.rs | 80 ++++++++++++++++++++++++
2 files changed, 81 insertions(+), 1 deletion(-)
diff --git a/drivers/gpu/nova-core/gsp/commands.rs b/drivers/gpu/nova-core/gsp/commands.rs
index 9473004a0921..436ea52689b9 100644
--- a/drivers/gpu/nova-core/gsp/commands.rs
+++ b/drivers/gpu/nova-core/gsp/commands.rs
@@ -126,7 +126,7 @@ pub(crate) fn gsp_init(
/// # Errors
///
/// - `EINVAL` if the payload is not a whole number of NVKV words, or if the stream is malformed
-/// or omits a required key.
+/// or omits a required key, or the FIFO engine count exceeds the supported table capacity.
/// - `ENOMEM` if the words or the decoded regions cannot be allocated.
fn decode_gsp_init_reply(payload_0: &[u8], payload_1: &[u8]) -> Result<GspStaticInfo> {
const WORD_SIZE: usize = size_of::<u64>();
diff --git a/drivers/gpu/nova-core/gsp/fw/commands.rs b/drivers/gpu/nova-core/gsp/fw/commands.rs
index d0715dca1be2..17ae4224e214 100644
--- a/drivers/gpu/nova-core/gsp/fw/commands.rs
+++ b/drivers/gpu/nova-core/gsp/fw/commands.rs
@@ -29,6 +29,7 @@
DecoderValue,
Encodable,
Encoder,
+ Indexed,
Key,
KeyId,
Required, //
@@ -317,6 +318,52 @@ pub(crate) fn new(
// Decode:
+const MAX_FIFO_ENGINES: usize = 64;
+
+/// Bit mask for `NVGMC_SC_ENGINE_FLAGS_IS_HOST_DRIVEN`.
+const ENGINE_FLAGS_IS_HOST_DRIVEN: u32 = 1 << 0;
+
+/// Host-driven GMC engine IDs in hardware FIFO order, including any repeated IDs.
+///
+/// # Invariants
+///
+/// `count` is at most [`MAX_FIFO_ENGINES`]. The first `count` slots are the retained engine IDs.
+#[derive(Copy, Clone)]
+pub(crate) struct FifoEngineList {
+ gmc_ids: [u32; MAX_FIFO_ENGINES],
+ count: usize,
+}
+
+impl FifoEngineList {
+ #[expect(dead_code)]
+ pub(crate) fn gmc_ids(&self) -> &[u32] {
+ // PANIC: The type invariant bounds `count` by the array capacity.
+ &self.gmc_ids[..self.count]
+ }
+}
+
+/// A FIFO engine count that fits the supported tables.
+///
+/// # Invariants
+///
+/// The count is at most [`MAX_FIFO_ENGINES`].
+#[derive(Default)]
+struct FifoEngineCount(usize);
+
+impl TryFrom<DecoderValue<'_>> for FifoEngineCount {
+ type Error = Error;
+
+ fn try_from(value: DecoderValue<'_>) -> Result<Self> {
+ let count = crate::num::u32_as_usize(u32::try_from(value)?);
+ if count > MAX_FIFO_ENGINES {
+ Err(EINVAL)
+ } else {
+ // INVARIANT: The count was checked against the table capacity above.
+ Ok(Self(count))
+ }
+ }
+}
+
// Should decode with UnknownKeyPolicy::Ignore.
nvkv_decode! {
/// Schema for the `GSP_INIT` response.
@@ -326,6 +373,9 @@ pub(crate) struct GspInitResponseSchema => GspStaticInfo {
fb_regions: Accumulated<FbRegionSchema>,
bar1_pde_base: Required<u64, { Self::BAR1_PDE_BASE_KEY }>,
vmmu_segment_size: Key<u64, { Self::VMMU_SEGMENT_SIZE_KEY }>,
+ fifo_engine_count: Key<FifoEngineCount, { Self::FIFO_ENGINE_COUNT_KEY }>,
+ fifo_engine_gmc_ids: Indexed<u32, MAX_FIFO_ENGINES, { Self::FIFO_ENGINE_GMC_ID_KEY }>,
+ fifo_engine_flags: Indexed<u32, MAX_FIFO_ENGINES, { Self::FIFO_ENGINE_FLAGS_KEY }>,
}
}
@@ -334,6 +384,9 @@ impl GspInitResponseSchema {
const GPU_NAME_STRING_KEY: KeyId = 0x2000;
const BAR1_PDE_BASE_KEY: KeyId = 0x1020;
const VMMU_SEGMENT_SIZE_KEY: KeyId = 0x1050;
+ const FIFO_ENGINE_COUNT_KEY: KeyId = 0x0500;
+ const FIFO_ENGINE_GMC_ID_KEY: KeyId = 0x0501;
+ const FIFO_ENGINE_FLAGS_KEY: KeyId = 0x0502;
}
/// The static GPU configuration, as decoded from the `GSP_INIT` reply.
@@ -343,6 +396,9 @@ pub(crate) struct GspStaticInfo {
bar1_pde_base: u64,
#[cfg_attr(not(CONFIG_KUNIT = "y"), expect(dead_code))]
vmmu_segment_size: u64,
+ fifo_engine_count: FifoEngineCount,
+ fifo_engine_gmc_ids: [u32; MAX_FIFO_ENGINES],
+ fifo_engine_flags: [u32; MAX_FIFO_ENGINES],
}
/// Error type for [`GspStaticInfo::gpu_name`].
@@ -407,6 +463,30 @@ pub(crate) fn total_fb_end(&self) -> Option<u64> {
.checked_add(1)
}
+ /// Returns the host-driven engines in their firmware FIFO order.
+ #[expect(dead_code)]
+ pub(crate) fn fifo_engine_list(&self) -> FifoEngineList {
+ // INVARIANT: The list starts empty and appends at most one ID per supported input slot.
+ let mut fifo_engine_list = FifoEngineList {
+ gmc_ids: [0; MAX_FIFO_ENGINES],
+ count: 0,
+ };
+ for (&gmc_id, &flags) in self
+ .fifo_engine_gmc_ids
+ .iter()
+ .zip(&self.fifo_engine_flags)
+ .take(self.fifo_engine_count.0)
+ {
+ if flags & ENGINE_FLAGS_IS_HOST_DRIVEN != 0 {
+ // PANIC: At most one slot is filled per input, and the input has at most
+ // MAX_FIFO_ENGINES entries, so the next retained ID always fits.
+ fifo_engine_list.gmc_ids[fifo_engine_list.count] = gmc_id;
+ fifo_engine_list.count += 1;
+ }
+ }
+ fifo_engine_list
+ }
+
/// Returns the BAR1 page directory entry base address.
pub(crate) fn bar1_pde_base(&self) -> u64 {
self.bar1_pde_base
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 06/31] gpu: nova-core: vgpu: initialize runtime parameters after GSP boot
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (4 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 05/31] gpu: nova-core: gsp: decode the FIFO engine table Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 07/31] gpu: nova-core: vgpu: reserve the 48-VM WPR2 heap Zhi Wang
` (24 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The vGPU manager needs the VMMU segment size and FIFO engine topology
returned by a successful GSP boot to divide instance resources.
Construct VgpuManager only for a vGPU-enabled boot. Retain the GPU-owned
channel pool, the ordered host-driven engine
list and the VMMU segment size. Expose the existing decoded VMMU value
from GspStaticInfo at its first consumer, preserving zero when firmware
omits the key.
Keep the runtime values in the manager for its instance module, and drop
the manager before the pool and memory resources it borrows.
Cc: Alexandre Courbot <acourbot@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 24 +++++++++++++++-
drivers/gpu/nova-core/gsp/commands.rs | 5 +++-
drivers/gpu/nova-core/gsp/fw/commands.rs | 4 +--
drivers/gpu/nova-core/vgpu.rs | 36 +++++++++++++++++++++++-
4 files changed, 63 insertions(+), 6 deletions(-)
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index c239b6614d44..11d9986c4b19 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -49,7 +49,10 @@
VramAddress, //
},
num,
- vgpu::VgpuState, //
+ vgpu::{
+ VgpuManager,
+ VgpuState, //
+ },
};
#[cfg_attr(not(CONFIG_KUNIT = "y"), expect(dead_code))]
@@ -323,6 +326,7 @@ fn static_info(&self) -> &gsp::commands::GspStaticInfo {
#[pin_data]
pub(crate) struct Gpu<'gpu> {
spec: Spec,
+ vgpu: Option<VgpuManager<'gpu>>,
/// GSP event interrupt registration.
///
/// Must be kept declared *before* `gsp_resources`, so that the handler is unregistered, and
@@ -457,6 +461,24 @@ pub(crate) fn new<'a>(
})?,
}),
+ vgpu: {
+ let info = &gsp_resources.boot_result.static_info;
+ match gsp_resources.vgpu_state {
+ VgpuState::Disabled => None,
+ VgpuState::Enabled { .. } => Some(VgpuManager::new(
+ // SAFETY: `chid_pool` is initialized above at its final pinned address.
+ // The private manager and its pool borrow cannot escape this `Gpu`.
+ // Completed field drop order drops the manager before the pool; on failure,
+ // pin-init drops it before the earlier-initialized pool.
+ unsafe { &*core::ptr::from_ref(chid_pool.as_ref().get_ref()) },
+ &info.fifo_engine_list(),
+ info.vmmu_segment_size,
+ TOTAL_CHANNELS,
+ )),
+ }
+ },
+
+ // GSP boot left the SWGEN0 latch set and pending bits in the tree.
_: {
irq::gsp::quiesce(bar, gsp_resources.spec.chipset, vectors_ref)?;
},
diff --git a/drivers/gpu/nova-core/gsp/commands.rs b/drivers/gpu/nova-core/gsp/commands.rs
index 436ea52689b9..97e32f81a39b 100644
--- a/drivers/gpu/nova-core/gsp/commands.rs
+++ b/drivers/gpu/nova-core/gsp/commands.rs
@@ -34,7 +34,10 @@
vgpu::VgpuState, //
};
-pub(crate) use fw::commands::GspStaticInfo;
+pub(crate) use fw::commands::{
+ FifoEngineList,
+ GspStaticInfo, //
+};
/// Builds the NVKV-encoded payload of a `GSP_INIT` request for `pdev`.
///
diff --git a/drivers/gpu/nova-core/gsp/fw/commands.rs b/drivers/gpu/nova-core/gsp/fw/commands.rs
index 17ae4224e214..7381b30f3680 100644
--- a/drivers/gpu/nova-core/gsp/fw/commands.rs
+++ b/drivers/gpu/nova-core/gsp/fw/commands.rs
@@ -394,8 +394,7 @@ pub(crate) struct GspStaticInfo {
gpu_name: ArrayVec<u8, { Self::MAX_GPU_NAME_LEN }>,
fb_regions: KVVec<FbRegion>,
bar1_pde_base: u64,
- #[cfg_attr(not(CONFIG_KUNIT = "y"), expect(dead_code))]
- vmmu_segment_size: u64,
+ pub(crate) vmmu_segment_size: u64,
fifo_engine_count: FifoEngineCount,
fifo_engine_gmc_ids: [u32; MAX_FIFO_ENGINES],
fifo_engine_flags: [u32; MAX_FIFO_ENGINES],
@@ -464,7 +463,6 @@ pub(crate) fn total_fb_end(&self) -> Option<u64> {
}
/// Returns the host-driven engines in their firmware FIFO order.
- #[expect(dead_code)]
pub(crate) fn fifo_engine_list(&self) -> FifoEngineList {
// INVARIANT: The list starts empty and appends at most one ID per supported input slot.
let mut fifo_engine_list = FifoEngineList {
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index b405ba49c490..5cd82adeb0f8 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -13,7 +13,11 @@
Fsp,
VgpuMode, //
},
- gpu::Chipset, //
+ gpu::{
+ ChannelIdPool,
+ Chipset, //
+ },
+ gsp::commands::FifoEngineList, //
};
mod hal;
@@ -80,3 +84,33 @@ fn query_state(
}
}
}
+
+/// Runtime resources for an enabled vGPU boot.
+pub(crate) struct VgpuManager<'gpu> {
+ #[expect(dead_code)]
+ chid_pool: &'gpu ChannelIdPool,
+ /// VMMU segment size in bytes, or zero if GSP-RM omitted it.
+ #[expect(dead_code)]
+ vmmu_segment_size: u64,
+ #[expect(dead_code)]
+ total_channels: u32,
+ #[expect(dead_code)]
+ fifo_engine_list: FifoEngineList,
+}
+
+impl<'gpu> VgpuManager<'gpu> {
+ /// Retains runtime parameters from a completed vGPU-enabled GSP boot.
+ pub(crate) fn new(
+ chid_pool: &'gpu ChannelIdPool,
+ fifo_engine_list: &FifoEngineList,
+ vmmu_segment_size: u64,
+ total_channels: u32,
+ ) -> Self {
+ Self {
+ chid_pool,
+ vmmu_segment_size,
+ total_channels,
+ fifo_engine_list: *fifo_engine_list,
+ }
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 07/31] gpu: nova-core: vgpu: reserve the 48-VM WPR2 heap
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (5 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 06/31] gpu: nova-core: vgpu: initialize runtime parameters after GSP boot Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 08/31] gpu: nova-core: mm: borrow BarUser for temporary BAR1 access Zhi Wang
` (23 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
GSP-RM needs the larger r000 vGPU WPR2 heap on devices that
advertise more than 32 VFs. The default heap is too small for the 48-VF
GB202 configuration and GSP-RM fails to start.
Select the firmware-defined 48-VM size from the detected VF count while
retaining the default size for configurations with 2 through 32 VFs.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/fb.rs | 2 +-
drivers/gpu/nova-core/gsp/fw.rs | 7 +++++--
drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs | 1 +
3 files changed, 7 insertions(+), 3 deletions(-)
diff --git a/drivers/gpu/nova-core/fb.rs b/drivers/gpu/nova-core/fb.rs
index 882827bcfd3a..ba68a812b095 100644
--- a/drivers/gpu/nova-core/fb.rs
+++ b/drivers/gpu/nova-core/fb.rs
@@ -302,7 +302,7 @@ fn wpr2_heap_params(ctx: &gsp::GspBootContext<'_, '_>, fb_size: u64) -> Result<(
),
VgpuState::Enabled { total_vfs } => (
u8::try_from(total_vfs.get()).map_err(|_| EINVAL)?,
- gsp::LibosParams::vgpu_wpr_heap_size(),
+ gsp::LibosParams::vgpu_wpr_heap_size(total_vfs.get()),
),
})
}
diff --git a/drivers/gpu/nova-core/gsp/fw.rs b/drivers/gpu/nova-core/gsp/fw.rs
index 717ba1cc4bf3..90bb324666e5 100644
--- a/drivers/gpu/nova-core/gsp/fw.rs
+++ b/drivers/gpu/nova-core/gsp/fw.rs
@@ -133,8 +133,11 @@ pub(crate) fn from_chipset(chipset: Chipset) -> &'static LibosParams {
}
/// Returns the WPR heap size to reserve when vGPU is enabled.
- pub(crate) fn vgpu_wpr_heap_size() -> u64 {
- u64::from(bindings::GSP_FW_HEAP_SIZE_VGPU_DEFAULT)
+ pub(crate) fn vgpu_wpr_heap_size(total_vfs: u16) -> u64 {
+ u64::from(match total_vfs {
+ 2..=32 => bindings::GSP_FW_HEAP_SIZE_VGPU_DEFAULT,
+ _ => r000_00::GSP_FW_HEAP_SIZE_VGPU_48VMS,
+ })
}
/// Returns the amount of memory (in bytes) to allocate for the WPR heap for a framebuffer size
diff --git a/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs b/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
index 454d0ea9360f..fa17d3e2edf6 100644
--- a/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
+++ b/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
@@ -84,6 +84,7 @@ impl<T> ::core::cmp::Eq for __BindgenUnionField<T> {}
pub const GSP_FW_HEAP_PARAM_SIZE_PER_GB: u32 = 98304;
pub const GSP_FW_HEAP_PARAM_CLIENT_ALLOC_SIZE: u32 = 100663296;
pub const GSP_FW_HEAP_SIZE_VGPU_DEFAULT: u32 = 609222656;
+pub const GSP_FW_HEAP_SIZE_VGPU_48VMS: u32 = 1436549120;
pub const GSP_FW_HEAP_SIZE_OVERRIDE_LIBOS2_MIN_MB: u32 = 64;
pub const GSP_FW_HEAP_SIZE_OVERRIDE_LIBOS2_MAX_MB: u32 = 256;
pub const GSP_FW_HEAP_SIZE_OVERRIDE_LIBOS3_BAREMETAL_MIN_MB: u32 = 88;
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 08/31] gpu: nova-core: mm: borrow BarUser for temporary BAR1 access
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (6 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 07/31] gpu: nova-core: vgpu: reserve the 48-VM WPR2 heap Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 09/31] gpu: nova-core: mm: add VramBlock Zhi Wang
` (22 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang, Alistair Popple
Temporary BAR1 access needs a borrow of the GPU-owned BarUser. Store
BarUser inline in Gpu and borrow it in BarUserAccess, avoiding a separate
allocation and reference counting for these accesses. Update the existing
self-test entry points to accept the borrowed interface.
No functional change is intended.
Cc: Alistair Popple <apopple@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 21 +++++++++------------
drivers/gpu/nova-core/mm.rs | 5 ++---
drivers/gpu/nova-core/mm/bar_user.rs | 27 ++++++++++++---------------
3 files changed, 23 insertions(+), 30 deletions(-)
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index 11d9986c4b19..97b33d4b2631 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -16,7 +16,6 @@
SizeConstants,
SZ_4K, //
},
- sync::Arc,
};
use crate::{
@@ -340,7 +339,8 @@ pub(crate) struct Gpu<'gpu> {
/// the GSP is still operational.
mm: GpuMm<'gpu>,
/// BAR1 user interface for CPU access to GPU virtual memory.
- bar_user: Arc<BarUser<'gpu>>,
+ #[pin]
+ bar_user: BarUser<'gpu>,
/// GSP and its resources.
#[pin]
gsp_resources: GspResources<'gpu>,
@@ -545,18 +545,15 @@ pub(crate) fn new<'a>(
},
// Create BAR1 user interface for CPU access to GPU virtual memory.
- bar_user: {
+ bar_user <- {
let pdb_addr = VramAddress::from_raw(gsp_resources.static_info().bar1_pde_base());
let bar1_idx = crate::driver::bar1_resource_index(pdev)?;
let bar1_size = pdev.resource_len(bar1_idx)?;
- Arc::pin_init(
- BarUser::new(
- pdb_addr,
- gsp_resources.spec.chipset,
- bar1_size,
- bar1,
- )?,
- GFP_KERNEL,
+ BarUser::new(
+ pdb_addr,
+ gsp_resources.spec.chipset,
+ bar1_size,
+ bar1,
)?
},
})
@@ -573,7 +570,7 @@ pub(crate) fn run_selftests(self: Pin<&mut Self>, pdev: &pci::Device<device::Bou
dev,
this.mm,
info.usable_fb_regions(),
- this.bar_user,
+ this.bar_user.as_ref().get_ref(),
info.bar1_pde_base(),
this.spec.chipset,
) {
diff --git a/drivers/gpu/nova-core/mm.rs b/drivers/gpu/nova-core/mm.rs
index ea85c821f0e2..a835180db6c5 100644
--- a/drivers/gpu/nova-core/mm.rs
+++ b/drivers/gpu/nova-core/mm.rs
@@ -298,8 +298,7 @@ pub(crate) mod selftest {
use kernel::{
device,
- sizes::SizeConstants,
- sync::Arc, //
+ sizes::SizeConstants, //
};
use super::*;
@@ -309,7 +308,7 @@ pub(crate) fn run(
dev: &device::Device<device::Bound>,
mm: &mut GpuMm<'_>,
mut usable_fb_regions: impl Iterator<Item = Range<u64>>,
- bar_user: &Arc<bar_user::BarUser<'_>>,
+ bar_user: &bar_user::BarUser<'_>,
bar1_pdb: u64,
chipset: Chipset,
) -> Result {
diff --git a/drivers/gpu/nova-core/mm/bar_user.rs b/drivers/gpu/nova-core/mm/bar_user.rs
index cd2d0271e64c..414c69e037ee 100644
--- a/drivers/gpu/nova-core/mm/bar_user.rs
+++ b/drivers/gpu/nova-core/mm/bar_user.rs
@@ -7,10 +7,7 @@
io::Io,
new_mutex,
prelude::*,
- sync::{
- Arc,
- Mutex, //
- },
+ sync::Mutex, //
};
use crate::{
@@ -60,12 +57,12 @@ pub(crate) fn new(
}
/// Map physical pages to a contiguous BAR1 virtual range.
- pub(crate) fn map(
- self: &Arc<Self>,
+ pub(crate) fn map<'access>(
+ &'access self,
mm: &mut GpuMm<'_>,
pfns: &[Pfn],
writable: bool,
- ) -> Result<BarUserAccess<'gpu>> {
+ ) -> Result<BarUserAccess<'access, 'gpu>> {
if pfns.is_empty() {
return Err(EINVAL);
}
@@ -73,21 +70,21 @@ pub(crate) fn map(
let mapped = vmm.map_pages(mm, pfns, None, writable)?;
Ok(BarUserAccess {
- bar_user: self.clone(),
+ bar_user: self,
mapped: Some(mapped),
})
}
}
/// Access object for a mapped BAR1 region.
-pub(crate) struct BarUserAccess<'gpu> {
- bar_user: Arc<BarUser<'gpu>>,
+pub(crate) struct BarUserAccess<'access, 'gpu> {
+ bar_user: &'access BarUser<'gpu>,
/// Cleared only after PTE invalidation and its TLB flush succeed.
mapped: Option<MappedRange>,
}
#[expect(dead_code)]
-impl BarUserAccess<'_> {
+impl BarUserAccess<'_, '_> {
/// Tear down the BAR1 mapping.
pub(crate) fn release(mut self, mm: &mut GpuMm<'_>) -> Result {
self.unmap(mm)
@@ -167,7 +164,7 @@ pub(crate) fn try_write64(&self, value: u64, offset: usize) -> Result {
}
}
-impl Drop for BarUserAccess<'_> {
+impl Drop for BarUserAccess<'_, '_> {
fn drop(&mut self) {
if self.mapped.is_some() {
kernel::pr_warn!(
@@ -185,10 +182,10 @@ fn drop(&mut self) {
/// address space. Uses the `GpuMm`'s buddy allocator to allocate page tables
/// and test pages as needed.
#[cfg(CONFIG_NOVA_CORE_SELFTESTS)]
-pub(crate) fn run_self_test(
+pub(super) fn run_self_test(
dev: &device::Device<device::Bound>,
mm: &mut GpuMm<'_>,
- bar_user: &Arc<BarUser<'_>>,
+ bar_user: &BarUser<'_>,
bar1_pdb: u64,
chipset: Chipset,
) -> Result {
@@ -401,7 +398,7 @@ pub(crate) fn run_self_test(
drop(vmm);
// Test 4: Exercise `BarUser::map()` end-to-end.
- let bar_user = Arc::pin_init(
+ let bar_user = KBox::pin_init(
BarUser::new(pdb_addr, chipset, SZ_64K.into_safe_cast(), bar1)?,
GFP_KERNEL,
)?;
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 09/31] gpu: nova-core: mm: add VramBlock
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (7 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 08/31] gpu: nova-core: mm: borrow BarUser for temporary BAR1 access Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 10/31] gpu: nova-core: mm: add VramRegion Zhi Wang
` (21 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang, Alistair Popple
vGPU memory slots need allocations at fixed offsets within usable VRAM.
Add GpuMm::alloc_vram_range and VramBlock to own these exact allocations
and return their storage to the buddy allocator on drop.
Use the buddy allocator's range validation and retain the checks for
absolute physical address overflow and alignment.
Cc: Alistair Popple <apopple@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/mm.rs | 2 +
drivers/gpu/nova-core/mm/vram.rs | 75 ++++++++++++++++++++++++++++++++
2 files changed, 77 insertions(+)
create mode 100644 drivers/gpu/nova-core/mm/vram.rs
diff --git a/drivers/gpu/nova-core/mm.rs b/drivers/gpu/nova-core/mm.rs
index a835180db6c5..a1e0bbffc460 100644
--- a/drivers/gpu/nova-core/mm.rs
+++ b/drivers/gpu/nova-core/mm.rs
@@ -67,6 +67,8 @@ macro_rules! impl_pfn_bounded {
mod regs;
pub(super) mod tlb;
pub(super) mod vmm;
+#[expect(dead_code)]
+pub(crate) mod vram;
/// GPU Memory Manager - owns all core MM components.
///
diff --git a/drivers/gpu/nova-core/mm/vram.rs b/drivers/gpu/nova-core/mm/vram.rs
new file mode 100644
index 000000000000..b6cc177825d7
--- /dev/null
+++ b/drivers/gpu/nova-core/mm/vram.rs
@@ -0,0 +1,75 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! VRAM allocation.
+
+use core::ops::Range;
+
+use kernel::{
+ gpu::buddy::{
+ AllocatedBlocks,
+ GpuBuddyAllocFlags,
+ GpuBuddyAllocMode, //
+ },
+ prelude::*,
+ ptr::Alignment,
+ sync::Arc, //
+};
+
+use crate::num::IntoSafeCast;
+
+use super::{
+ GpuMm,
+ PAGE_SIZE, //
+};
+
+/// A physically contiguous VRAM allocation.
+pub(crate) struct VramBlock {
+ _blocks: Pin<KBox<AllocatedBlocks>>,
+ address: u64,
+ size: u64,
+}
+
+impl VramBlock {
+ pub(crate) const fn address(&self) -> u64 {
+ self.address
+ }
+}
+
+impl GpuMm<'_> {
+ /// Allocates an exact byte range relative to the buddy allocator's base.
+ pub(crate) fn alloc_vram_range(&self, range: Range<u64>, align: u64) -> Result<Arc<VramBlock>> {
+ let size = range.end.checked_sub(range.start).ok_or(EINVAL)?;
+
+ let align = align.max(PAGE_SIZE.into_safe_cast());
+ let min_block_size = usize::try_from(align).map_err(|_| EOVERFLOW)?;
+ let min_block_size = Alignment::new_checked(min_block_size).ok_or(EINVAL)?;
+
+ let buddy = self.buddy();
+ let base = buddy.base_offset();
+ let address = base.checked_add(range.start).ok_or(EOVERFLOW)?;
+ base.checked_add(range.end).ok_or(EOVERFLOW)?;
+ if !address.is_multiple_of(align) {
+ return Err(EINVAL);
+ }
+
+ let blocks = KBox::pin_init(
+ buddy.alloc_blocks(
+ GpuBuddyAllocMode::Range(range),
+ size,
+ min_block_size,
+ GpuBuddyAllocFlags::default(),
+ ),
+ GFP_KERNEL,
+ )?;
+
+ Ok(Arc::new(
+ VramBlock {
+ _blocks: blocks,
+ address,
+ size,
+ },
+ GFP_KERNEL,
+ )?)
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 10/31] gpu: nova-core: mm: add VramRegion
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (8 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 09/31] gpu: nova-core: mm: add VramBlock Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 11/31] gpu: nova-core: mm: add BarMapping Zhi Wang
` (20 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang, Alistair Popple
BAR mappings do not necessarily cover an entire VRAM block. A region
layer lets a mapping select a byte range while keeping the underlying
allocation alive.
For vGPU, communication buffers occupy only part of the management heap
and need a BAR mapping of that range.
Add VramRegion to describe ranges within a shared VramBlock and support
bounds-checked subregions.
Cc: Alistair Popple <apopple@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/mm/vram.rs | 56 ++++++++++++++++++++++++++++++++
1 file changed, 56 insertions(+)
diff --git a/drivers/gpu/nova-core/mm/vram.rs b/drivers/gpu/nova-core/mm/vram.rs
index b6cc177825d7..bed90a3de8bd 100644
--- a/drivers/gpu/nova-core/mm/vram.rs
+++ b/drivers/gpu/nova-core/mm/vram.rs
@@ -34,6 +34,62 @@ impl VramBlock {
pub(crate) const fn address(&self) -> u64 {
self.address
}
+
+ /// Creates a byte range relative to this allocation, retaining its backing storage.
+ pub(crate) fn region(self: &Arc<Self>, range: Range<u64>) -> Result<VramRegion> {
+ VramRegion::new(self.clone(), range)
+ }
+}
+
+/// A byte range that keeps its VRAM allocation alive.
+#[derive(Clone)]
+pub(crate) struct VramRegion {
+ backing: Arc<VramBlock>,
+ address: u64,
+ size: u64,
+}
+
+impl VramRegion {
+ fn new(backing: Arc<VramBlock>, range: Range<u64>) -> Result<Self> {
+ if range.start >= range.end || range.end > backing.size {
+ return Err(EINVAL);
+ }
+
+ let size = range.end - range.start;
+ let address = backing.address.checked_add(range.start).ok_or(EOVERFLOW)?;
+ backing.address.checked_add(range.end).ok_or(EOVERFLOW)?;
+
+ Ok(Self {
+ backing,
+ address,
+ size,
+ })
+ }
+
+ pub(crate) const fn address(&self) -> u64 {
+ self.address
+ }
+
+ pub(crate) const fn size(&self) -> u64 {
+ self.size
+ }
+
+ /// Creates a nonempty subregion relative to this view's start.
+ pub(crate) fn subregion(&self, range: Range<u64>) -> Result<Self> {
+ if range.start >= range.end || range.end > self.size {
+ return Err(EINVAL);
+ }
+
+ let size = range.end - range.start;
+ let address = self.address.checked_add(range.start).ok_or(EOVERFLOW)?;
+ address.checked_add(size).ok_or(EOVERFLOW)?;
+
+ Ok(Self {
+ backing: self.backing.clone(),
+ address,
+ size,
+ })
+ }
}
impl GpuMm<'_> {
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 11/31] gpu: nova-core: mm: add BarMapping
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (9 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 10/31] gpu: nova-core: mm: add VramRegion Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 12/31] gpu: nova-core: gsp: add synchronous GMC transactions Zhi Wang
` (19 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang, Alistair Popple
vGPU communication buffers need CPU access to regions within their
VramBlocks. Add BarMapping to map these regions through BAR1, including
regions that start or end within a page.
Retain the backing region and borrow BarUser and the MM mutex for the
mapping's lifetime. Restrict CPU accesses to the region's byte range,
and unmap on drop using the retained MM dependency.
Provide an explicit unmap operation so owners can observe errors before
releasing resources. Keep the mapping and backing region on failure,
remember the first error and reject subsequent accesses. Owners must
retain failed mappings until device teardown; drop does not retry an
unmap that already failed.
Cc: Alistair Popple <apopple@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/mm/bar_user.rs | 145 +++++++++++++++++++++++++++
1 file changed, 145 insertions(+)
diff --git a/drivers/gpu/nova-core/mm/bar_user.rs b/drivers/gpu/nova-core/mm/bar_user.rs
index 414c69e037ee..964aee8bfdb1 100644
--- a/drivers/gpu/nova-core/mm/bar_user.rs
+++ b/drivers/gpu/nova-core/mm/bar_user.rs
@@ -7,6 +7,10 @@
io::Io,
new_mutex,
prelude::*,
+ ptr::{
+ Alignable,
+ Alignment, //
+ },
sync::Mutex, //
};
@@ -18,6 +22,7 @@
MappedRange,
Vmm, //
},
+ vram::VramRegion,
GpuMm,
Pfn,
Vfn,
@@ -176,6 +181,146 @@ fn drop(&mut self) {
}
}
+/// A BAR mapping that retains its backing VRAM region.
+#[must_use]
+pub(crate) struct BarMapping<'map, 'gpu> {
+ access: BarUserAccess<'map, 'gpu>,
+ mm: &'map Mutex<GpuMm<'gpu>>,
+ region: VramRegion,
+ region_offset: usize,
+ region_size: usize,
+ unmap_error: Option<Error>,
+}
+
+#[expect(dead_code)]
+impl<'map, 'gpu> BarMapping<'map, 'gpu> {
+ /// Maps the containing pages while restricting CPU access to the requested byte range.
+ pub(crate) fn new(
+ bar_user: &'map BarUser<'gpu>,
+ mm: &'map Mutex<GpuMm<'gpu>>,
+ region: VramRegion,
+ writable: bool,
+ ) -> Result<Self> {
+ let page_size: u64 = PAGE_SIZE.into_safe_cast();
+ let page_align = Alignment::new::<PAGE_SIZE>();
+
+ let region_start = region.address();
+ let region_end = region_start.checked_add(region.size()).ok_or(EOVERFLOW)?;
+
+ let map_start = region_start.align_down(page_align);
+ let map_end = region_end.align_up(page_align).ok_or(EOVERFLOW)?;
+ let map_size = map_end.checked_sub(map_start).ok_or(EINVAL)?;
+ let num_pages = usize::try_from(map_size / page_size).map_err(|_| EOVERFLOW)?;
+ if num_pages == 0 {
+ return Err(EINVAL);
+ }
+
+ let region_offset = usize::try_from(region_start - map_start).map_err(|_| EOVERFLOW)?;
+ let region_size = usize::try_from(region.size()).map_err(|_| EOVERFLOW)?;
+
+ let mut pfns = KVec::with_capacity(num_pages, GFP_KERNEL)?;
+ for page in 0..num_pages {
+ let page = u64::try_from(page).map_err(|_| EOVERFLOW)?;
+ let byte_offset = page.checked_mul(page_size).ok_or(EOVERFLOW)?;
+ let address = map_start.checked_add(byte_offset).ok_or(EOVERFLOW)?;
+ pfns.push(Pfn::from(VramAddress::from_raw(address)), GFP_KERNEL)?;
+ }
+
+ let access = bar_user.map(&mut mm.lock(), &pfns, writable)?;
+
+ Ok(Self {
+ access,
+ mm,
+ region,
+ region_offset,
+ region_size,
+ unmap_error: None,
+ })
+ }
+
+ pub(crate) fn region(&self) -> &VramRegion {
+ &self.region
+ }
+
+ pub(crate) fn gpu_va_addr(&self) -> Result<u64> {
+ self.access()?
+ .base()
+ .into_raw()
+ .checked_add(u64::try_from(self.region_offset).map_err(|_| EOVERFLOW)?)
+ .ok_or(EOVERFLOW)
+ }
+
+ pub(crate) const fn size(&self) -> usize {
+ self.region_size
+ }
+
+ fn access(&self) -> Result<&BarUserAccess<'map, 'gpu>> {
+ if self.access.mapped.is_none() {
+ return Err(ENODEV);
+ }
+ if let Some(error) = self.unmap_error {
+ return Err(error);
+ }
+ Ok(&self.access)
+ }
+
+ fn access_offset(&self, offset: usize, width: usize) -> Result<usize> {
+ let region_end = offset.checked_add(width).ok_or(EOVERFLOW)?;
+ if region_end > self.region_size {
+ return Err(EINVAL);
+ }
+
+ let access_offset = self.region_offset.checked_add(offset).ok_or(EOVERFLOW)?;
+ if !access_offset.is_multiple_of(width) {
+ return Err(EINVAL);
+ }
+
+ Ok(access_offset)
+ }
+
+ pub(crate) fn try_read32(&self, offset: usize) -> Result<u32> {
+ self.access()?
+ .try_read32(self.access_offset(offset, size_of::<u32>())?)
+ }
+
+ pub(crate) fn try_write32(&self, value: u32, offset: usize) -> Result {
+ self.access()?
+ .try_write32(value, self.access_offset(offset, size_of::<u32>())?)
+ }
+
+ pub(crate) fn try_write64(&self, value: u64, offset: usize) -> Result {
+ self.access()?
+ .try_write64(value, self.access_offset(offset, size_of::<u64>())?)
+ }
+
+ /// Unmap while retaining the mapping and its backing storage on failure.
+ ///
+ /// The first error is retained; later calls do not retry the operation.
+ pub(crate) fn unmap(&mut self) -> Result {
+ if let Some(error) = self.unmap_error {
+ return Err(error);
+ }
+ let result = self.access.unmap(&mut self.mm.lock());
+ if let Err(error) = result {
+ self.unmap_error = Some(error);
+ }
+ result
+ }
+}
+
+impl Drop for BarMapping<'_, '_> {
+ /// Drop unmaps the region and may sleep. Call [`Self::unmap`] to observe errors;
+ /// on failure the owner must retain this object until device teardown.
+ fn drop(&mut self) {
+ if self.unmap_error.is_some() {
+ return;
+ }
+ if let Err(error) = self.unmap() {
+ kernel::pr_err!("failed to unmap BAR1 region: {:?}\n", error);
+ }
+ }
+}
+
/// Run MM subsystem self-tests during probe.
///
/// Tests page table infrastructure and `BAR1` MMIO access using the `BAR1`
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 12/31] gpu: nova-core: gsp: add synchronous GMC transactions
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (10 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 11/31] gpu: nova-core: mm: add BarMapping Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 13/31] gpu: nova-core: gsp: wait for GMC completion events Zhi Wang
` (18 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
vGPU management queries need an owned response payload after the receive
queue releases its slot. Sending and receiving must share one queue
guard so another receiver cannot consume the reply.
Match the response flag, command ID and request sequence, then copy the
payload and return the raw firmware status. Support the default receive
timeout and a caller-selected timeout without converting firmware
rejection into a transport error.
Share the receive deadline and retry loop in CmdqInner with the existing
GMC boot response wait. Preserve each caller's matching and status
policy, consume every valid element once even when its callback fails,
and keep a single deadline after sending. Document unmatched-message
consumption and distinguish a zero response payload capacity from a
command that sends no reply.
Co-developed-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
Documentation/gpu/nova/core/interrupts.rst | 17 +-
drivers/gpu/nova-core/gsp/cmdq.rs | 174 +++++++++++++++++----
drivers/gpu/nova-core/gsp/fw.rs | 6 +-
3 files changed, 161 insertions(+), 36 deletions(-)
diff --git a/Documentation/gpu/nova/core/interrupts.rst b/Documentation/gpu/nova/core/interrupts.rst
index fa124bc8ca05..45ba788bfe27 100644
--- a/Documentation/gpu/nova/core/interrupts.rst
+++ b/Documentation/gpu/nova/core/interrupts.rst
@@ -576,9 +576,12 @@ command reply or an unsolicited event, and the two differ in the function code.
records, lifecycle notices) need no action and get no line of their own,
because the receive trace at debug level already records every message's
arrival with its sequence number, function code, and length.
-* A GMC message carries a command id in place of a function code. Only the
- ``GSP_INIT`` wait during boot claims GMC messages, so one that arrives
- anywhere else is logged at warning level and dropped.
+* A GMC message carries a command id in place of a function code. The
+ ``GSP_INIT`` wait and synchronous GMC transactions claim responses with the
+ expected command id and sequence number. These waits consume interleaved RPC
+ messages as events. Unmatched GMC messages are handled by the wait's callback
+ or logged and consumed; a queue drain with no waiting caller logs them at
+ warning level and drops them.
A command's reply must carry the RPC sequence number that nova-core wrote into
the command, as well as its function code. A message with the awaited function
@@ -600,9 +603,11 @@ the device is reset.
The polling path and the IRQ thread both read the queue under the command-queue
mutex. Replies and events share one queue and one read pointer, so one lock is
held across the whole drain. A thread waiting for a reply logs each event that
-arrives before the reply and keeps waiting. One deadline of 5 seconds applies
-to the whole wait, rather than a fresh timeout after each message, and the
-thread holds the mutex for the whole wait, so no other caller consumes the
+arrives before the reply and keeps waiting. One receive deadline applies to the
+whole wait, rather than a fresh timeout after each message. The default timeout
+is 5 seconds; GMC transactions can specify another timeout. For send-and-wait operations the receive deadline starts after sending,
+so it does not bound waiting for the mutex or for command-queue space. The thread
+holds the mutex from sending through receiving, so no other caller consumes the
message it waits for.
With one lock, a drain waits for an in-flight command's receive to finish or
diff --git a/drivers/gpu/nova-core/gsp/cmdq.rs b/drivers/gpu/nova-core/gsp/cmdq.rs
index c63cd450df41..df3d45a13a13 100644
--- a/drivers/gpu/nova-core/gsp/cmdq.rs
+++ b/drivers/gpu/nova-core/gsp/cmdq.rs
@@ -548,6 +548,16 @@ fn payload_length(&self) -> Option<usize> {
}
}
+/// Response from a GMC API command.
+pub(crate) struct GmcResponse {
+ /// Response status (`NV_STATUS` code). Zero means success.
+ #[expect(dead_code)]
+ pub(crate) status: u32,
+ /// Response payload copied out of the message queue.
+ #[expect(dead_code)]
+ pub(crate) payload: KVec<u8>,
+}
+
/// GSP command queue.
///
/// Provides the ability to send commands and receive messages from the GSP using a shared memory
@@ -688,6 +698,99 @@ pub(crate) fn send_gmc_no_wait(
.send_gmc(command_id, payload, max_response_size)
}
+ /// Sends a GMC API command and waits for its matching response.
+ ///
+ /// Uses [`Self::RECEIVE_TIMEOUT`] as the receive timeout. Firmware rejection is returned in
+ /// [`GmcResponse::status`], not as an error.
+ ///
+ /// See [`Self::send_gmc_and_receive_timeout`] for message consumption and locking behavior.
+ ///
+ /// # Errors
+ ///
+ /// Returns the same errors as [`Self::send_gmc_and_receive_timeout`].
+ #[expect(dead_code)]
+ pub(crate) fn send_gmc_and_receive(
+ &self,
+ command_id: u32,
+ payload: &[u8],
+ max_response_size: u32,
+ ) -> Result<GmcResponse> {
+ self.send_gmc_and_receive_timeout(
+ command_id,
+ payload,
+ max_response_size,
+ Self::RECEIVE_TIMEOUT,
+ )
+ }
+
+ /// Sends a GMC API command and waits up to `timeout` for its matching response.
+ ///
+ /// Matches both the command ID and the request's sequence number. A nonzero firmware status
+ /// is returned in [`GmcResponse::status`]; the caller decides how to handle rejection.
+ /// `max_response_size` advertises the response payload capacity in bytes, so zero still
+ /// permits a status-only response.
+ ///
+ /// Unmatched GMC responses and events are debug-logged and consumed. Interleaved RPC messages
+ /// are logged as events and consumed. The matching response is also consumed if copying its
+ /// payload fails.
+ ///
+ /// The queue stays locked from sending through receiving. One receive deadline starts after
+ /// sending and is not extended by other messages. It does not bound waiting for the mutex or
+ /// for space to send the request.
+ ///
+ /// # Errors
+ ///
+ /// - `EMSGSIZE` if the request exceeds the command queue's maximum element size.
+ /// - `ETIMEDOUT` if space does not become available to send the request, or if the matching
+ /// response does not arrive before the receive deadline.
+ /// - `EIO` if the command queue slot cannot hold the request headers, the receive queue is
+ /// poisoned, or a received element fails framing validation.
+ /// - `ENOMEM` if the response payload cannot be allocated.
+ ///
+ /// Errors from initializing the request headers are propagated as-is.
+ pub(crate) fn send_gmc_and_receive_timeout(
+ &self,
+ command_id: u32,
+ payload: &[u8],
+ max_response_size: u32,
+ timeout: Delta,
+ ) -> Result<GmcResponse> {
+ let mut inner = self.inner.lock();
+ let expected_sequence = inner.send_gmc(command_id, payload, max_response_size)?;
+ let dev = inner.dev;
+
+ let deadline = Instant::<Monotonic>::now() + timeout;
+ inner.await_gmc(deadline, |header, payload_0, payload_1| {
+ let header = &header.gmc;
+ if !header.is_response_to(command_id, expected_sequence) {
+ let kind = if header.is_response() {
+ "response"
+ } else {
+ "event"
+ };
+ dev_dbg!(
+ dev,
+ "GSP GMC: skip {} seq {} cmd {:#x}; want response seq {} cmd {:#x}\n",
+ kind,
+ header.sequence,
+ header.command_id(),
+ expected_sequence,
+ command_id,
+ );
+ return Ok(None);
+ }
+
+ // Each byte slice is at most `isize::MAX` bytes, so their sum fits in `usize`.
+ let mut payload = KVec::with_capacity(payload_0.len() + payload_1.len(), GFP_KERNEL)?;
+ payload.extend_from_slice(payload_0, GFP_KERNEL)?;
+ payload.extend_from_slice(payload_1, GFP_KERNEL)?;
+ Ok(Some(GmcResponse {
+ status: header.status(),
+ payload,
+ }))
+ })
+ }
+
/// Sends a GMC API request that GSP-RM does not answer.
///
/// # Errors
@@ -1395,6 +1498,35 @@ fn receive_gmc_and_dispatch<R>(
})
}
+ /// Waits until `handler` returns a value for a GMC element, using one receive deadline.
+ ///
+ /// Unclaimed elements do not extend the deadline. Every valid element is consumed, including
+ /// one for which `handler` returns an error; RPC elements are logged as events and consumed.
+ /// The caller retains the queue guard throughout the wait and the handler calls.
+ ///
+ /// # Errors
+ ///
+ /// - `ETIMEDOUT` if no GMC element satisfies `handler` before `deadline`.
+ /// - `EIO` if the queue is poisoned or an element fails framing validation.
+ ///
+ /// Errors from `handler` are propagated as-is.
+ fn await_gmc<R>(
+ &mut self,
+ deadline: Instant<Monotonic>,
+ mut handler: impl FnMut(&GspGmcMsgElement, &[u8], &[u8]) -> Result<Option<R>>,
+ ) -> Result<R> {
+ loop {
+ let remaining = deadline - Instant::<Monotonic>::now();
+ if remaining.is_negative() {
+ return Err(ETIMEDOUT);
+ }
+
+ if let Some(value) = self.receive_gmc_and_dispatch(remaining, &mut handler)? {
+ return Ok(value);
+ }
+ }
+ }
+
/// Waits for the response to the GMC request with command id `command_id` and RPC sequence
/// number `sequence`, up to [`Cmdq::RECEIVE_TIMEOUT`] from the call.
///
@@ -1420,35 +1552,23 @@ fn await_gmc_response<R>(
) -> Result<R> {
let dev = self.dev;
let deadline = Instant::<Monotonic>::now() + Cmdq::RECEIVE_TIMEOUT;
- loop {
- let remaining = deadline - Instant::<Monotonic>::now();
- if remaining.is_negative() {
- break Err(ETIMEDOUT);
+ self.await_gmc(deadline, |header, payload_0, payload_1| {
+ if !header.gmc.is_response_to(command_id, sequence) {
+ return on_other(header, payload_0, payload_1).map(|()| None);
}
- let response =
- self.receive_gmc_and_dispatch(remaining, |header, payload_0, payload_1| {
- if !header.gmc.is_response_to(command_id, sequence) {
- return on_other(header, payload_0, payload_1).map(|()| None);
- }
-
- let status = header.gmc.status();
- if status != 0 {
- dev_err!(
- dev,
- "GSP GMC: command 0x{:x} failed, status={:#x}\n",
- command_id,
- status
- );
- return Err(EIO);
- }
-
- decode(payload_0, payload_1).map(Some)
- })?;
-
- if let Some(response) = response {
- break Ok(response);
+ let status = header.gmc.status();
+ if status != 0 {
+ dev_err!(
+ dev,
+ "GSP GMC: command 0x{:x} failed, status={:#x}\n",
+ command_id,
+ status
+ );
+ return Err(EIO);
}
- }
+
+ decode(payload_0, payload_1).map(Some)
+ })
}
}
diff --git a/drivers/gpu/nova-core/gsp/fw.rs b/drivers/gpu/nova-core/gsp/fw.rs
index 90bb324666e5..71732418e7a5 100644
--- a/drivers/gpu/nova-core/gsp/fw.rs
+++ b/drivers/gpu/nova-core/gsp/fw.rs
@@ -814,7 +814,7 @@ pub(crate) fn status(&self) -> u32 {
/// sequence number `sequence`.
pub(crate) fn is_response_to(&self, command_id: u32, sequence: u32) -> bool {
self.is_response()
- && self.command_id() == command_id
+ && self.command_id() == (command_id & GMCAPI_COMMAND_ID_MASK)
&& self.sequence == u64::from(sequence)
}
}
@@ -911,8 +911,8 @@ impl GspGmcMsgElement {
/// Creates the queue element header and the GMC API header of a request that carries
/// `payload_size` bytes of payload under the RPC sequence number `sequence`.
///
- /// `max_response_size` is the largest response that the sender accepts, and zero for a request
- /// that GSP-RM does not answer.
+ /// `max_response_size` is the response payload capacity in bytes. Zero permits a status-only
+ /// response; whether GSP-RM sends a response at all is determined by the command's protocol.
///
/// # Errors
///
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 13/31] gpu: nova-core: gsp: wait for GMC completion events
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (11 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 12/31] gpu: nova-core: gsp: add synchronous GMC transactions Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 14/31] gpu: nova-core: vgpu: add r000 plugin bindings Zhi Wang
` (17 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
GSP plugin shutdown completes through a separate event carrying the
GFID. Releasing the queue guard after sending would let another receiver
consume that completion.
Keep the guard from sending through the matching event and use the
shared Inner receive deadline. Offer GMC events to the completion
predicate, pass unmatched events to the caller's handler, and consume
interleaved responses and RPC messages. Represent completion with a
value and all continuing cases with None.
Propagate callback errors after consuming the current valid element.
Document callback non-reentrancy and that the receive timeout starts
after sending, excluding mutex acquisition and command-queue space
waits.
Co-developed-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
Documentation/gpu/nova/core/interrupts.rst | 6 ++-
drivers/gpu/nova-core/gsp/cmdq.rs | 55 ++++++++++++++++++++++
drivers/gpu/nova-core/gsp/fw.rs | 7 +++
3 files changed, 66 insertions(+), 2 deletions(-)
diff --git a/Documentation/gpu/nova/core/interrupts.rst b/Documentation/gpu/nova/core/interrupts.rst
index 45ba788bfe27..f78b1294905b 100644
--- a/Documentation/gpu/nova/core/interrupts.rst
+++ b/Documentation/gpu/nova/core/interrupts.rst
@@ -578,7 +578,8 @@ command reply or an unsolicited event, and the two differ in the function code.
arrival with its sequence number, function code, and length.
* A GMC message carries a command id in place of a function code. The
``GSP_INIT`` wait and synchronous GMC transactions claim responses with the
- expected command id and sequence number. These waits consume interleaved RPC
+ expected command id and sequence number. A GMC completion-event wait uses its
+ caller's predicate to select an event. These waits consume interleaved RPC
messages as events. Unmatched GMC messages are handled by the wait's callback
or logged and consumed; a queue drain with no waiting caller logs them at
warning level and drops them.
@@ -605,7 +606,8 @@ mutex. Replies and events share one queue and one read pointer, so one lock is
held across the whole drain. A thread waiting for a reply logs each event that
arrives before the reply and keeps waiting. One receive deadline applies to the
whole wait, rather than a fresh timeout after each message. The default timeout
-is 5 seconds; GMC transactions can specify another timeout. For send-and-wait operations the receive deadline starts after sending,
+is 5 seconds; GMC transactions and completion-event waits can specify another
+timeout. For send-and-wait operations the receive deadline starts after sending,
so it does not bound waiting for the mutex or for command-queue space. The thread
holds the mutex from sending through receiving, so no other caller consumes the
message it waits for.
diff --git a/drivers/gpu/nova-core/gsp/cmdq.rs b/drivers/gpu/nova-core/gsp/cmdq.rs
index df3d45a13a13..43038c7fe823 100644
--- a/drivers/gpu/nova-core/gsp/cmdq.rs
+++ b/drivers/gpu/nova-core/gsp/cmdq.rs
@@ -791,6 +791,61 @@ pub(crate) fn send_gmc_and_receive_timeout(
})
}
+ /// Sends an asynchronous GMC command and waits atomically for its event.
+ ///
+ /// The command queue remains locked from the send through the matching
+ /// event, preventing another transaction from consuming its completion.
+ /// GMC events are passed to `predicate`, then to `handler` if the predicate returns false.
+ /// GMC responses are debug-logged and consumed without invoking either callback. Interleaved
+ /// RPC messages are logged as events and consumed.
+ ///
+ /// One receive deadline starts after sending and is not extended by other messages. It does
+ /// not bound waiting for the mutex or for space to send the request.
+ ///
+ /// Both callbacks run with the queue locked and must not reenter this queue or reset it.
+ /// Their second argument is the raw `max_resp_or_status` word of the event header.
+ ///
+ /// # Errors
+ ///
+ /// - `EMSGSIZE` if the request exceeds the command queue's maximum element size.
+ /// - `ETIMEDOUT` if space does not become available to send the request, or if no event
+ /// satisfies `predicate` before the receive deadline.
+ /// - `EIO` if the command queue slot cannot hold the request headers, the receive queue is
+ /// poisoned, or a received element fails framing validation.
+ ///
+ /// Errors from initializing the request headers and from either callback are propagated
+ /// as-is. The current valid element is consumed before a callback error is returned.
+ #[expect(dead_code)]
+ pub(crate) fn send_gmc_and_wait_event(
+ &self,
+ command_id: u32,
+ payload: &[u8],
+ timeout: Delta,
+ mut predicate: impl FnMut(u32, u32, u64, &[u8], &[u8]) -> Result<bool>,
+ mut handler: impl FnMut(u32, u32, u64, &[u8], &[u8]) -> Result,
+ ) -> Result {
+ let mut inner = self.inner.lock();
+ inner.send_gmc(command_id, payload, 0)?;
+ let deadline = Instant::<Monotonic>::now() + timeout;
+
+ inner.await_gmc(deadline, |header, payload_0, payload_1| {
+ let header = &header.gmc;
+ if header.is_response() {
+ return Ok(None);
+ }
+
+ let command = header.command_id();
+ let max_resp_or_status = header.raw_status_word();
+ let sequence = header.sequence;
+ if predicate(command, max_resp_or_status, sequence, payload_0, payload_1)? {
+ return Ok(Some(()));
+ }
+
+ handler(command, max_resp_or_status, sequence, payload_0, payload_1)?;
+ Ok(None)
+ })
+ }
+
/// Sends a GMC API request that GSP-RM does not answer.
///
/// # Errors
diff --git a/drivers/gpu/nova-core/gsp/fw.rs b/drivers/gpu/nova-core/gsp/fw.rs
index 71732418e7a5..a416826737bb 100644
--- a/drivers/gpu/nova-core/gsp/fw.rs
+++ b/drivers/gpu/nova-core/gsp/fw.rs
@@ -802,6 +802,13 @@ pub(crate) fn sequence_number(&self) -> u64 {
self.sequence & !GMC_EVENT_SEQUENCE_BASE
}
+ /// Returns the raw `max_resp_or_status` word carried by a GMC header.
+ ///
+ /// Its meaning for an event is defined by that event's command.
+ pub(crate) fn raw_status_word(&self) -> u32 {
+ self.max_resp_or_status
+ }
+
/// Returns the `NV_STATUS` that a response carries.
///
/// The value is meaningful only when [`Self::is_response`] is `true`. In a request, the same
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 14/31] gpu: nova-core: vgpu: add r000 plugin bindings
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (12 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 13/31] gpu: nova-core: gsp: wait for GMC completion events Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 15/31] gpu: nova-core: vgpu: add VRAM slot allocator Zhi Wang
` (16 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The GSP plugin communication area and RPC protocol are defined by the r000
firmware interface. Keeping local copies of that ABI risks letting the vGPU
manager drift from the firmware layout.
Extend the existing r000 binding set with the nova-core subset of
dev_vgpu_gsp_shared.h. The bindings expose the control and response
region types, message IDs, and firmware-defined region sizes. Keep the
raw declarations private and expose only the symbols needed by the GSP
plugin communication layer.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
.../gpu/nova-core/gsp/fw/r000_00/bindings.rs | 173 ++++++++++++++++++
1 file changed, 173 insertions(+)
diff --git a/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs b/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
index fa17d3e2edf6..3cc718bb2ab8 100644
--- a/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
+++ b/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
@@ -1,4 +1,5 @@
// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#[repr(C)]
#[derive(Default)]
@@ -850,3 +851,175 @@ fn default() -> Self {
}
}
}
+pub const GSP_PLUGIN_BOOTLOADED: u32 = 1315261039;
+pub const VGPU_CPU_GSP_CTRL_BUFF_VERSION: u32 = 2;
+pub const VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE: u32 = 4096;
+pub const VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE: u32 = 4096;
+pub const VGPU_CPU_GSP_MESSAGE_BUFF_REGION_SIZE: u32 = 4096;
+pub const VGPU_CPU_GSP_MIGRATION_BUFF_REGION_SIZE: u32 = 2097152;
+pub const VGPU_CPU_GSP_ERROR_BUFF_REGION_SIZE: u32 = 4096;
+pub const VGPU_CPU_GSP_INIT_TASK_LOG_BUFF_REGION_SIZE: u32 = 131072;
+pub const VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE: u32 = 262144;
+pub const VGPU_CPU_GSP_KERNEL_TASK_LOG_BUFF_REGION_SIZE: u32 = 65536;
+pub const VGPU_CPU_GSP_GUEST_RPC_TRACE_BUFF_REGION_SIZE: u32 = 65536;
+pub const VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE: u32 = 2637824;
+pub type VGPU_CPU_GSP_BOOL = u32_;
+#[repr(C)]
+#[derive(Debug, Default, Copy, Clone, MaybeZeroable)]
+pub struct VGPU_CPU_GSP_VGX_VERSION {
+ pub major_number: u32_,
+ pub minor_number: u32_,
+}
+#[repr(C)]
+#[derive(Debug, Copy, Clone, MaybeZeroable)]
+pub struct VGPU_CPU_GSP_GUEST_INFO {
+ pub vgx_version: VGPU_CPU_GSP_VGX_VERSION,
+ pub guest_driver_version_buffer_length: u32_,
+ pub guest_version_buffer_length: u32_,
+ pub guest_title_buffer_length: u32_,
+ pub guest_changelist_number: u32_,
+ pub guest_driver_version_buffer: [ffi::c_char; 256usize],
+ pub guest_version_buffer: [ffi::c_char; 256usize],
+ pub guest_title_buffer: [ffi::c_char; 256usize],
+ pub guest_branch_buffer: [ffi::c_char; 256usize],
+}
+impl Default for VGPU_CPU_GSP_GUEST_INFO {
+ fn default() -> Self {
+ let mut s = ::core::mem::MaybeUninit::<Self>::uninit();
+ unsafe {
+ ::core::ptr::write_bytes(s.as_mut_ptr(), 0, 1);
+ s.assume_init()
+ }
+ }
+}
+#[repr(C)]
+#[derive(Copy, Clone, MaybeZeroable)]
+pub union VGPU_CPU_GSP_CTRL_BUFF_REGION {
+ pub buf: [u8_; 4096usize],
+ pub __bindgen_anon_1: VGPU_CPU_GSP_CTRL_BUFF_REGION__bindgen_ty_1,
+}
+#[repr(C)]
+#[derive(Debug, Default, Copy, Clone, MaybeZeroable)]
+pub struct VGPU_CPU_GSP_CTRL_BUFF_REGION__bindgen_ty_1 {
+ pub version: u32_,
+ pub message_type: u32_,
+ pub message_seq_num: u32_,
+ pub __bindgen_padding_0: [u8; 4usize],
+ pub response_buff_offset: u64_,
+ pub message_buff_offset: u64_,
+ pub migration_buff_offset: u64_,
+ pub error_buff_offset: u64_,
+ pub guest_rpc_trace_buff_offset: u64_,
+ pub migration_buf_cpu_access_offset: u32_,
+ pub is_migration_in_progress: u8_,
+ pub __bindgen_padding_1: [u8; 3usize],
+ pub error_buff_cpu_get_idx: u32_,
+ pub guest_rpc_trace_buff_cpu_get_idx: u32_,
+ pub attached_vgpu_count: u32_,
+ pub is_gr_init_done: u8_,
+ pub __bindgen_padding_2: [u8; 3usize],
+ pub host_info: [VGPU_CPU_GSP_CTRL_BUFF_REGION__bindgen_ty_1__bindgen_ty_1; 16usize],
+}
+#[repr(C)]
+#[derive(Debug, Default, Copy, Clone, MaybeZeroable)]
+pub struct VGPU_CPU_GSP_CTRL_BUFF_REGION__bindgen_ty_1__bindgen_ty_1 {
+ pub vgpu_type_id: u32_,
+ pub host_gpu_pci_id: u32_,
+ pub pci_dev_id: u32_,
+ pub vgpu_uuid: [u8_; 16usize],
+}
+impl Default for VGPU_CPU_GSP_CTRL_BUFF_REGION {
+ fn default() -> Self {
+ let mut s = ::core::mem::MaybeUninit::<Self>::uninit();
+ unsafe {
+ ::core::ptr::write_bytes(s.as_mut_ptr(), 0, 1);
+ s.assume_init()
+ }
+ }
+}
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_VERSION_NEGOTIATION: MESSAGE = 1;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_SETUP_CONFIG_PARAMS_AND_INIT: MESSAGE = 2;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_RESET: MESSAGE = 3;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_STOP_WORK: MESSAGE = 4;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_CANCEL_STOP: MESSAGE = 5;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_SAVE_STATE: MESSAGE = 6;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_CANCEL_SAVE: MESSAGE = 7;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_RESTORE_STATE: MESSAGE = 8;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_RESTORE_DEFERRED_STATE: MESSAGE = 9;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MIGRATION_RESUME_WORK: MESSAGE = 10;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_CONSOLE_VNC_STATE: MESSAGE = 11;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_VF_BAR0_REG_ACCESS: MESSAGE = 12;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_UPDATE_BME_STATE: MESSAGE = 13;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_RESET_MIGRATION_BUFFER_PTR: MESSAGE = 14;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_RELEASE_CLIENT_DATABASE: MESSAGE = 15;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_CHECK_IS_ALIVE: MESSAGE = 16;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_SEND_STATIC_INFO: MESSAGE = 17;
+pub const MESSAGE_NV_VGPU_CPU_RPC_MSG_MAX: MESSAGE = 18;
+pub type MESSAGE = ffi::c_uint;
+#[repr(C)]
+#[derive(Debug, Default, Copy, Clone, MaybeZeroable)]
+pub struct VGPU_CPU_GSP_DISPLAYLESS_SURFACE {
+ pub sequence_update_start: u64_,
+ pub sequence_update_end: u64_,
+ pub effective_fb_page_size: u32_,
+ pub rect_width: u32_,
+ pub rect_height: u32_,
+ pub surface_width: u32_,
+ pub surface_height: u32_,
+ pub surface_size: u32_,
+ pub surface_offset: u32_,
+ pub surface_format: u32_,
+ pub surface_kind: u32_,
+ pub surface_pitch: u32_,
+ pub surface_type: u32_,
+ pub surface_block_height: u8_,
+ pub __bindgen_padding_0: [u8; 3usize],
+ pub is_blanking_enabled: VGPU_CPU_GSP_BOOL,
+ pub is_flip_pending: VGPU_CPU_GSP_BOOL,
+ pub is_free_pending: VGPU_CPU_GSP_BOOL,
+ pub is_memory_blocklinear: VGPU_CPU_GSP_BOOL,
+}
+#[repr(C)]
+#[derive(Copy, Clone, MaybeZeroable)]
+pub union VGPU_CPU_GSP_RESPONSE_BUFF_REGION {
+ pub buf: [u8_; 4096usize],
+ pub __bindgen_anon_1: VGPU_CPU_GSP_RESPONSE_BUFF_REGION__bindgen_ty_1,
+}
+#[repr(C)]
+#[derive(Debug, Copy, Clone, MaybeZeroable)]
+pub struct VGPU_CPU_GSP_RESPONSE_BUFF_REGION__bindgen_ty_1 {
+ pub message_seq_num_received: u32_,
+ pub message_seq_num_processed: u32_,
+ pub result_code: u32_,
+ pub guest_rpc_version: u32_,
+ pub migration_buf_gsp_access_offset: u32_,
+ pub migration_state_save_complete: u32_,
+ pub is_migration_allowed: VGPU_CPU_GSP_BOOL,
+ pub __bindgen_padding_0: [u8; 4usize],
+ pub surface: [VGPU_CPU_GSP_DISPLAYLESS_SURFACE; 4usize],
+ pub error_buff_gsp_put_idx: u32_,
+ pub grid_license_state: u32_,
+ pub guest_os_type: u32_,
+ pub frl_config: u32_,
+ pub guest_info: VGPU_CPU_GSP_GUEST_INFO,
+ pub is_guest_info_populated: VGPU_CPU_GSP_BOOL,
+ pub guest_rpc_trace_buff_gsp_put_idx: u32_,
+}
+impl Default for VGPU_CPU_GSP_RESPONSE_BUFF_REGION__bindgen_ty_1 {
+ fn default() -> Self {
+ let mut s = ::core::mem::MaybeUninit::<Self>::uninit();
+ unsafe {
+ ::core::ptr::write_bytes(s.as_mut_ptr(), 0, 1);
+ s.assume_init()
+ }
+ }
+}
+impl Default for VGPU_CPU_GSP_RESPONSE_BUFF_REGION {
+ fn default() -> Self {
+ let mut s = ::core::mem::MaybeUninit::<Self>::uninit();
+ unsafe {
+ ::core::ptr::write_bytes(s.as_mut_ptr(), 0, 1);
+ s.assume_init()
+ }
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 15/31] gpu: nova-core: vgpu: add VRAM slot allocator
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (13 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 14/31] gpu: nova-core: vgpu: add r000 plugin bindings Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 16/31] gpu: nova-core: gsp: factor out NVKV payload conversion Zhi Wang
` (15 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
From: Alok Kumar <alkumar@nvidia.com>
A vGPU type specifies the guest VRAM and GSP plugin management-heap
sizes for each instance. Allocating these regions independently fragments
VRAM and does not preserve a stable layout for the vGPU type.
Add a standalone slot allocator that reserves one exact VRAM range and
tracks assignments in a shared, mutex-protected bitmap. Validate the
vGPU type's layout and reject an allocation request whose layout differs
from the active pool.
Lay out all guest VRAM slots first and all management-heap slots
second, pairing them by bitmap index. Require the guest VRAM stride and
pool base to satisfy the required VMMU-segment alignment.
Return an owned slot that clears its bitmap entry on drop. Keep the
bitmap alive while slots exist and release the entry if constructing
the VRAM regions fails. Require users to stop device accesses and remove
mappings before dropping a slot.
Signed-off-by: Alok Kumar <alkumar@nvidia.com>
Co-developed-by: Zhi Wang <zhiw@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/vgpu.rs | 1 +
drivers/gpu/nova-core/vgpu/vram.rs | 181 +++++++++++++++++++++++++++++
2 files changed, 182 insertions(+)
create mode 100644 drivers/gpu/nova-core/vgpu/vram.rs
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 5cd82adeb0f8..1402e37541b7 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -21,6 +21,7 @@
};
mod hal;
+mod vram;
/// vGPU state detected during GPU construction.
#[derive(Debug, Clone, Copy)]
diff --git a/drivers/gpu/nova-core/vgpu/vram.rs b/drivers/gpu/nova-core/vgpu/vram.rs
new file mode 100644
index 000000000000..65e6a945ea6f
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/vram.rs
@@ -0,0 +1,181 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! VRAM slot allocation for vGPU instances.
+
+use kernel::{
+ bitmap::BitmapVec,
+ prelude::*,
+ sync::{
+ new_mutex,
+ Arc,
+ Mutex, //
+ }, //
+};
+
+use crate::mm::{
+ vram::{
+ VramBlock,
+ VramRegion, //
+ },
+ GpuMm, //
+};
+
+const VRAM_SLOT_MIN_ALIGN: u64 = 4096;
+
+#[derive(Clone, Copy, PartialEq)]
+pub(super) struct VgpuVramLayout {
+ pub(super) type_id: u32,
+ pub(super) max_slots: u32,
+ pub(super) fb_size: u64,
+ pub(super) heap_size: u64,
+ pub(super) fb_align: u64,
+}
+
+impl VgpuVramLayout {
+ fn validated(mut self) -> Result<Self> {
+ if self.max_slots == 0 || self.fb_size == 0 || self.heap_size == 0 {
+ return Err(EINVAL);
+ }
+ self.fb_align = core::cmp::max(self.fb_align, VRAM_SLOT_MIN_ALIGN);
+ if !self.fb_align.is_power_of_two() {
+ return Err(EINVAL);
+ }
+ if self.fb_size & (self.fb_align - 1) != 0
+ || self.heap_size & (VRAM_SLOT_MIN_ALIGN - 1) != 0
+ {
+ return Err(EINVAL);
+ }
+ Ok(self)
+ }
+}
+
+/// Slot bitmap shared by the allocator and its outstanding slots.
+#[pin_data]
+struct SlotBitmap {
+ #[pin]
+ used: Mutex<BitmapVec>,
+}
+
+impl SlotBitmap {
+ fn new(count: usize) -> Result<Arc<Self>> {
+ let used = BitmapVec::new(count, GFP_KERNEL)?;
+ Arc::pin_init(
+ pin_init!(Self {
+ used <- new_mutex!(used),
+ }),
+ GFP_KERNEL,
+ )
+ }
+
+ fn alloc(self: &Arc<Self>) -> Result<Slot> {
+ let mut used = self.used.lock();
+ let index = used.next_zero_bit(0).ok_or(ENOSPC)?;
+ used.set_bit(index);
+ Ok(Slot {
+ bitmap: self.clone(),
+ index,
+ })
+ }
+
+ fn is_empty(&self) -> bool {
+ self.used.lock().last_bit().is_none()
+ }
+}
+
+/// A slot that clears its bitmap entry on drop.
+struct Slot {
+ bitmap: Arc<SlotBitmap>,
+ index: usize,
+}
+
+impl Drop for Slot {
+ fn drop(&mut self) {
+ let mut used = self.bitmap.used.lock();
+ debug_assert_eq!(used.next_bit(self.index), Some(self.index));
+ used.clear_bit(self.index);
+ }
+}
+
+/// A VRAM slot that clears its bitmap entry when dropped.
+///
+/// All device accesses and mappings of its regions must end before dropping the slot.
+/// Clearing the entry locks a sleeping mutex, so dropping the slot may sleep.
+#[must_use]
+#[expect(dead_code)]
+pub(super) struct VgpuVramSlot {
+ pub(super) fbmem: VramRegion,
+ pub(super) mgmt_heap: VramRegion,
+ _slot: Slot,
+}
+
+pub(super) struct VgpuVramSlotAllocator {
+ backing: Arc<VramBlock>,
+ layout: VgpuVramLayout,
+ fb_region_size: u64,
+ slot_bitmap: Arc<SlotBitmap>,
+}
+
+#[expect(dead_code)]
+impl VgpuVramSlotAllocator {
+ pub(super) fn new(mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<Self> {
+ let layout = layout.validated()?;
+ let max_slots = u64::from(layout.max_slots);
+ // Keep guest VRAM slots contiguous so each starts at `fb_align`;
+ // interleaving page-aligned heaps could misalign later slots.
+ let fb_region_size = layout.fb_size.checked_mul(max_slots).ok_or(EINVAL)?;
+ let heap_region_size = layout.heap_size.checked_mul(max_slots).ok_or(EINVAL)?;
+ let pool_size = fb_region_size.checked_add(heap_region_size).ok_or(EINVAL)?;
+
+ let slot_bitmap = SlotBitmap::new(usize::try_from(layout.max_slots).map_err(|_| EINVAL)?)?;
+ let backing = mm.alloc_vram_range(0..pool_size, VRAM_SLOT_MIN_ALIGN)?;
+ if !backing.address().is_multiple_of(layout.fb_align) {
+ return Err(EINVAL);
+ }
+
+ Ok(Self {
+ backing,
+ layout,
+ fb_region_size,
+ slot_bitmap,
+ })
+ }
+
+ pub(super) fn matches_layout(&self, layout: VgpuVramLayout) -> Result<bool> {
+ Ok(self.layout == layout.validated()?)
+ }
+
+ pub(super) fn alloc(&mut self, layout: VgpuVramLayout) -> Result<VgpuVramSlot> {
+ if !self.matches_layout(layout)? {
+ return Err(EBUSY);
+ }
+
+ let slot = self.slot_bitmap.alloc()?;
+ let index = u64::try_from(slot.index).map_err(|_| EINVAL)?;
+
+ let fb_size = self.layout.fb_size;
+ let fb_offset = fb_size.checked_mul(index).ok_or(EINVAL)?;
+ let fb_end = fb_offset.checked_add(fb_size).ok_or(EINVAL)?;
+
+ let heap_size = self.layout.heap_size;
+ let heap_slot_offset = heap_size.checked_mul(index).ok_or(EINVAL)?;
+ let heap_offset = self
+ .fb_region_size
+ .checked_add(heap_slot_offset)
+ .ok_or(EINVAL)?;
+ let heap_end = heap_offset.checked_add(heap_size).ok_or(EINVAL)?;
+
+ let fbmem = self.backing.region(fb_offset..fb_end)?;
+ let mgmt_heap = self.backing.region(heap_offset..heap_end)?;
+
+ Ok(VgpuVramSlot {
+ fbmem,
+ mgmt_heap,
+ _slot: slot,
+ })
+ }
+
+ pub(super) fn is_empty(&self) -> bool {
+ self.slot_bitmap.is_empty()
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 16/31] gpu: nova-core: gsp: factor out NVKV payload conversion
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (14 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 15/31] gpu: nova-core: vgpu: add VRAM slot allocator Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 17/31] gpu: nova-core: vgpu: query VF assignments and properties Zhi Wang
` (14 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Both GSP initialization and vGPU property queries decode their replies as
NVKV streams.
Move the byte-to-word conversion helper into the NVKV module and make
that module accessible to the vGPU firmware interface.
No functional change is intended.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gsp.rs | 2 +-
drivers/gpu/nova-core/gsp/commands.rs | 17 ++--------------
drivers/gpu/nova-core/gsp/nvkv.rs | 28 +++++++++++++++++++++++++++
3 files changed, 31 insertions(+), 16 deletions(-)
diff --git a/drivers/gpu/nova-core/gsp.rs b/drivers/gpu/nova-core/gsp.rs
index 3dbc37a3b35b..25b4bf6eb848 100644
--- a/drivers/gpu/nova-core/gsp.rs
+++ b/drivers/gpu/nova-core/gsp.rs
@@ -28,7 +28,7 @@
pub(crate) mod cmdq;
pub(crate) mod commands;
mod fw;
-mod nvkv;
+pub(crate) mod nvkv;
mod regs;
pub(crate) use fw::{
diff --git a/drivers/gpu/nova-core/gsp/commands.rs b/drivers/gpu/nova-core/gsp/commands.rs
index 97e32f81a39b..742e46efd413 100644
--- a/drivers/gpu/nova-core/gsp/commands.rs
+++ b/drivers/gpu/nova-core/gsp/commands.rs
@@ -22,6 +22,7 @@
GMCAPI_CMD_GSP_SUSPEND, //
},
nvkv::{
+ nvkv_words,
Decoder,
Encodable,
EncodedStream,
@@ -30,7 +31,6 @@
},
GspBootContext, //
},
- sbuffer::SBufferIter,
vgpu::VgpuState, //
};
@@ -132,20 +132,7 @@ pub(crate) fn gsp_init(
/// or omits a required key, or the FIFO engine count exceeds the supported table capacity.
/// - `ENOMEM` if the words or the decoded regions cannot be allocated.
fn decode_gsp_init_reply(payload_0: &[u8], payload_1: &[u8]) -> Result<GspStaticInfo> {
- const WORD_SIZE: usize = size_of::<u64>();
-
- let len = payload_0.len() + payload_1.len();
- if len % WORD_SIZE != 0 {
- return Err(EINVAL);
- }
-
- let mut words = KVVec::with_capacity(len / WORD_SIZE, GFP_KERNEL)?;
- let mut bytes = SBufferIter::new_reader([payload_0, payload_1]);
- for _ in 0..len / WORD_SIZE {
- let mut word = [0u8; WORD_SIZE];
- bytes.read_exact(&mut word)?;
- words.push(u64::from_le_bytes(word), GFP_KERNEL)?;
- }
+ let words = nvkv_words(payload_0, payload_1)?;
let decoder = Decoder::new(&words, UnknownKeyPolicy::Ignore);
let mut schema = GspInitResponseSchema::default();
diff --git a/drivers/gpu/nova-core/gsp/nvkv.rs b/drivers/gpu/nova-core/gsp/nvkv.rs
index e7a9549919fb..05b1ba5a4f39 100644
--- a/drivers/gpu/nova-core/gsp/nvkv.rs
+++ b/drivers/gpu/nova-core/gsp/nvkv.rs
@@ -27,6 +27,34 @@
};
use zerocopy::Immutable;
+use crate::sbuffer::SBufferIter;
+
+/// Joins the two halves of a wrapped payload into the `u64` words an NVKV stream is made of.
+///
+/// # Errors
+///
+/// - `EINVAL` if the combined length is not a whole number of words.
+/// - `ENOMEM` if the buffer cannot be allocated.
+pub(crate) fn nvkv_words(payload_0: &[u8], payload_1: &[u8]) -> Result<KVVec<u64>> {
+ const WORD_SIZE: usize = size_of::<u64>();
+
+ // Each byte slice is at most `isize::MAX` bytes, so their sum fits in `usize`.
+ let len = payload_0.len() + payload_1.len();
+ if len % WORD_SIZE != 0 {
+ return Err(EINVAL);
+ }
+
+ let mut words = KVVec::with_capacity(len / WORD_SIZE, GFP_KERNEL)?;
+ let mut bytes = SBufferIter::new_reader([payload_0, payload_1]);
+ for _ in 0..len / WORD_SIZE {
+ let mut word = [0u8; WORD_SIZE];
+ bytes.read_exact(&mut word)?;
+ words.push(u64::from_le_bytes(word), GFP_KERNEL)?;
+ }
+
+ Ok(words)
+}
+
mod encode;
pub(crate) use encode::*;
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 17/31] gpu: nova-core: vgpu: query VF assignments and properties
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (15 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 16/31] gpu: nova-core: gsp: factor out NVKV payload conversion Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 18/31] gpu: nova-core: vgpu: add instance create/destroy Zhi Wang
` (13 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Creating an instance requires its VF assignment and the corresponding
vGPU properties before reserving resources.
Call Cmdq's synchronous GMC interface with each query's command ID,
request payload and receive capacity. Check the matching reply's
firmware status before decoding it, and report the command and raw
NV_STATUS when firmware rejects the request.
Decode the VF assignment and NVKV property replies at their query entry
points. Validate the returned type ID and nonzero instance limit, and
initialize the decoded properties on the heap. Keep the raw queue
interface available to firmware-control clients.
Co-developed-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gsp.rs | 1 +
drivers/gpu/nova-core/gsp/cmdq.rs | 3 -
drivers/gpu/nova-core/gsp/fw.rs | 2 +-
drivers/gpu/nova-core/vgpu.rs | 2 +
drivers/gpu/nova-core/vgpu/commands.rs | 91 ++++++++++++++++++++++
drivers/gpu/nova-core/vgpu/fw.rs | 17 ++++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 95 +++++++++++++++++++++++
7 files changed, 207 insertions(+), 4 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/commands.rs
create mode 100644 drivers/gpu/nova-core/vgpu/fw.rs
create mode 100644 drivers/gpu/nova-core/vgpu/fw/commands.rs
diff --git a/drivers/gpu/nova-core/gsp.rs b/drivers/gpu/nova-core/gsp.rs
index 25b4bf6eb848..45645bc7c9d8 100644
--- a/drivers/gpu/nova-core/gsp.rs
+++ b/drivers/gpu/nova-core/gsp.rs
@@ -32,6 +32,7 @@
mod regs;
pub(crate) use fw::{
+ r000_00 as bindings,
GspFmcBootParams,
GspFwWprMeta,
LibosMemoryRegionInitArgument,
diff --git a/drivers/gpu/nova-core/gsp/cmdq.rs b/drivers/gpu/nova-core/gsp/cmdq.rs
index 43038c7fe823..abd9edadff64 100644
--- a/drivers/gpu/nova-core/gsp/cmdq.rs
+++ b/drivers/gpu/nova-core/gsp/cmdq.rs
@@ -551,10 +551,8 @@ fn payload_length(&self) -> Option<usize> {
/// Response from a GMC API command.
pub(crate) struct GmcResponse {
/// Response status (`NV_STATUS` code). Zero means success.
- #[expect(dead_code)]
pub(crate) status: u32,
/// Response payload copied out of the message queue.
- #[expect(dead_code)]
pub(crate) payload: KVec<u8>,
}
@@ -708,7 +706,6 @@ pub(crate) fn send_gmc_no_wait(
/// # Errors
///
/// Returns the same errors as [`Self::send_gmc_and_receive_timeout`].
- #[expect(dead_code)]
pub(crate) fn send_gmc_and_receive(
&self,
command_id: u32,
diff --git a/drivers/gpu/nova-core/gsp/fw.rs b/drivers/gpu/nova-core/gsp/fw.rs
index a416826737bb..aef283dda087 100644
--- a/drivers/gpu/nova-core/gsp/fw.rs
+++ b/drivers/gpu/nova-core/gsp/fw.rs
@@ -2,7 +2,7 @@
// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
pub(crate) mod commands;
-mod r000_00;
+pub(crate) mod r000_00;
// Alias to avoid repeating the version number with every use.
use r000_00 as bindings;
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 1402e37541b7..d910445dd2b0 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -20,6 +20,8 @@
gsp::commands::FifoEngineList, //
};
+mod commands;
+mod fw;
mod hal;
mod vram;
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
new file mode 100644
index 000000000000..679c22b0381a
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -0,0 +1,91 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! vGPU firmware command operations.
+//!
+//! Sends requests, checks responses and coordinates the firmware command
+//! sequences used by instance lifecycle operations.
+
+use kernel::{
+ device,
+ prelude::*, //
+};
+
+use crate::gsp::{
+ cmdq::Cmdq,
+ nvkv::{
+ nvkv_words,
+ Decoder,
+ UnknownKeyPolicy, //
+ }, //
+};
+
+use super::{
+ fw::{
+ commands::{
+ VgpuPropertiesSchema, //
+ },
+ GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE,
+ GMCAPI_CMD_QUERY_VGPU_PROPERTIES, //
+ }, //
+};
+
+pub(super) use super::fw::commands::{
+ Dbdf,
+ VgpuProperties, //
+};
+
+/// Reports a matching command's raw firmware status before translating rejection to `EIO`.
+fn check_status(dev: &device::Device<device::Bound>, command_id: u32, status: u32) -> Result {
+ if status == 0 {
+ Ok(())
+ } else {
+ dev_err!(
+ dev,
+ "GMC command {:#x} rejected: NV_STATUS={:#x}\n",
+ command_id,
+ status,
+ );
+ Err(EIO)
+ }
+}
+
+/// Query the vGPU type assigned to a VF by its DBDF.
+#[expect(dead_code)]
+pub(super) fn query_assigned_vf_type(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ dbdf: Dbdf,
+) -> Result<u32> {
+ let command_id = GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE;
+ let payload = u64::from(dbdf.into_raw()).to_le_bytes();
+ // Preserve the firmware receive budget; only the leading type ID is consumed.
+ let response = cmdq.send_gmc_and_receive(command_id, &payload, 64)?;
+ check_status(dev, command_id, response.status)?;
+
+ let bytes = response.payload.first_chunk::<4>().ok_or(ENODEV)?;
+ Ok(u32::from_le_bytes(*bytes))
+}
+
+/// Query and decode the firmware properties of one vGPU type.
+#[expect(dead_code)]
+pub(super) fn query_vgpu_properties(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ type_id: u32,
+) -> Result<KBox<VgpuProperties>> {
+ let command_id = GMCAPI_CMD_QUERY_VGPU_PROPERTIES;
+ // NVKV replies vary in length. This is a receive capacity, not their encoded size.
+ let response = cmdq.send_gmc_and_receive(command_id, &type_id.to_le_bytes(), 4096)?;
+ check_status(dev, command_id, response.status)?;
+
+ let words = nvkv_words(&response.payload, &[])?;
+ let decoder = Decoder::new(&words, UnknownKeyPolicy::Ignore);
+ let mut schema = VgpuPropertiesSchema::default();
+ let properties = KBox::try_init(decoder.decode(&mut schema)?, GFP_KERNEL)?;
+ if properties.type_id != type_id || properties.max_instance == 0 {
+ Err(EINVAL)
+ } else {
+ Ok(properties)
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
new file mode 100644
index 000000000000..23097cb13121
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -0,0 +1,17 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! vGPU firmware interface.
+//!
+//! Exposes command constants and communication buffer layout types from
+//! raw firmware bindings.
+
+pub(super) mod commands;
+
+use crate::gsp::bindings;
+
+pub(super) const GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE;
+
+pub(super) const GMCAPI_CMD_QUERY_VGPU_PROPERTIES: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_QUERY_VGPU_PROPERTIES;
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
new file mode 100644
index 000000000000..a4d23eae8233
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -0,0 +1,95 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! Wire types and codecs for vGPU commands.
+//!
+//! Defines request encoders and response schemas independently of command
+//! submission and instance lifecycle operations.
+
+use kernel::{
+ alloc::ArrayVec,
+ bitfield, //
+};
+
+use crate::{
+ gsp::nvkv::{
+ nvkv_decode,
+ Array,
+ Key,
+ KeyId,
+ Required, //
+ }, //
+};
+
+bitfield! {
+ pub(crate) struct Dbdf(u32) {
+ 2:0 function;
+ 7:3 device;
+ 15:8 bus;
+ 31:16 domain;
+ }
+}
+
+nvkv_decode! {
+ pub(crate) struct VgpuPropertiesSchema => VgpuProperties {
+ // TODO: `name`/`class` required?
+ name: Array<u8, { VgpuProperties::STRING_LEN }, { Self::TYPE_NAME_KEY }>,
+ class: Array<u8, { VgpuProperties::STRING_LEN }, { Self::CLASS_KEY }>,
+ type_id: Required<u32, { Self::TYPE_ID_KEY }>,
+ bar1_length: Required<u64, { Self::BAR1_LENGTH_KEY }>,
+ max_instance: Required<u32, { Self::MAX_INSTANCE_KEY }>,
+ ecc: Key<u32, { Self::ECC_KEY }>,
+ profile_size: Required<u64, { Self::PROFILE_SIZE_KEY }>,
+ max_fps: Key<u32, { Self::MAX_FPS_KEY }>,
+ num_heads: Key<u32, { Self::NUM_HEADS_KEY }>,
+ max_res_x: Key<u32, { Self::MAX_RES_X_KEY }>,
+ max_res_y: Key<u32, { Self::MAX_RES_Y_KEY }>,
+ dev_id: Required<u32, { Self::DEV_ID_KEY }>,
+ subsystem_id: Required<u32, { Self::SUBSYSTEM_ID_KEY }>,
+ fb_length: Required<u64, { Self::FB_LENGTH_KEY }>,
+ gsp_heap_size: Required<u64, { Self::GSP_HEAP_SIZE_KEY }>,
+ fb_reservation: Required<u64, { Self::FB_RESERVATION_KEY }>,
+ }
+}
+
+impl VgpuPropertiesSchema {
+ const TYPE_NAME_KEY: KeyId = 0x3100;
+ const CLASS_KEY: KeyId = 0x3101;
+ const TYPE_ID_KEY: KeyId = 0x3102;
+ const BAR1_LENGTH_KEY: KeyId = 0x3103;
+ const MAX_INSTANCE_KEY: KeyId = 0x3104;
+ const ECC_KEY: KeyId = 0x3105;
+ const PROFILE_SIZE_KEY: KeyId = 0x3106;
+ const MAX_FPS_KEY: KeyId = 0x3107;
+ const NUM_HEADS_KEY: KeyId = 0x3108;
+ const MAX_RES_X_KEY: KeyId = 0x3109;
+ const MAX_RES_Y_KEY: KeyId = 0x310A;
+ const DEV_ID_KEY: KeyId = 0x310B;
+ const SUBSYSTEM_ID_KEY: KeyId = 0x310C;
+ const FB_LENGTH_KEY: KeyId = 0x310D;
+ const GSP_HEAP_SIZE_KEY: KeyId = 0x310E;
+ const FB_RESERVATION_KEY: KeyId = 0x310F;
+}
+
+pub(crate) struct VgpuProperties {
+ name: ArrayVec<u8, { Self::STRING_LEN }>,
+ class: ArrayVec<u8, { Self::STRING_LEN }>,
+ pub(crate) type_id: u32,
+ pub(crate) bar1_length: u64,
+ pub(crate) max_instance: u32,
+ ecc: u32,
+ profile_size: u64,
+ max_fps: u32,
+ num_heads: u32,
+ max_res_x: u32,
+ max_res_y: u32,
+ pub(crate) dev_id: u32,
+ pub(crate) subsystem_id: u32,
+ pub(crate) fb_length: u64,
+ pub(crate) gsp_heap_size: u64,
+ fb_reservation: u64,
+}
+
+impl VgpuProperties {
+ const STRING_LEN: usize = 64;
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 18/31] gpu: nova-core: vgpu: add instance create/destroy
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (16 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 17/31] gpu: nova-core: vgpu: query VF assignments and properties Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 19/31] gpu: nova-core: vgpu: encode vGPU boot requests Zhi Wang
` (12 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
A vGPU instance owns the host resources reserved for a VF. Its assigned
type determines the channel range and VRAM and management-heap
reservation.
Add a manager-owned instance registry, reject duplicate GFID or DBDF
entries and enforce the profile's instance limit. Reserve registry
capacity before acquiring resources, then move the channel and VRAM
reservations into the instance. Removing an instance or dropping the
registry releases these host reservations through their owners.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 12 +-
drivers/gpu/nova-core/vgpu.rs | 24 ++-
drivers/gpu/nova-core/vgpu/instance.rs | 218 +++++++++++++++++++++++++
drivers/gpu/nova-core/vgpu/vram.rs | 1 -
4 files changed, 240 insertions(+), 15 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/instance.rs
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index 97b33d4b2631..b41a9921381d 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -54,14 +54,16 @@
},
};
-#[cfg_attr(not(CONFIG_KUNIT = "y"), expect(dead_code))]
mod channel;
mod hal;
mod regs;
use self::channel::TOTAL_CHANNELS;
-pub(crate) use self::channel::ChannelIdPool;
+pub(crate) use self::channel::{
+ ChannelIdPool,
+ ChannelIdReservation, //
+};
macro_rules! define_chipset {
({ $($variant:ident = $value:expr),* $(,)* }) =>
@@ -325,7 +327,7 @@ fn static_info(&self) -> &gsp::commands::GspStaticInfo {
#[pin_data]
pub(crate) struct Gpu<'gpu> {
spec: Spec,
- vgpu: Option<VgpuManager<'gpu>>,
+ vgpu: Option<Pin<KBox<VgpuManager<'gpu>>>>,
/// GSP event interrupt registration.
///
/// Must be kept declared *before* `gsp_resources`, so that the handler is unregistered, and
@@ -465,7 +467,7 @@ pub(crate) fn new<'a>(
let info = &gsp_resources.boot_result.static_info;
match gsp_resources.vgpu_state {
VgpuState::Disabled => None,
- VgpuState::Enabled { .. } => Some(VgpuManager::new(
+ VgpuState::Enabled { .. } => Some(KBox::pin_init(VgpuManager::new(
// SAFETY: `chid_pool` is initialized above at its final pinned address.
// The private manager and its pool borrow cannot escape this `Gpu`.
// Completed field drop order drops the manager before the pool; on failure,
@@ -474,7 +476,7 @@ pub(crate) fn new<'a>(
&info.fifo_engine_list(),
info.vmmu_segment_size,
TOTAL_CHANNELS,
- )),
+ ), GFP_KERNEL)?),
}
},
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index d910445dd2b0..fb4f1b5f7754 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -4,8 +4,10 @@
use kernel::{
device,
+ new_mutex,
pci,
- prelude::*, //
+ prelude::*,
+ sync::Mutex, //
};
use crate::{
@@ -23,6 +25,7 @@
mod commands;
mod fw;
mod hal;
+mod instance;
mod vram;
/// vGPU state detected during GPU construction.
@@ -88,16 +91,17 @@ fn query_state(
}
}
+use self::instance::VgpuInstances;
+
/// Runtime resources for an enabled vGPU boot.
+#[pin_data]
pub(crate) struct VgpuManager<'gpu> {
- #[expect(dead_code)]
+ #[pin]
+ instances: Mutex<VgpuInstances<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
/// VMMU segment size in bytes, or zero if GSP-RM omitted it.
- #[expect(dead_code)]
vmmu_segment_size: u64,
- #[expect(dead_code)]
total_channels: u32,
- #[expect(dead_code)]
fifo_engine_list: FifoEngineList,
}
@@ -108,12 +112,14 @@ pub(crate) fn new(
fifo_engine_list: &FifoEngineList,
vmmu_segment_size: u64,
total_channels: u32,
- ) -> Self {
- Self {
+ ) -> impl PinInit<Self> + use<'gpu> {
+ let fifo_engine_list = *fifo_engine_list;
+ pin_init!(Self {
+ instances <- new_mutex!(VgpuInstances::new(), "nova-core::vgpu-instances"),
chid_pool,
vmmu_segment_size,
total_channels,
- fifo_engine_list: *fifo_engine_list,
- }
+ fifo_engine_list,
+ })
}
}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
new file mode 100644
index 000000000000..a90aaef28b35
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -0,0 +1,218 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+use core::num::{
+ NonZero,
+ NonZeroUsize, //
+};
+
+use kernel::{
+ prelude::*,
+ ptr::Alignment,
+ sizes::SizeConstants, //
+};
+
+use crate::{
+ gpu::ChannelIdReservation,
+ mm::GpuMm, //
+};
+
+use super::{
+ commands::Dbdf,
+ vram::{
+ VgpuVramLayout,
+ VgpuVramSlot,
+ VgpuVramSlotAllocator, //
+ },
+ VgpuManager, //
+};
+
+/// Guest Function ID validated against one device's total number of VFs.
+///
+/// GFID 0 is reserved for the PF; VFs start at 1. Keeping the nonzero `u16`
+/// preserves the PCI SR-IOV range when forming a plugin doorbell handle.
+#[repr(transparent)]
+#[derive(Clone, Copy, PartialEq, Eq)]
+pub(super) struct Gfid(NonZero<u16>);
+
+impl Gfid {
+ /// Validates an external GFID for a device supporting `total_vfs` VFs.
+ #[expect(dead_code)]
+ pub(super) fn new(gfid: u32, total_vfs: NonZero<u16>) -> Result<Self> {
+ let gfid = u16::try_from(gfid).map_err(|_| EINVAL)?;
+ let gfid = NonZero::new(gfid).ok_or(EINVAL)?;
+
+ if gfid > total_vfs {
+ Err(EINVAL)
+ } else {
+ Ok(Self(gfid))
+ }
+ }
+
+ #[expect(dead_code)]
+ pub(super) const fn get(self) -> u16 {
+ self.0.get()
+ }
+}
+
+/// Resource requirements and device identity for one vGPU type.
+#[expect(dead_code)]
+pub(super) struct VgpuType {
+ vgpu_type_id: u32,
+ bar1_length: u64,
+ max_instance: u32,
+ pci_dev_id: u32,
+ pci_subsys_id: u32,
+ fb_length: u64,
+ gsp_heap_size: u64,
+}
+
+/// A vGPU instance and the resources reserved for it.
+#[expect(dead_code)]
+struct VgpuInstance<'gpu> {
+ gfid: Gfid,
+ dbdf: Dbdf,
+ vgpu_type: VgpuType,
+ vm_pid: u32,
+ chids: ChannelIdReservation<'gpu>,
+ vram_slot: VgpuVramSlot,
+}
+
+/// Identity and firmware profile used to allocate an instance.
+pub(super) struct InstanceInfo {
+ gfid: Gfid,
+ dbdf: Dbdf,
+ vgpu_type: VgpuType,
+ vm_pid: u32,
+}
+
+#[expect(dead_code)]
+impl InstanceInfo {
+ pub(super) const fn new(gfid: Gfid, dbdf: Dbdf, vgpu_type: VgpuType, vm_pid: u32) -> Self {
+ Self {
+ gfid,
+ dbdf,
+ vgpu_type,
+ vm_pid,
+ }
+ }
+}
+
+/// Registry of live vGPU instances.
+pub(super) struct VgpuInstances<'gpu> {
+ instances: KVec<VgpuInstance<'gpu>>,
+ vram_slots: Option<VgpuVramSlotAllocator>,
+}
+
+#[expect(dead_code)]
+impl<'gpu> VgpuInstances<'gpu> {
+ pub(super) const fn new() -> Self {
+ Self {
+ instances: KVec::new(),
+ vram_slots: None,
+ }
+ }
+
+ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<VgpuVramSlot> {
+ let replace_empty_pool = match self.vram_slots.as_ref() {
+ Some(allocator) if allocator.is_empty() => !allocator.matches_layout(layout)?,
+ _ => false,
+ };
+ if replace_empty_pool {
+ self.vram_slots = None;
+ }
+
+ if let Some(allocator) = self.vram_slots.as_mut() {
+ return allocator.alloc(layout);
+ }
+
+ let mut allocator = VgpuVramSlotAllocator::new(mm, layout)?;
+ let slot = allocator.alloc(layout)?;
+ self.vram_slots = Some(allocator);
+ Ok(slot)
+ }
+
+ /// Allocate resources and register a new inactive vGPU instance.
+ fn allocate_instance(
+ &mut self,
+ mm: &GpuMm<'_>,
+ vgpu: &VgpuManager<'gpu>,
+ info: InstanceInfo,
+ ) -> Result<Gfid> {
+ let InstanceInfo {
+ gfid,
+ dbdf,
+ vgpu_type,
+ vm_pid,
+ } = info;
+
+ let instance_exists = self
+ .instances
+ .iter()
+ .any(|instance| instance.gfid == gfid || instance.dbdf == dbdf);
+ if instance_exists {
+ return Err(EEXIST);
+ }
+
+ let type_id = vgpu_type.vgpu_type_id;
+ let num_type_instances = self
+ .instances
+ .iter()
+ .filter(|instance| instance.vgpu_type.vgpu_type_id == type_id)
+ .count();
+ let max_instances = usize::try_from(vgpu_type.max_instance).map_err(|_| EOVERFLOW)?;
+ if max_instances == 0 || num_type_instances >= max_instances {
+ return Err(ENOSPC);
+ }
+
+ // Reserve registry capacity before acquiring resources so publishing
+ // the completed instance cannot fail due to memory pressure.
+ self.instances.reserve(1, GFP_KERNEL)?;
+
+ let channels_per_instance = vgpu
+ .total_channels
+ .checked_div(vgpu_type.max_instance)
+ .ok_or(EINVAL)?;
+ let channels_per_instance =
+ usize::try_from(channels_per_instance).map_err(|_| EOVERFLOW)?;
+ let channels_per_instance = NonZeroUsize::new(channels_per_instance).ok_or(EINVAL)?;
+ let chids = vgpu
+ .chid_pool
+ .reserve_ids(channels_per_instance, Alignment::SZ_1)?;
+
+ let vram_layout = VgpuVramLayout {
+ type_id,
+ max_slots: vgpu_type.max_instance,
+ fb_size: vgpu_type.fb_length,
+ heap_size: vgpu_type.gsp_heap_size,
+ fb_align: vgpu.vmmu_segment_size,
+ };
+ let vram_slot = self.alloc_vram_slot(mm, vram_layout)?;
+
+ let instance = VgpuInstance {
+ gfid,
+ dbdf,
+ vgpu_type,
+ vm_pid,
+ chids,
+ vram_slot,
+ };
+ self.instances
+ .push_within_capacity(instance)
+ .map_err(|_| EIO)?;
+
+ Ok(gfid)
+ }
+
+ /// Remove an instance and release its channel and VRAM reservations.
+ fn destroy_instance(&mut self, gfid: Gfid) -> Result {
+ let instance_index = self
+ .instances
+ .iter()
+ .position(|instance| instance.gfid == gfid)
+ .ok_or(ENOENT)?;
+ let instance = self.instances.remove(instance_index).map_err(|_| EIO)?;
+ drop(instance);
+ Ok(())
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/vram.rs b/drivers/gpu/nova-core/vgpu/vram.rs
index 65e6a945ea6f..faaeea106009 100644
--- a/drivers/gpu/nova-core/vgpu/vram.rs
+++ b/drivers/gpu/nova-core/vgpu/vram.rs
@@ -116,7 +116,6 @@ pub(super) struct VgpuVramSlotAllocator {
slot_bitmap: Arc<SlotBitmap>,
}
-#[expect(dead_code)]
impl VgpuVramSlotAllocator {
pub(super) fn new(mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<Self> {
let layout = layout.validated()?;
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 19/31] gpu: nova-core: vgpu: encode vGPU boot requests
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (17 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 18/31] gpu: nova-core: vgpu: add instance create/destroy Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 20/31] gpu: nova-core: vgpu: add GSP plugin communication buffers Zhi Wang
` (11 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The boot request describes a vGPU's identity, channel map, VRAM,
plugin heap and log buffers to GSP-RM.
Add the NVKV schema and encode it from a named boot description. Pass
VRAM region views so each physical address stays paired with its length
until encoding, and retain the control offset relative to the plugin
heap. Represent channel-map engine and index fields in their 16-bit wire
domains.
Describe one guest VRAM segment and the current whole-GPU,
non-SMIG resource layout. Keep separate MIG-RM heap fields zero for that
supported firmware path.
Co-developed-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gsp/nvkv.rs | 11 ++
drivers/gpu/nova-core/gsp/nvkv/encode.rs | 12 ++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 169 +++++++++++++++++++++-
3 files changed, 189 insertions(+), 3 deletions(-)
diff --git a/drivers/gpu/nova-core/gsp/nvkv.rs b/drivers/gpu/nova-core/gsp/nvkv.rs
index 05b1ba5a4f39..02a12bd831d6 100644
--- a/drivers/gpu/nova-core/gsp/nvkv.rs
+++ b/drivers/gpu/nova-core/gsp/nvkv.rs
@@ -172,6 +172,17 @@ pub(crate) struct Array<T: Default + Copy, const N: usize, const KEY_ID: KeyId>
vec: ArrayVec<T, N>,
}
+impl<T: Default + Copy, const N: usize, const KEY_ID: KeyId> Array<T, N, KEY_ID> {
+ /// Creates an array filled from `values`.
+ ///
+ /// Fails with `EINVAL` if `values` is longer than `N`.
+ pub(crate) fn new(values: &[T]) -> Result<Self> {
+ let mut vec = ArrayVec::default();
+ vec.extend_from_slice(values)?;
+ Ok(Self { vec })
+ }
+}
+
bitfield! {
/// The op word that starts each NVKV operation.
struct Op(u64) {
diff --git a/drivers/gpu/nova-core/gsp/nvkv/encode.rs b/drivers/gpu/nova-core/gsp/nvkv/encode.rs
index 0047be65e8a9..1f4df24d8c9c 100644
--- a/drivers/gpu/nova-core/gsp/nvkv/encode.rs
+++ b/drivers/gpu/nova-core/gsp/nvkv/encode.rs
@@ -6,6 +6,7 @@
use kernel::prelude::*;
use super::{
+ Array,
EncodedStream,
Index,
Key,
@@ -145,6 +146,17 @@ fn encode(&self, encoder: &mut Encoder) -> Result {
}
}
+impl<T, const N: usize, const KEY_ID: KeyId> Encodable for Array<T, N, KEY_ID>
+where
+ T: Default + Copy,
+ for<'a> IndexedKey<&'a [T], KEY_ID>: Encodable,
+{
+ #[inline(always)]
+ fn encode(&self, encoder: &mut Encoder) -> Result {
+ IndexedKey::<&[T], KEY_ID>::new(Index::new::<0>(), self.vec.as_slice()).encode(encoder)
+ }
+}
+
impl<T: Encodable> Encodable for Option<T> {
#[inline(always)]
fn encode(&self, encoder: &mut Encoder) -> Result {
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index a4d23eae8233..83048473d35f 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -8,17 +8,24 @@
use kernel::{
alloc::ArrayVec,
- bitfield, //
+ bitfield,
+ prelude::*, //
};
use crate::{
gsp::nvkv::{
nvkv_decode,
+ nvkv_encode,
Array,
+ Encodable,
+ EncodedStream,
+ Encoder,
+ Index, //
Key,
KeyId,
- Required, //
- }, //
+ Required,
+ },
+ mm::vram::VramRegion, //
};
bitfield! {
@@ -30,6 +37,162 @@ pub(crate) struct Dbdf(u32) {
}
}
+#[derive(Clone, Copy)]
+struct SwizzId(u32);
+
+impl SwizzId {
+ const WHOLE_GPU: Self = Self(0xFFFF_FFFF);
+}
+
+impl From<SwizzId> for u32 {
+ fn from(value: SwizzId) -> Self {
+ value.0
+ }
+}
+
+bitfield! {
+ pub(crate) struct ChannelMapEntry(u64) {
+ 15:0 engine_type;
+ 31:16 index;
+ 63:32 chid_offset;
+ }
+}
+
+impl ChannelMapEntry {
+ const KEY: KeyId = 0x1001;
+
+ #[expect(dead_code)]
+ pub(crate) fn new(engine_type: u16, index: u16, chid_offset: u32) -> Self {
+ Self::zeroed()
+ .with_engine_type(engine_type)
+ .with_index(index)
+ .with_chid_offset(chid_offset)
+ }
+}
+
+impl Encodable for KVVec<ChannelMapEntry> {
+ fn encode(&self, encoder: &mut Encoder) -> Result {
+ // SAFETY: `ChannelMapEntry` is a `bitfield!` over `u64`, i.e.
+ // `#[repr(transparent)]` around a `u64`, so the entries are
+ // layout-compatible with `u64` and can be viewed as a `u64` slice.
+ let slice = unsafe { core::slice::from_raw_parts(self.as_ptr().cast::<u64>(), self.len()) };
+ encoder.encode_array64(ChannelMapEntry::KEY, Index::new::<0>(), slice)
+ }
+}
+
+bitfield! {
+ struct VgpuBootloadOptions(u64) {
+ }
+}
+
+nvkv_encode! {
+ struct VgpuBootloadRequest {
+ dbdf: Key<Dbdf, { Self::DBDF_KEY }, u32>,
+ gfid: Key<u32, { Self::GFID_KEY }>,
+ vgpu_type: Key<u32, { Self::VGPU_TYPE_KEY }>,
+ vm_pid: Key<u32, { Self::VM_PID_KEY }>,
+ swizz_id: Key<SwizzId, { Self::SWIZZ_ID_KEY }, u32>,
+ num_channels: Key<u32, { Self::NUM_CHANNELS_KEY }>,
+ num_plugin_channels: Key<u32, { Self::NUM_PLUGIN_CHANNELS_KEY }>,
+ guest_fb_segment_count: Key<u32, { Self::GUEST_FB_SEGMENT_COUNT_KEY }>,
+ options: Key<VgpuBootloadOptions, { Self::OPTIONS_KEY }, u64>,
+ channel_mapping: KVVec<ChannelMapEntry>,
+ guest_fb_segment_phys_addr: Array<u64, 8, { Self::GUEST_FB_SEGMENT_PHYS_ADDR_KEY }>,
+ guest_fb_segment_length: Array<u64, 8, { Self::GUEST_FB_SEGMENT_LENGTH_KEY }>,
+ plugin_heap_phys_addr: Key<u64, { Self::PLUGIN_HEAP_PHYS_ADDR_KEY }>,
+ plugin_heap_length: Key<u64, { Self::PLUGIN_HEAP_LENGTH_KEY }>,
+ ctrl_buff_offset: Key<u64, { Self::CTRL_BUFF_OFFSET_KEY }>,
+ init_task_log_offset: Key<u64, { Self::INIT_TASK_LOG_OFFSET_KEY }>,
+ init_task_log_size: Key<u64, { Self::INIT_TASK_LOG_SIZE_KEY }>,
+ vgpu_task_log_offset: Key<u64, { Self::VGPU_TASK_LOG_OFFSET_KEY }>,
+ vgpu_task_log_size: Key<u64, { Self::VGPU_TASK_LOG_SIZE_KEY }>,
+ kernel_log_offset: Key<u64, { Self::KERNEL_LOG_OFFSET_KEY }>,
+ kernel_log_size: Key<u64, { Self::KERNEL_LOG_SIZE_KEY }>,
+ mig_rm_heap_phys_addr: Key<u64, { Self::MIG_RM_HEAP_PHYS_ADDR_KEY }>,
+ mig_rm_heap_length: Key<u64, { Self::MIG_RM_HEAP_LENGTH_KEY }>,
+ }
+}
+
+impl VgpuBootloadRequest {
+ const DBDF_KEY: KeyId = 0x0001;
+ const GFID_KEY: KeyId = 0x0002;
+ const VGPU_TYPE_KEY: KeyId = 0x0003;
+ const VM_PID_KEY: KeyId = 0x0004;
+ const SWIZZ_ID_KEY: KeyId = 0x0005;
+ const NUM_CHANNELS_KEY: KeyId = 0x0006;
+ const NUM_PLUGIN_CHANNELS_KEY: KeyId = 0x0007;
+ const GUEST_FB_SEGMENT_COUNT_KEY: KeyId = 0x0008;
+ const OPTIONS_KEY: KeyId = 0x1000;
+ const GUEST_FB_SEGMENT_PHYS_ADDR_KEY: KeyId = 0x1002;
+ const GUEST_FB_SEGMENT_LENGTH_KEY: KeyId = 0x1003;
+ const PLUGIN_HEAP_PHYS_ADDR_KEY: KeyId = 0x1004;
+ const PLUGIN_HEAP_LENGTH_KEY: KeyId = 0x1005;
+ const CTRL_BUFF_OFFSET_KEY: KeyId = 0x1006;
+ const INIT_TASK_LOG_OFFSET_KEY: KeyId = 0x1007;
+ const INIT_TASK_LOG_SIZE_KEY: KeyId = 0x1008;
+ const VGPU_TASK_LOG_OFFSET_KEY: KeyId = 0x1009;
+ const VGPU_TASK_LOG_SIZE_KEY: KeyId = 0x100A;
+ const KERNEL_LOG_OFFSET_KEY: KeyId = 0x100B;
+ const KERNEL_LOG_SIZE_KEY: KeyId = 0x100C;
+ const MIG_RM_HEAP_PHYS_ADDR_KEY: KeyId = 0x100D;
+ const MIG_RM_HEAP_LENGTH_KEY: KeyId = 0x100E;
+}
+
+/// Identity, channel mapping and VRAM regions passed to a GSP plugin at boot.
+///
+/// Region addresses are physical VRAM addresses, including the log fields whose wire
+/// names contain `offset`. The control-buffer offset is relative to the plugin heap.
+pub(crate) struct BootloadInfo<'a> {
+ pub(crate) dbdf: Dbdf,
+ pub(crate) gfid: u32,
+ pub(crate) vgpu_type: u32,
+ pub(crate) vm_pid: u32,
+ pub(crate) num_channels: u32,
+ pub(crate) num_plugin_channels: u32,
+ pub(crate) channel_mapping: KVVec<ChannelMapEntry>,
+ pub(crate) guest_fb: &'a VramRegion,
+ pub(crate) plugin_heap: &'a VramRegion,
+ pub(crate) ctrl_buffer_offset: u64,
+ pub(crate) init_log: &'a VramRegion,
+ pub(crate) vgpu_log: &'a VramRegion,
+ pub(crate) kernel_log: &'a VramRegion,
+}
+
+/// Encodes a `VGPU_BOOTLOAD` request using the typed NVKV schema.
+#[expect(dead_code)]
+pub(crate) fn encode_vgpu_bootload(info: BootloadInfo<'_>) -> Result<EncodedStream> {
+ let request = VgpuBootloadRequest {
+ dbdf: info.dbdf.into(),
+ gfid: info.gfid.into(),
+ vgpu_type: info.vgpu_type.into(),
+ vm_pid: info.vm_pid.into(),
+ swizz_id: SwizzId::WHOLE_GPU.into(),
+ num_channels: info.num_channels.into(),
+ num_plugin_channels: info.num_plugin_channels.into(),
+ guest_fb_segment_count: 1.into(),
+ options: VgpuBootloadOptions::zeroed().into(),
+ channel_mapping: info.channel_mapping,
+ guest_fb_segment_phys_addr: Array::new(&[info.guest_fb.address()])?,
+ guest_fb_segment_length: Array::new(&[info.guest_fb.size()])?,
+ plugin_heap_phys_addr: info.plugin_heap.address().into(),
+ plugin_heap_length: info.plugin_heap.size().into(),
+ ctrl_buff_offset: info.ctrl_buffer_offset.into(),
+ init_task_log_offset: info.init_log.address().into(),
+ init_task_log_size: info.init_log.size().into(),
+ vgpu_task_log_offset: info.vgpu_log.address().into(),
+ vgpu_task_log_size: info.vgpu_log.size().into(),
+ kernel_log_offset: info.kernel_log.address().into(),
+ kernel_log_size: info.kernel_log.size().into(),
+ // The current non-SMIG firmware path does not map a separate MIG-RM heap.
+ mig_rm_heap_phys_addr: 0.into(),
+ mig_rm_heap_length: 0.into(),
+ };
+
+ let mut encoder = Encoder::new();
+ request.encode(&mut encoder)?;
+ Ok(encoder.finish())
+}
+
nvkv_decode! {
pub(crate) struct VgpuPropertiesSchema => VgpuProperties {
// TODO: `name`/`class` required?
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 20/31] gpu: nova-core: vgpu: add GSP plugin communication buffers
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (18 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 19/31] gpu: nova-core: vgpu: encode vGPU boot requests Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 21/31] gpu: nova-core: vgpu: add instance boot Zhi Wang
` (10 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The GSP plugin keeps its boot-ready marker, control data and logs in a
communication region within the instance's management heap.
Map that region through BAR1 and expose its firmware-defined subregions.
Check fixed layout totals and the control structure size at build time
while retaining runtime checks against the allocated heap and mapping
bounds.
Own the BarMapping so dropping the communication region unmaps it, and
retain its BarUser and MM borrows for that lifetime. Expose explicit
unmap to report teardown failures without consuming the owner. Provide
the ready-marker access needed to boot a reused instance slot.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/mm.rs | 1 -
drivers/gpu/nova-core/vgpu.rs | 1 +
drivers/gpu/nova-core/vgpu/fw.rs | 15 ++
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 205 ++++++++++++++++++
4 files changed, 221 insertions(+), 1 deletion(-)
create mode 100644 drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
diff --git a/drivers/gpu/nova-core/mm.rs b/drivers/gpu/nova-core/mm.rs
index a1e0bbffc460..72e96a1dde5a 100644
--- a/drivers/gpu/nova-core/mm.rs
+++ b/drivers/gpu/nova-core/mm.rs
@@ -67,7 +67,6 @@ macro_rules! impl_pfn_bounded {
mod regs;
pub(super) mod tlb;
pub(super) mod vmm;
-#[expect(dead_code)]
pub(crate) mod vram;
/// GPU Memory Manager - owns all core MM components.
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index fb4f1b5f7754..024cb13694f4 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -24,6 +24,7 @@
mod commands;
mod fw;
+mod gsp_plugin_comm;
mod hal;
mod instance;
mod vram;
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index 23097cb13121..aabe114b55c2 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -10,6 +10,21 @@
use crate::gsp::bindings;
+pub(super) use bindings::{
+ GSP_PLUGIN_BOOTLOADED,
+ VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE,
+ VGPU_CPU_GSP_CTRL_BUFF_REGION as RawControlRegion,
+ VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_ERROR_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_GUEST_RPC_TRACE_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_INIT_TASK_LOG_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_KERNEL_TASK_LOG_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_MESSAGE_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_MIGRATION_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE, //
+};
+
pub(super) const GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE;
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
new file mode 100644
index 000000000000..b945e216a1d6
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -0,0 +1,205 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! GSP plugin communication buffer mappings and access.
+
+use kernel::{
+ num::casts::u32_as_usize,
+ prelude::*,
+ sync::Mutex, //
+};
+
+use crate::mm::{
+ bar_user::{
+ BarMapping,
+ BarUser, //
+ },
+ vram::VramRegion,
+ GpuMm, //
+};
+
+use super::fw::{
+ self,
+ RawControlRegion, //
+};
+
+static_assert!(
+ fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_MESSAGE_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_MIGRATION_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_ERROR_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_INIT_TASK_LOG_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_KERNEL_TASK_LOG_BUFF_REGION_SIZE
+ + fw::VGPU_CPU_GSP_GUEST_RPC_TRACE_BUFF_REGION_SIZE
+ == fw::VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE
+);
+static_assert!(
+ size_of::<RawControlRegion>() == u32_as_usize(fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE)
+);
+/// Physical VRAM regions containing the vGPU plugin logs.
+#[expect(dead_code)]
+pub(super) struct PluginLogRegions {
+ pub(super) init: VramRegion,
+ pub(super) vgpu: VramRegion,
+ pub(super) kernel: VramRegion,
+}
+
+fn take_region(region: &VramRegion, cursor: &mut u64, size: u32) -> Result<VramRegion> {
+ let end = cursor.checked_add(u64::from(size)).ok_or(EOVERFLOW)?;
+ let subregion = region.subregion(*cursor..end)?;
+ *cursor = end;
+ Ok(subregion)
+}
+
+/// BAR1 mapping of the plugin communication region in its management heap.
+///
+/// r000 layout, with byte offsets from the management heap (not to scale):
+///
+/// ```text
+/// 0x000000 +----------------------------------------+
+/// | Control (boot-ready marker) 4 KiB |
+/// 0x001000 +----------------------------------------+
+/// | Response 4 KiB |
+/// 0x002000 +----------------------------------------+
+/// | Message 4 KiB |
+/// 0x003000 +----------------------------------------+
+/// | Migration 2 MiB |
+/// 0x203000 +----------------------------------------+
+/// | Error 4 KiB |
+/// 0x204000 +----------------------------------------+
+/// | Init task log 128 KiB |
+/// 0x224000 +----------------------------------------+
+/// | vGPU task log 256 KiB |
+/// 0x264000 +----------------------------------------+
+/// | Kernel task log 64 KiB |
+/// 0x274000 +----------------------------------------+
+/// | Guest RPC trace 64 KiB |
+/// 0x284000 +----------------------------------------+
+/// ```
+pub(super) struct CommBufferRegion<'map, 'gpu> {
+ map: BarMapping<'map, 'gpu>,
+ control: VramRegion,
+ init_log: VramRegion,
+ vgpu_log: VramRegion,
+ kernel_log: VramRegion,
+}
+
+#[expect(dead_code)]
+impl<'map, 'gpu> CommBufferRegion<'map, 'gpu> {
+ /// Map the communication portion of a plugin management heap.
+ pub(super) fn new(
+ bar_user: &'map BarUser<'gpu>,
+ mm: &'map Mutex<GpuMm<'gpu>>,
+ management_heap: &VramRegion,
+ ) -> Result<Self> {
+ let total_size = u64::from(fw::VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE);
+ let region = management_heap.subregion(0..total_size)?;
+ let mut cursor = 0;
+
+ let control = take_region(®ion, &mut cursor, fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE)?;
+ take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE,
+ )?;
+ take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_MESSAGE_BUFF_REGION_SIZE,
+ )?;
+ take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_MIGRATION_BUFF_REGION_SIZE,
+ )?;
+ take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_ERROR_BUFF_REGION_SIZE,
+ )?;
+ let init_log = take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_INIT_TASK_LOG_BUFF_REGION_SIZE,
+ )?;
+ let vgpu_log = take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE,
+ )?;
+ let kernel_log = take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_KERNEL_TASK_LOG_BUFF_REGION_SIZE,
+ )?;
+ take_region(
+ ®ion,
+ &mut cursor,
+ fw::VGPU_CPU_GSP_GUEST_RPC_TRACE_BUFF_REGION_SIZE,
+ )?;
+
+ let map = BarMapping::new(bar_user, mm, region, true)?;
+
+ Ok(Self {
+ map,
+ control,
+ init_log,
+ vgpu_log,
+ kernel_log,
+ })
+ }
+
+ fn region_offset(&self, region: &VramRegion) -> Result<usize> {
+ let offset = region
+ .address()
+ .checked_sub(self.map.region().address())
+ .ok_or(EINVAL)?;
+ if offset.checked_add(region.size()).ok_or(EOVERFLOW)? > self.map.region().size() {
+ return Err(EINVAL);
+ }
+
+ usize::try_from(offset).map_err(|_| EOVERFLOW)
+ }
+
+ fn io_offset(&self, region: &VramRegion, field: usize, width: usize) -> Result<usize> {
+ let field_end = field.checked_add(width).ok_or(EOVERFLOW)?;
+ if u64::try_from(field_end).map_err(|_| EOVERFLOW)? > region.size() {
+ return Err(EINVAL);
+ }
+
+ self.region_offset(region)?
+ .checked_add(field)
+ .ok_or(EOVERFLOW)
+ }
+
+ fn read_u32(&self, region: &VramRegion, field: usize) -> Result<u32> {
+ self.map
+ .try_read32(self.io_offset(region, field, size_of::<u32>())?)
+ }
+
+ /// Return the physical regions occupied by the three plugin logs.
+ pub(super) fn plugin_logs(&self) -> PluginLogRegions {
+ PluginLogRegions {
+ init: self.init_log.clone(),
+ vgpu: self.vgpu_log.clone(),
+ kernel: self.kernel_log.clone(),
+ }
+ }
+
+ /// Return whether firmware has published the plugin boot marker.
+ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
+ let value = self.read_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_seq_num),
+ )?;
+
+ Ok(value == fw::GSP_PLUGIN_BOOTLOADED)
+ }
+
+ /// Invalidate the PTEs and release the communication mapping.
+ pub(super) fn unmap(&mut self) -> Result {
+ self.map.unmap()
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 21/31] gpu: nova-core: vgpu: add instance boot
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (19 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 20/31] gpu: nova-core: vgpu: add GSP plugin communication buffers Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 22/31] gpu: nova-core: vgpu: add instance shutdown Zhi Wang
` (9 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Start an allocated instance's GSP plugin with its channel map, VRAM,
plugin heap and log regions. Clear the old ready marker before BOOTLOAD
and wait for firmware to publish a new one.
Retain the command queue and memory dependencies needed by the instance,
and map its communication buffers before boot.
Co-developed-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 71 +++++---
drivers/gpu/nova-core/gsp/fw/commands.rs | 1 -
drivers/gpu/nova-core/vgpu.rs | 30 +++-
drivers/gpu/nova-core/vgpu/commands.rs | 21 ++-
drivers/gpu/nova-core/vgpu/fw.rs | 3 +
drivers/gpu/nova-core/vgpu/fw/commands.rs | 2 -
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 16 +-
drivers/gpu/nova-core/vgpu/instance.rs | 152 ++++++++++++++++--
drivers/gpu/nova-core/vgpu/vram.rs | 1 -
9 files changed, 245 insertions(+), 52 deletions(-)
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index b41a9921381d..012c7e6abca6 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -8,6 +8,7 @@
fmt,
gpu::buddy::GpuBuddyParams,
io::Io,
+ new_mutex,
num::Bounded,
pci,
prelude::*,
@@ -16,6 +17,7 @@
SizeConstants,
SZ_4K, //
},
+ sync::Mutex,
};
use crate::{
@@ -327,6 +329,7 @@ fn static_info(&self) -> &gsp::commands::GspStaticInfo {
#[pin_data]
pub(crate) struct Gpu<'gpu> {
spec: Spec,
+ /// Drops before the MM, BAR1 mappings and GSP needed for instance teardown.
vgpu: Option<Pin<KBox<VgpuManager<'gpu>>>>,
/// GSP event interrupt registration.
///
@@ -339,7 +342,8 @@ pub(crate) struct Gpu<'gpu> {
///
/// Must be kept declared *before* `gsp_resources`, so that its components are dropped while
/// the GSP is still operational.
- mm: GpuMm<'gpu>,
+ #[pin]
+ mm: Mutex<GpuMm<'gpu>>,
/// BAR1 user interface for CPU access to GPU virtual memory.
#[pin]
bar_user: BarUser<'gpu>,
@@ -463,22 +467,7 @@ pub(crate) fn new<'a>(
})?,
}),
- vgpu: {
- let info = &gsp_resources.boot_result.static_info;
- match gsp_resources.vgpu_state {
- VgpuState::Disabled => None,
- VgpuState::Enabled { .. } => Some(KBox::pin_init(VgpuManager::new(
- // SAFETY: `chid_pool` is initialized above at its final pinned address.
- // The private manager and its pool borrow cannot escape this `Gpu`.
- // Completed field drop order drops the manager before the pool; on failure,
- // pin-init drops it before the earlier-initialized pool.
- unsafe { &*core::ptr::from_ref(chid_pool.as_ref().get_ref()) },
- &info.fifo_engine_list(),
- info.vmmu_segment_size,
- TOTAL_CHANNELS,
- ), GFP_KERNEL)?),
- }
- },
+
// GSP boot left the SWGEN0 latch set and pending bits in the tree.
_: {
@@ -529,7 +518,7 @@ pub(crate) fn new<'a>(
},
// Create GPU memory manager owning memory management resources.
- mm: {
+ mm <- {
let info = gsp_resources.static_info();
let usable_vram = info.usable_fb_regions().next().ok_or(ENODEV)?;
let buddy_params = GpuBuddyParams {
@@ -538,12 +527,15 @@ pub(crate) fn new<'a>(
chunk_size: Alignment::new::<SZ_4K>(),
};
- GpuMm::new(
- bar,
- gsp_resources.spec.chipset,
- buddy_params,
- VramAddress::from_raw(info.total_fb_end().ok_or(ENODEV)?),
- )?
+ new_mutex!(
+ GpuMm::new(
+ bar,
+ gsp_resources.spec.chipset,
+ buddy_params,
+ VramAddress::from_raw(info.total_fb_end().ok_or(ENODEV)?),
+ )?,
+ "nova-core::gpu-mm",
+ )
},
// Create BAR1 user interface for CPU access to GPU virtual memory.
@@ -558,6 +550,34 @@ pub(crate) fn new<'a>(
bar1,
)?
},
+
+ vgpu: {
+ match gsp_resources.vgpu_state {
+ VgpuState::Disabled => None,
+ VgpuState::Enabled { .. } => {
+ // SAFETY: These sibling fields are initialized at their final pinned
+ // addresses. The private manager cannot escape this `Gpu`, and is dropped
+ // before all its dependencies, both here on failure and on normal removal.
+ let (cmdq, bar_user, mm, chid_pool) = unsafe {
+ (
+ &*core::ptr::from_ref(&gsp_resources.gsp.cmdq),
+ &*core::ptr::from_ref(bar_user.as_ref().get_ref()),
+ &*core::ptr::from_ref(mm.as_ref().get_ref()),
+ &*core::ptr::from_ref(chid_pool.as_ref().get_ref()),
+ )
+ };
+ Some(KBox::pin_init(VgpuManager::new(
+ dev,
+ cmdq,
+ bar_user,
+ mm,
+ chid_pool,
+ gsp_resources.static_info(),
+ TOTAL_CHANNELS,
+ ), GFP_KERNEL)?)
+ }
+ }
+ },
})
}
@@ -567,10 +587,11 @@ pub(crate) fn run_selftests(self: Pin<&mut Self>, pdev: &pci::Device<device::Bou
let this = self.project();
let dev = pdev.as_ref();
let info = this.gsp_resources.static_info();
+ let mut mm = this.mm.lock();
if let Err(err) = crate::mm::selftest::run(
dev,
- this.mm,
+ &mut mm,
info.usable_fb_regions(),
this.bar_user.as_ref().get_ref(),
info.bar1_pde_base(),
diff --git a/drivers/gpu/nova-core/gsp/fw/commands.rs b/drivers/gpu/nova-core/gsp/fw/commands.rs
index 7381b30f3680..b1087d950b30 100644
--- a/drivers/gpu/nova-core/gsp/fw/commands.rs
+++ b/drivers/gpu/nova-core/gsp/fw/commands.rs
@@ -335,7 +335,6 @@ pub(crate) struct FifoEngineList {
}
impl FifoEngineList {
- #[expect(dead_code)]
pub(crate) fn gmc_ids(&self) -> &[u32] {
// PANIC: The type invariant bounds `count` by the array capacity.
&self.gmc_ids[..self.count]
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 024cb13694f4..84c4d0b3839d 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -19,7 +19,17 @@
ChannelIdPool,
Chipset, //
},
- gsp::commands::FifoEngineList, //
+ gsp::{
+ cmdq::Cmdq,
+ commands::{
+ FifoEngineList,
+ GspStaticInfo, //
+ }, //
+ },
+ mm::{
+ bar_user::BarUser,
+ GpuMm, //
+ }, //
};
mod commands;
@@ -99,6 +109,10 @@ fn query_state(
pub(crate) struct VgpuManager<'gpu> {
#[pin]
instances: Mutex<VgpuInstances<'gpu>>,
+ dev: &'gpu device::Device<device::Bound>,
+ cmdq: &'gpu Cmdq<'gpu>,
+ bar_user: &'gpu BarUser<'gpu>,
+ mm: &'gpu Mutex<GpuMm<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
/// VMMU segment size in bytes, or zero if GSP-RM omitted it.
vmmu_segment_size: u64,
@@ -109,14 +123,22 @@ pub(crate) struct VgpuManager<'gpu> {
impl<'gpu> VgpuManager<'gpu> {
/// Retains runtime parameters from a completed vGPU-enabled GSP boot.
pub(crate) fn new(
+ dev: &'gpu device::Device<device::Bound>,
+ cmdq: &'gpu Cmdq<'gpu>,
+ bar_user: &'gpu BarUser<'gpu>,
+ mm: &'gpu Mutex<GpuMm<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
- fifo_engine_list: &FifoEngineList,
- vmmu_segment_size: u64,
+ info: &GspStaticInfo,
total_channels: u32,
) -> impl PinInit<Self> + use<'gpu> {
- let fifo_engine_list = *fifo_engine_list;
+ let fifo_engine_list = info.fifo_engine_list();
+ let vmmu_segment_size = info.vmmu_segment_size;
pin_init!(Self {
instances <- new_mutex!(VgpuInstances::new(), "nova-core::vgpu-instances"),
+ dev,
+ cmdq,
+ bar_user,
+ mm,
chid_pool,
vmmu_segment_size,
total_channels,
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 679c22b0381a..2e9c250b65ab 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -8,7 +8,9 @@
use kernel::{
device,
- prelude::*, //
+ prelude::*,
+ time::Delta,
+ transmute::AsBytes, //
};
use crate::gsp::{
@@ -25,6 +27,7 @@
commands::{
VgpuPropertiesSchema, //
},
+ GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK,
GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE,
GMCAPI_CMD_QUERY_VGPU_PROPERTIES, //
}, //
@@ -89,3 +92,19 @@ pub(super) fn query_vgpu_properties(
Ok(properties)
}
}
+
+/// Send BOOTLOAD and check its firmware status.
+pub(super) fn send_bootload(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ payload: &[u64],
+) -> Result {
+ let command_id = GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK;
+ let response = cmdq.send_gmc_and_receive_timeout(
+ command_id,
+ AsBytes::as_bytes(payload),
+ 0,
+ Delta::from_secs(10),
+ )?;
+ check_status(dev, command_id, response.status)
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index aabe114b55c2..82c29dfb467a 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -30,3 +30,6 @@
pub(super) const GMCAPI_CMD_QUERY_VGPU_PROPERTIES: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_QUERY_VGPU_PROPERTIES;
+
+pub(super) const GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK;
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index 83048473d35f..bf8d44819e13 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -61,7 +61,6 @@ pub(crate) struct ChannelMapEntry(u64) {
impl ChannelMapEntry {
const KEY: KeyId = 0x1001;
- #[expect(dead_code)]
pub(crate) fn new(engine_type: u16, index: u16, chid_offset: u32) -> Self {
Self::zeroed()
.with_engine_type(engine_type)
@@ -159,7 +158,6 @@ pub(crate) struct BootloadInfo<'a> {
}
/// Encodes a `VGPU_BOOTLOAD` request using the typed NVKV schema.
-#[expect(dead_code)]
pub(crate) fn encode_vgpu_bootload(info: BootloadInfo<'_>) -> Result<EncodedStream> {
let request = VgpuBootloadRequest {
dbdf: info.dbdf.into(),
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index b945e216a1d6..0b4d03c5f0fc 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -39,7 +39,6 @@
size_of::<RawControlRegion>() == u32_as_usize(fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE)
);
/// Physical VRAM regions containing the vGPU plugin logs.
-#[expect(dead_code)]
pub(super) struct PluginLogRegions {
pub(super) init: VramRegion,
pub(super) vgpu: VramRegion,
@@ -86,7 +85,6 @@ pub(super) struct CommBufferRegion<'map, 'gpu> {
kernel_log: VramRegion,
}
-#[expect(dead_code)]
impl<'map, 'gpu> CommBufferRegion<'map, 'gpu> {
/// Map the communication portion of a plugin management heap.
pub(super) fn new(
@@ -188,6 +186,19 @@ pub(super) fn plugin_logs(&self) -> PluginLogRegions {
}
}
+ /// Clear a previous boot marker before starting the plugin.
+ pub(super) fn clear_plugin_ready(&self) -> Result {
+ let offset = self.io_offset(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_seq_num),
+ size_of::<u32>(),
+ )?;
+ self.map.try_write32(0, offset)?;
+ // Complete the posted clear before firmware can publish its new marker.
+ self.map.try_read32(offset)?;
+ Ok(())
+ }
+
/// Return whether firmware has published the plugin boot marker.
pub(super) fn is_plugin_ready(&self) -> Result<bool> {
let value = self.read_u32(
@@ -199,6 +210,7 @@ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
}
/// Invalidate the PTEs and release the communication mapping.
+ #[expect(dead_code)]
pub(super) fn unmap(&mut self) -> Result {
self.map.unmap()
}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index a90aaef28b35..b2a0b25c0e15 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -7,18 +7,38 @@
};
use kernel::{
+ device,
prelude::*,
ptr::Alignment,
- sizes::SizeConstants, //
+ sizes::SizeConstants,
+ time::{
+ delay::fsleep,
+ Delta,
+ Instant,
+ Monotonic, //
+ }, //
};
use crate::{
gpu::ChannelIdReservation,
- mm::GpuMm, //
+ gsp::{
+ cmdq::Cmdq,
+ commands::FifoEngineList, //
+ },
+ mm::GpuMm,
};
use super::{
- commands::Dbdf,
+ commands::{
+ send_bootload,
+ Dbdf, //
+ },
+ fw::commands::{
+ encode_vgpu_bootload,
+ BootloadInfo,
+ ChannelMapEntry, //
+ },
+ gsp_plugin_comm::CommBufferRegion,
vram::{
VgpuVramLayout,
VgpuVramSlot,
@@ -27,6 +47,52 @@
VgpuManager, //
};
+/// Ready limit used by `vmiopd_negotiate_cpu_gsp_version()` for the same marker.
+const PLUGIN_READY_TIMEOUT: Delta = Delta::from_secs(10);
+
+/// Per-engine channel budget reserved by the full SR-IOV plugin.
+///
+/// The supported GB20x path keeps firmware's non-heavy default, which uses
+/// `PLUGIN_ALLOCATED_CHANNELS_PER_ENGINE` rather than the heavy-mode budget.
+const PLUGIN_CHANNELS_PER_ENGINE: u32 = 3;
+
+/// Build the typed channel mapping from the GSP FIFO engine list.
+fn channel_mapping(
+ fifo_engine_list: &FifoEngineList,
+ chid_offset: u32,
+) -> Result<KVVec<ChannelMapEntry>> {
+ let mut mapping = KVVec::new();
+ for &gmc_id in fifo_engine_list.gmc_ids() {
+ // CAST: The mask leaves only the low 16 bits of the GMC engine ID.
+ let engine_type = (gmc_id & u32::from(u16::MAX)) as u16;
+ // CAST: Shifting a `u32` by 16 leaves at most 16 bits.
+ let index = (gmc_id >> u16::BITS) as u16;
+ mapping.push(
+ ChannelMapEntry::new(engine_type, index, chid_offset),
+ GFP_KERNEL,
+ )?;
+ }
+ Ok(mapping)
+}
+
+fn wait_plugin_ready(
+ dev: &device::Device<device::Bound>,
+ comm: &CommBufferRegion<'_, '_>,
+) -> Result {
+ let start = Instant::<Monotonic>::now();
+
+ loop {
+ if comm.is_plugin_ready()? {
+ dev_dbg!(dev, "vGPU plugin ready after {:?}\n", start.elapsed());
+ return Ok(());
+ }
+ if start.elapsed() >= PLUGIN_READY_TIMEOUT {
+ return Err(ETIMEDOUT);
+ }
+ fsleep(Delta::from_millis(1));
+ }
+}
+
/// Guest Function ID validated against one device's total number of VFs.
///
/// GFID 0 is reserved for the PF; VFs start at 1. Keeping the nonzero `u16`
@@ -49,7 +115,6 @@ pub(super) fn new(gfid: u32, total_vfs: NonZero<u16>) -> Result<Self> {
}
}
- #[expect(dead_code)]
pub(super) const fn get(self) -> u16 {
self.0.get()
}
@@ -68,14 +133,72 @@ pub(super) struct VgpuType {
}
/// A vGPU instance and the resources reserved for it.
-#[expect(dead_code)]
struct VgpuInstance<'gpu> {
gfid: Gfid,
dbdf: Dbdf,
vgpu_type: VgpuType,
vm_pid: u32,
- chids: ChannelIdReservation<'gpu>,
+ num_plugin_channels: u32,
+ comm: CommBufferRegion<'gpu, 'gpu>,
+ // Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
+ chids: ChannelIdReservation<'gpu>,
+}
+
+impl<'gpu> VgpuInstance<'gpu> {
+ #[expect(dead_code)]
+ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
+ let dev = vgpu.dev;
+ self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
+
+ Ok(())
+ }
+
+ /// Bootload the GSP vGPU plugin and wait for its BAR1 ready indication.
+ fn bootload(
+ &mut self,
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ fifo_engine_list: &FifoEngineList,
+ ) -> Result {
+ let fb = &self.vram_slot.fbmem;
+ let mgmt = &self.vram_slot.mgmt_heap;
+ let logs = self.comm.plugin_logs();
+
+ let payload = encode_vgpu_bootload(BootloadInfo {
+ dbdf: self.dbdf,
+ gfid: u32::from(self.gfid.get()),
+ vgpu_type: self.vgpu_type.vgpu_type_id,
+ vm_pid: self.vm_pid,
+ num_channels: u32::try_from(self.chids.len()).map_err(|_| EOVERFLOW)?,
+ num_plugin_channels: self.num_plugin_channels,
+ channel_mapping: channel_mapping(
+ fifo_engine_list,
+ u32::try_from(self.chids.start).map_err(|_| EOVERFLOW)?,
+ )?,
+ guest_fb: fb,
+ plugin_heap: mgmt,
+ ctrl_buffer_offset: 0,
+ init_log: &logs.init,
+ vgpu_log: &logs.vgpu,
+ kernel_log: &logs.kernel,
+ })?;
+
+ dev_dbg!(
+ dev,
+ "bootload: gfid={} sending {} typed NVKV bytes\n",
+ self.gfid.get(),
+ payload.len() * size_of::<u64>(),
+ );
+
+ self.comm.clear_plugin_ready()?;
+ send_bootload(dev, cmdq, &payload)?;
+
+ wait_plugin_ready(dev, &self.comm)?;
+
+ dev_dbg!(dev, "bootload: gfid={} plugin ready\n", self.gfid.get());
+ Ok(())
+ }
}
/// Identity and firmware profile used to allocate an instance.
@@ -133,12 +256,7 @@ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<
}
/// Allocate resources and register a new inactive vGPU instance.
- fn allocate_instance(
- &mut self,
- mm: &GpuMm<'_>,
- vgpu: &VgpuManager<'gpu>,
- info: InstanceInfo,
- ) -> Result<Gfid> {
+ fn allocate_instance(&mut self, vgpu: &VgpuManager<'gpu>, info: InstanceInfo) -> Result<Gfid> {
let InstanceInfo {
gfid,
dbdf,
@@ -165,8 +283,7 @@ fn allocate_instance(
return Err(ENOSPC);
}
- // Reserve registry capacity before acquiring resources so publishing
- // the completed instance cannot fail due to memory pressure.
+ // Reserve capacity before acquiring resources so registration cannot allocate.
self.instances.reserve(1, GFP_KERNEL)?;
let channels_per_instance = vgpu
@@ -187,15 +304,18 @@ fn allocate_instance(
heap_size: vgpu_type.gsp_heap_size,
fb_align: vgpu.vmmu_segment_size,
};
- let vram_slot = self.alloc_vram_slot(mm, vram_layout)?;
+ let vram_slot = self.alloc_vram_slot(&vgpu.mm.lock(), vram_layout)?;
+ let comm = CommBufferRegion::new(vgpu.bar_user, vgpu.mm, &vram_slot.mgmt_heap)?;
let instance = VgpuInstance {
gfid,
dbdf,
vgpu_type,
vm_pid,
- chids,
+ num_plugin_channels: PLUGIN_CHANNELS_PER_ENGINE,
+ comm,
vram_slot,
+ chids,
};
self.instances
.push_within_capacity(instance)
diff --git a/drivers/gpu/nova-core/vgpu/vram.rs b/drivers/gpu/nova-core/vgpu/vram.rs
index faaeea106009..bd7d1b4bf039 100644
--- a/drivers/gpu/nova-core/vgpu/vram.rs
+++ b/drivers/gpu/nova-core/vgpu/vram.rs
@@ -102,7 +102,6 @@ fn drop(&mut self) {
/// All device accesses and mappings of its regions must end before dropping the slot.
/// Clearing the entry locks a sleeping mutex, so dropping the slot may sleep.
#[must_use]
-#[expect(dead_code)]
pub(super) struct VgpuVramSlot {
pub(super) fbmem: VramRegion,
pub(super) mgmt_heap: VramRegion,
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 22/31] gpu: nova-core: vgpu: add instance shutdown
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (20 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 21/31] gpu: nova-core: vgpu: add instance boot Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 23/31] gpu: nova-core: vgpu: initialize GSP plugin RPC buffers Zhi Wang
` (8 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Stop the GSP plugin, wait for SHUTDOWN completion, and release its
firmware resources with CLEANUP before unmapping the communication
buffers.
Add PendingInstance rollback and manager teardown. Register host resources
before firmware work, and retain them until device removal if firmware
state is uncertain or teardown fails. Keep the manager's dependencies
alive until cleanup finishes.
Co-developed-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Alok Kumar <alkumar@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gsp/cmdq.rs | 1 -
drivers/gpu/nova-core/vgpu.rs | 9 +-
drivers/gpu/nova-core/vgpu/commands.rs | 55 +++++++-
drivers/gpu/nova-core/vgpu/fw.rs | 9 ++
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 1 -
drivers/gpu/nova-core/vgpu/instance.rs | 118 +++++++++++++++++-
6 files changed, 182 insertions(+), 11 deletions(-)
diff --git a/drivers/gpu/nova-core/gsp/cmdq.rs b/drivers/gpu/nova-core/gsp/cmdq.rs
index abd9edadff64..8730fd0079c4 100644
--- a/drivers/gpu/nova-core/gsp/cmdq.rs
+++ b/drivers/gpu/nova-core/gsp/cmdq.rs
@@ -812,7 +812,6 @@ pub(crate) fn send_gmc_and_receive_timeout(
///
/// Errors from initializing the request headers and from either callback are propagated
/// as-is. The current valid element is consumed before a callback error is returned.
- #[expect(dead_code)]
pub(crate) fn send_gmc_and_wait_event(
&self,
command_id: u32,
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 84c4d0b3839d..f7184e7009a5 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -105,7 +105,7 @@ fn query_state(
use self::instance::VgpuInstances;
/// Runtime resources for an enabled vGPU boot.
-#[pin_data]
+#[pin_data(PinnedDrop)]
pub(crate) struct VgpuManager<'gpu> {
#[pin]
instances: Mutex<VgpuInstances<'gpu>>,
@@ -146,3 +146,10 @@ pub(crate) fn new(
})
}
}
+
+#[pinned_drop]
+impl PinnedDrop for VgpuManager<'_> {
+ fn drop(self: Pin<&mut Self>) {
+ self.instances.lock().release_all(&self);
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 2e9c250b65ab..52c6fc0ee335 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -28,9 +28,13 @@
VgpuPropertiesSchema, //
},
GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK,
+ GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES,
GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE,
- GMCAPI_CMD_QUERY_VGPU_PROPERTIES, //
- }, //
+ GMCAPI_CMD_QUERY_VGPU_PROPERTIES,
+ GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK,
+ GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE, //
+ },
+ instance::Gfid, //
};
pub(super) use super::fw::commands::{
@@ -108,3 +112,50 @@ pub(super) fn send_bootload(
)?;
check_status(dev, command_id, response.status)
}
+
+/// Shut down a vGPU plugin task and wait for its completion event.
+pub(super) fn send_shutdown(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+) -> Result {
+ let payload = u32::from(gfid.get()).to_le_bytes();
+ cmdq.send_gmc_and_wait_event(
+ GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK,
+ &payload,
+ Delta::from_secs(10),
+ |command_id, _max_response_size, _sequence, payload_0, payload_1| {
+ Ok(
+ command_id == GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE
+ && payload
+ .iter()
+ .copied()
+ .eq(Iterator::chain(payload_0.iter(), payload_1)
+ .take(payload.len())
+ .copied()),
+ )
+ },
+ |command_id, _max_response_size, _sequence, _payload_0, _payload_1| {
+ dev_dbg!(
+ dev,
+ "shutdown: ignoring unrelated event command={:#x}\n",
+ command_id,
+ );
+ Ok(())
+ },
+ )
+}
+
+/// Release firmware resources after a plugin task has stopped.
+pub(super) fn send_cleanup(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+) -> Result {
+ let command_id = GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES;
+ let response =
+ cmdq.send_gmc_and_receive(command_id, &u32::from(gfid.get()).to_le_bytes(), 0)?;
+ check_status(dev, command_id, response.status)?;
+ dev_dbg!(dev, "cleanup: gfid={} done\n", gfid.get());
+ Ok(())
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index 82c29dfb467a..1528cc56ce74 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -33,3 +33,12 @@
pub(super) const GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK;
+
+pub(super) const GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK;
+
+pub(super) const GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE;
+
+pub(super) const GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES;
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index 0b4d03c5f0fc..a00f4d2affc7 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -210,7 +210,6 @@ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
}
/// Invalidate the PTEs and release the communication mapping.
- #[expect(dead_code)]
pub(super) fn unmap(&mut self) -> Result {
self.map.unmap()
}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index b2a0b25c0e15..bf02595590e1 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -31,6 +31,8 @@
use super::{
commands::{
send_bootload,
+ send_cleanup,
+ send_shutdown,
Dbdf, //
},
fw::commands::{
@@ -143,10 +145,12 @@ struct VgpuInstance<'gpu> {
// Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
chids: ChannelIdReservation<'gpu>,
+ needs_teardown: bool,
+ /// An uncertain or failed operation retains resources until device removal.
+ failure: Option<Error>,
}
impl<'gpu> VgpuInstance<'gpu> {
- #[expect(dead_code)]
fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
let dev = vgpu.dev;
self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
@@ -154,6 +158,26 @@ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
Ok(())
}
+ /// Tear down once, retaining all remaining resources if an operation fails.
+ fn teardown(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
+ if let Some(error) = self.failure {
+ return Err(error);
+ }
+ let result = (|| {
+ self.shutdown(vgpu.dev, vgpu.cmdq)?;
+
+ if self.needs_teardown {
+ send_cleanup(vgpu.dev, vgpu.cmdq, self.gfid)?;
+ self.needs_teardown = false;
+ }
+ self.comm.unmap()
+ })();
+ if let Err(error) = result {
+ self.failure = Some(error);
+ }
+ result
+ }
+
/// Bootload the GSP vGPU plugin and wait for its BAR1 ready indication.
fn bootload(
&mut self,
@@ -192,6 +216,7 @@ fn bootload(
);
self.comm.clear_plugin_ready()?;
+ self.needs_teardown = true;
send_bootload(dev, cmdq, &payload)?;
wait_plugin_ready(dev, &self.comm)?;
@@ -199,6 +224,15 @@ fn bootload(
dev_dbg!(dev, "bootload: gfid={} plugin ready\n", self.gfid.get());
Ok(())
}
+
+ /// Stop the plugin when firmware may own instance resources.
+ fn shutdown(&mut self, dev: &device::Device<device::Bound>, cmdq: &Cmdq<'_>) -> Result {
+ if self.needs_teardown {
+ send_shutdown(dev, cmdq, self.gfid)?;
+ dev_dbg!(dev, "shutdown: gfid={} stopped\n", self.gfid.get());
+ }
+ Ok(())
+ }
}
/// Identity and firmware profile used to allocate an instance.
@@ -255,8 +289,12 @@ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<
Ok(slot)
}
- /// Allocate resources and register a new inactive vGPU instance.
- fn allocate_instance(&mut self, vgpu: &VgpuManager<'gpu>, info: InstanceInfo) -> Result<Gfid> {
+ /// Register host resources before submitting any firmware work.
+ fn allocate_instance<'a>(
+ &'a mut self,
+ vgpu: &'a VgpuManager<'gpu>,
+ info: InstanceInfo,
+ ) -> Result<PendingInstance<'a, 'gpu>> {
let InstanceInfo {
gfid,
dbdf,
@@ -316,23 +354,91 @@ fn allocate_instance(&mut self, vgpu: &VgpuManager<'gpu>, info: InstanceInfo) ->
comm,
vram_slot,
chids,
+ needs_teardown: false,
+ failure: None,
};
self.instances
.push_within_capacity(instance)
.map_err(|_| EIO)?;
- Ok(gfid)
+ Ok(PendingInstance {
+ instances: self,
+ vgpu,
+ gfid,
+ committed: false,
+ })
}
- /// Remove an instance and release its channel and VRAM reservations.
- fn destroy_instance(&mut self, gfid: Gfid) -> Result {
+ /// Remove an instance only after firmware and mapping teardown have succeeded.
+ fn destroy_instance(&mut self, vgpu: &VgpuManager<'gpu>, gfid: Gfid) -> Result {
let instance_index = self
.instances
.iter()
.position(|instance| instance.gfid == gfid)
.ok_or(ENOENT)?;
+ let instance = self.instances.get_mut(instance_index).ok_or(EIO)?;
+ instance.teardown(vgpu)?;
+
let instance = self.instances.remove(instance_index).map_err(|_| EIO)?;
drop(instance);
Ok(())
}
+
+ /// Release the registry while the manager's MM and firmware dependencies remain available.
+ pub(super) fn release_all(&mut self, vgpu: &VgpuManager<'gpu>) {
+ for instance in &mut self.instances {
+ // Failed operations retain their resources until this final device removal.
+ // Do not resubmit commands whose firmware outcome was uncertain.
+ if instance.failure.is_none() {
+ if let Err(error) = instance.teardown(vgpu) {
+ dev_err!(
+ vgpu.dev,
+ "vGPU teardown failed for gfid={}: {:?}\n",
+ instance.gfid.get(),
+ error,
+ );
+ }
+ }
+ }
+ self.instances.clear();
+ }
+}
+
+/// Rolls back this creation attempt unless ownership has passed to the caller.
+struct PendingInstance<'a, 'gpu> {
+ instances: &'a mut VgpuInstances<'gpu>,
+ vgpu: &'a VgpuManager<'gpu>,
+ gfid: Gfid,
+ committed: bool,
+}
+
+#[expect(dead_code)]
+impl PendingInstance<'_, '_> {
+ fn activate(mut self) -> Result {
+ let instance = self
+ .instances
+ .instances
+ .iter_mut()
+ .find(|instance| instance.gfid == self.gfid)
+ .ok_or(EIO)?;
+ instance.activate(self.vgpu)?;
+ self.committed = true;
+ Ok(())
+ }
+}
+
+impl Drop for PendingInstance<'_, '_> {
+ fn drop(&mut self) {
+ if self.committed {
+ return;
+ }
+ if let Err(error) = self.instances.destroy_instance(self.vgpu, self.gfid) {
+ dev_err!(
+ self.vgpu.dev,
+ "vGPU creation could not release gfid={}: {:?}; resources retained until removal\n",
+ self.gfid.get(),
+ error,
+ );
+ }
+ }
}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 23/31] gpu: nova-core: vgpu: initialize GSP plugin RPC buffers
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (21 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 22/31] gpu: nova-core: vgpu: add instance shutdown Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 24/31] gpu: nova-core: vgpu: add GSP plugin RPC transactions Zhi Wang
` (7 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The first plugin RPC needs region offsets, a protocol version and
initialized control and response state in the shared communication
buffer.
Add BAR1 accessors to initialize those fields and verify the fixed
response structure size at build time. The control sequence field first
carries the boot-ready marker, so RPC initialization must follow the
plugin's ready indication.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/mm/bar_user.rs | 10 ++
drivers/gpu/nova-core/vgpu/fw.rs | 2 +
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 141 +++++++++++++++++-
3 files changed, 147 insertions(+), 6 deletions(-)
diff --git a/drivers/gpu/nova-core/mm/bar_user.rs b/drivers/gpu/nova-core/mm/bar_user.rs
index 964aee8bfdb1..bcbef1571fb9 100644
--- a/drivers/gpu/nova-core/mm/bar_user.rs
+++ b/drivers/gpu/nova-core/mm/bar_user.rs
@@ -150,6 +150,11 @@ pub(crate) fn try_read32(&self, offset: usize) -> Result<u32> {
self.bar_user.bar1.try_read32(off)
}
+ fn try_write8(&self, value: u8, offset: usize) -> Result {
+ let off = self.bar_offset(offset)?;
+ self.bar_user.bar1.try_write8(value, off)
+ }
+
/// Write a 32-bit value at the given offset.
pub(crate) fn try_write32(&self, value: u32, offset: usize) -> Result {
let off = self.bar_offset(offset)?;
@@ -283,6 +288,11 @@ pub(crate) fn try_read32(&self, offset: usize) -> Result<u32> {
.try_read32(self.access_offset(offset, size_of::<u32>())?)
}
+ pub(crate) fn try_write8(&self, value: u8, offset: usize) -> Result {
+ self.access()?
+ .try_write8(value, self.access_offset(offset, size_of::<u8>())?)
+ }
+
pub(crate) fn try_write32(&self, value: u32, offset: usize) -> Result {
self.access()?
.try_write32(value, self.access_offset(offset, size_of::<u32>())?)
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index 1528cc56ce74..03225b9833c3 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -15,12 +15,14 @@
VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE,
VGPU_CPU_GSP_CTRL_BUFF_REGION as RawControlRegion,
VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_CTRL_BUFF_VERSION,
VGPU_CPU_GSP_ERROR_BUFF_REGION_SIZE,
VGPU_CPU_GSP_GUEST_RPC_TRACE_BUFF_REGION_SIZE,
VGPU_CPU_GSP_INIT_TASK_LOG_BUFF_REGION_SIZE,
VGPU_CPU_GSP_KERNEL_TASK_LOG_BUFF_REGION_SIZE,
VGPU_CPU_GSP_MESSAGE_BUFF_REGION_SIZE,
VGPU_CPU_GSP_MIGRATION_BUFF_REGION_SIZE,
+ VGPU_CPU_GSP_RESPONSE_BUFF_REGION as RawResponseRegion,
VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE,
VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE, //
};
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index a00f4d2affc7..31da31de4fd2 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -20,7 +20,8 @@
use super::fw::{
self,
- RawControlRegion, //
+ RawControlRegion,
+ RawResponseRegion, //
};
static_assert!(
@@ -38,6 +39,10 @@
static_assert!(
size_of::<RawControlRegion>() == u32_as_usize(fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE)
);
+static_assert!(
+ size_of::<RawResponseRegion>() == u32_as_usize(fw::VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE)
+);
+
/// Physical VRAM regions containing the vGPU plugin logs.
pub(super) struct PluginLogRegions {
pub(super) init: VramRegion,
@@ -80,9 +85,14 @@ fn take_region(region: &VramRegion, cursor: &mut u64, size: u32) -> Result<VramR
pub(super) struct CommBufferRegion<'map, 'gpu> {
map: BarMapping<'map, 'gpu>,
control: VramRegion,
+ response: VramRegion,
+ message: VramRegion,
+ migration: VramRegion,
+ error: VramRegion,
init_log: VramRegion,
vgpu_log: VramRegion,
kernel_log: VramRegion,
+ guest_trace: VramRegion,
}
impl<'map, 'gpu> CommBufferRegion<'map, 'gpu> {
@@ -97,22 +107,22 @@ pub(super) fn new(
let mut cursor = 0;
let control = take_region(®ion, &mut cursor, fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE)?;
- take_region(
+ let response = take_region(
®ion,
&mut cursor,
fw::VGPU_CPU_GSP_RESPONSE_BUFF_REGION_SIZE,
)?;
- take_region(
+ let message = take_region(
®ion,
&mut cursor,
fw::VGPU_CPU_GSP_MESSAGE_BUFF_REGION_SIZE,
)?;
- take_region(
+ let migration = take_region(
®ion,
&mut cursor,
fw::VGPU_CPU_GSP_MIGRATION_BUFF_REGION_SIZE,
)?;
- take_region(
+ let error = take_region(
®ion,
&mut cursor,
fw::VGPU_CPU_GSP_ERROR_BUFF_REGION_SIZE,
@@ -132,7 +142,7 @@ pub(super) fn new(
&mut cursor,
fw::VGPU_CPU_GSP_KERNEL_TASK_LOG_BUFF_REGION_SIZE,
)?;
- take_region(
+ let guest_trace = take_region(
®ion,
&mut cursor,
fw::VGPU_CPU_GSP_GUEST_RPC_TRACE_BUFF_REGION_SIZE,
@@ -143,9 +153,14 @@ pub(super) fn new(
Ok(Self {
map,
control,
+ response,
+ message,
+ migration,
+ error,
init_log,
vgpu_log,
kernel_log,
+ guest_trace,
})
}
@@ -177,6 +192,21 @@ fn read_u32(&self, region: &VramRegion, field: usize) -> Result<u32> {
.try_read32(self.io_offset(region, field, size_of::<u32>())?)
}
+ fn write_u8(&self, region: &VramRegion, field: usize, value: u8) -> Result {
+ self.map
+ .try_write8(value, self.io_offset(region, field, size_of::<u8>())?)
+ }
+
+ fn write_u32(&self, region: &VramRegion, field: usize, value: u32) -> Result {
+ self.map
+ .try_write32(value, self.io_offset(region, field, size_of::<u32>())?)
+ }
+
+ fn write_u64(&self, region: &VramRegion, field: usize, value: u64) -> Result {
+ self.map
+ .try_write64(value, self.io_offset(region, field, size_of::<u64>())?)
+ }
+
/// Return the physical regions occupied by the three plugin logs.
pub(super) fn plugin_logs(&self) -> PluginLogRegions {
PluginLogRegions {
@@ -209,6 +239,105 @@ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
Ok(value == fw::GSP_PLUGIN_BOOTLOADED)
}
+ /// Initialize the shared control and response buffers for plugin RPC.
+ #[expect(dead_code)]
+ pub(super) fn initialize(&self) -> Result {
+ self.write_u64(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.response_buff_offset),
+ u64::try_from(self.region_offset(&self.response)?)?,
+ )?;
+ self.write_u64(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_buff_offset),
+ u64::try_from(self.region_offset(&self.message)?)?,
+ )?;
+ self.write_u64(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.migration_buff_offset),
+ u64::try_from(self.region_offset(&self.migration)?)?,
+ )?;
+ self.write_u64(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.error_buff_offset),
+ u64::try_from(self.region_offset(&self.error)?)?,
+ )?;
+ self.write_u64(
+ &self.control,
+ core::mem::offset_of!(
+ RawControlRegion,
+ __bindgen_anon_1.guest_rpc_trace_buff_offset
+ ),
+ u64::try_from(self.region_offset(&self.guest_trace)?)?,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(
+ RawControlRegion,
+ __bindgen_anon_1.migration_buf_cpu_access_offset
+ ),
+ 0,
+ )?;
+ self.write_u8(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.is_migration_in_progress),
+ 0,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.error_buff_cpu_get_idx),
+ 0,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(
+ RawControlRegion,
+ __bindgen_anon_1.guest_rpc_trace_buff_cpu_get_idx
+ ),
+ 0,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.attached_vgpu_count),
+ 1,
+ )?;
+ self.write_u8(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.is_gr_init_done),
+ 0,
+ )?;
+
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_type),
+ 0,
+ )?;
+ // Replace the boot-ready marker with the initial RPC sequence.
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_seq_num),
+ 0,
+ )?;
+ self.write_u32(
+ &self.response,
+ core::mem::offset_of!(
+ RawResponseRegion,
+ __bindgen_anon_1.message_seq_num_processed
+ ),
+ 0,
+ )?;
+ self.write_u32(
+ &self.response,
+ core::mem::offset_of!(RawResponseRegion, __bindgen_anon_1.result_code),
+ 0,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.version),
+ fw::VGPU_CPU_GSP_CTRL_BUFF_VERSION,
+ )
+ }
+
/// Invalidate the PTEs and release the communication mapping.
pub(super) fn unmap(&mut self) -> Result {
self.map.unmap()
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 24/31] gpu: nova-core: vgpu: add GSP plugin RPC transactions
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (22 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 23/31] gpu: nova-core: vgpu: initialize GSP plugin RPC buffers Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 25/31] gpu: nova-core: vgpu: negotiate the GSP plugin RPC version Zhi Wang
` (6 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
A GSP plugin consumes a shared-buffer request after its VF doorbell is
rung and publishes completion in the response region.
Add PluginRpc to own the communication mapping together with the
instance's stable BAR0 and validated GFID. Accept typed RPC message IDs,
copy the payload before publishing its sequence, advance the local
sequence, ring and read back the doorbell, then wait for a matching
completion and check its firmware status.
Keep one sequence identity for each published request, including a
request whose doorbell or wait later fails, and bound completion polling
with the RPC timeout. Delegate explicit unmap to the owned communication
region so teardown errors retain its owner; ordinary drop releases the
mapping through BarMapping.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/regs.rs | 31 +++-
drivers/gpu/nova-core/vgpu.rs | 1 +
drivers/gpu/nova-core/vgpu/fw.rs | 13 ++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 10 ++
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 51 +++++-
drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs | 155 ++++++++++++++++++
drivers/gpu/nova-core/vgpu/instance.rs | 15 +-
7 files changed, 267 insertions(+), 9 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
diff --git a/drivers/gpu/nova-core/regs.rs b/drivers/gpu/nova-core/regs.rs
index ec8e05dc3351..cadaa3495b4e 100644
--- a/drivers/gpu/nova-core/regs.rs
+++ b/drivers/gpu/nova-core/regs.rs
@@ -7,13 +7,17 @@
Io,
Mmio, //
},
+ prelude::*,
sizes::SizeConstants,
time, //
};
use pin_init::Zeroable;
use crate::{
- driver::NovaRegisters,
+ driver::{
+ Bar0,
+ NovaRegisters, //
+ },
falcon::{
DmaTrfCmdSize,
FalconCoreRev,
@@ -39,6 +43,31 @@
pub(crate) NV_PBUS_SW_SCRATCH(u32)[64] @ 0x00001400 {}
}
+// VIRTUAL_FUNCTION
+
+register! {
+ base: NovaRegisters;
+
+ // PF BAR0 exposes the virtual-function register window at 0x00b8_0000.
+ pub(crate) NV_VIRTUAL_FUNCTION_PRIV_DOORBELL(u32) @ 0x00b8_2200 {
+ 31:0 handle;
+ }
+}
+
+impl NV_VIRTUAL_FUNCTION_PRIV_DOORBELL {
+ const DOORBELL_STRIDE: u32 = 32;
+ const DOORBELL_VECTOR: u32 = 17;
+
+ /// Notify the GSP plugin for the given guest function and read back the doorbell.
+ pub(crate) fn ring_gsp_plugin(bar0: Bar0<'_>, gfid: u16) -> Result {
+ // A `u16` GFID produces a handle of at most 0x1f_fff1, which fits in `u32`.
+ let value = u32::from(gfid) * Self::DOORBELL_STRIDE + Self::DOORBELL_VECTOR;
+ bar0.try_write_reg(Self::zeroed().with_handle(value))?;
+ bar0.try_read(NV_VIRTUAL_FUNCTION_PRIV_DOORBELL)?;
+ Ok(())
+ }
+}
+
// PGC6 register space.
//
// `GC6` is a GPU low-power state where VRAM is in self-refresh and the GPU is powered down (except
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index f7184e7009a5..d6bcdabafb42 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -35,6 +35,7 @@
mod commands;
mod fw;
mod gsp_plugin_comm;
+mod gsp_plugin_rpc;
mod hal;
mod instance;
mod vram;
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index 03225b9833c3..d80f36465296 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -10,6 +10,8 @@
use crate::gsp::bindings;
+pub(super) use commands::RpcMessage;
+
pub(super) use bindings::{
GSP_PLUGIN_BOOTLOADED,
VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE,
@@ -44,3 +46,14 @@
pub(super) const GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES;
+
+/// State observed in the response buffer for an expected RPC sequence.
+pub(super) enum RpcResponse {
+ Pending {
+ /// Last sequence completed by firmware.
+ sequence: u32,
+ },
+ Complete {
+ status: u32,
+ },
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index bf8d44819e13..91d43cf6dd85 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -28,6 +28,16 @@
mm::vram::VramRegion, //
};
+use super::bindings;
+
+/// Message types supported by the nova-core plugin RPC channel.
+#[expect(dead_code)]
+#[derive(Clone, Copy)]
+#[repr(u32)]
+pub(crate) enum RpcMessage {
+ VersionNegotiation = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_VERSION_NEGOTIATION,
+}
+
bitfield! {
pub(crate) struct Dbdf(u32) {
2:0 function;
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index 31da31de4fd2..3d2c69e8cd79 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -21,7 +21,9 @@
use super::fw::{
self,
RawControlRegion,
- RawResponseRegion, //
+ RawResponseRegion,
+ RpcMessage,
+ RpcResponse, //
};
static_assert!(
@@ -240,7 +242,6 @@ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
}
/// Initialize the shared control and response buffers for plugin RPC.
- #[expect(dead_code)]
pub(super) fn initialize(&self) -> Result {
self.write_u64(
&self.control,
@@ -338,6 +339,52 @@ pub(super) fn initialize(&self) -> Result {
)
}
+ /// Copy and publish one RPC request to firmware.
+ pub(super) fn submit(&self, message: RpcMessage, sequence: u32, data: &[u8]) -> Result {
+ if u64::try_from(data.len()).map_err(|_| EOVERFLOW)? > self.message.size() {
+ return Err(E2BIG);
+ }
+
+ for (index, chunk) in data.chunks(size_of::<u32>()).enumerate() {
+ let mut bytes = [0u8; size_of::<u32>()];
+ bytes[..chunk.len()].copy_from_slice(chunk);
+ let field = index.checked_mul(size_of::<u32>()).ok_or(EOVERFLOW)?;
+ self.write_u32(&self.message, field, u32::from_le_bytes(bytes))?;
+ }
+
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_type),
+ // CAST: `RpcMessage` has a `u32` representation.
+ message as u32,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_seq_num),
+ sequence,
+ )
+ }
+
+ /// Read firmware's response for an expected RPC sequence.
+ pub(super) fn response(&self, expected_sequence: u32) -> Result<RpcResponse> {
+ let sequence = self.read_u32(
+ &self.response,
+ core::mem::offset_of!(
+ RawResponseRegion,
+ __bindgen_anon_1.message_seq_num_processed
+ ),
+ )?;
+ if sequence != expected_sequence {
+ return Ok(RpcResponse::Pending { sequence });
+ }
+
+ let status = self.read_u32(
+ &self.response,
+ core::mem::offset_of!(RawResponseRegion, __bindgen_anon_1.result_code),
+ )?;
+ Ok(RpcResponse::Complete { status })
+ }
+
/// Invalidate the PTEs and release the communication mapping.
pub(super) fn unmap(&mut self) -> Result {
self.map.unmap()
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
new file mode 100644
index 000000000000..0ee32f011f63
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
@@ -0,0 +1,155 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! GSP plugin RPC.
+//!
+//! ```text
+//! Host (PluginRpc) Shared RPC buffer (VRAM) GSP plugin
+//! | | |
+//! |-- BAR1: payload --------->| Message |
+//! |-- BAR1: type, sequence -->| Control |
+//! | | |
+//! |-- BAR0: VF doorbell ----------------------------->|
+//! | |<-- read request ------|
+//! | | | process RPC
+//! | |<-- completion --------|
+//! |-- poll processed seq ---->| Response |
+//! |<-- matching seq, status --| |
+//! ```
+
+use kernel::{
+ device,
+ prelude::*,
+ time::{
+ delay::fsleep,
+ Delta,
+ Instant,
+ Monotonic, //
+ },
+};
+
+use crate::{
+ driver::Bar0,
+ regs::NV_VIRTUAL_FUNCTION_PRIV_DOORBELL, //
+};
+
+use super::{
+ fw::{
+ RpcMessage,
+ RpcResponse, //
+ },
+ gsp_plugin_comm::CommBufferRegion,
+ instance::Gfid, //
+};
+
+/// BAR1-backed channel used to communicate with one GSP plugin.
+pub(super) struct PluginRpc<'map, 'gpu> {
+ comm: CommBufferRegion<'map, 'gpu>,
+ bar0: Bar0<'gpu>,
+ gfid: Gfid,
+ message_sequence: u32,
+}
+
+impl<'map, 'gpu> PluginRpc<'map, 'gpu> {
+ pub(super) fn new(comm: CommBufferRegion<'map, 'gpu>, bar0: Bar0<'gpu>, gfid: Gfid) -> Self {
+ Self {
+ comm,
+ bar0,
+ gfid,
+ message_sequence: 0,
+ }
+ }
+
+ pub(super) fn comm(&self) -> &CommBufferRegion<'map, 'gpu> {
+ &self.comm
+ }
+
+ /// Initialize the control and response buffers for the first RPC.
+ #[expect(dead_code)]
+ pub(super) fn init_rpc(&mut self) -> Result {
+ self.comm.initialize()?;
+ self.message_sequence = 0;
+ Ok(())
+ }
+
+ fn next_sequence(&self) -> u32 {
+ let sequence = self.message_sequence.wrapping_add(1);
+ if sequence == 0 {
+ 1
+ } else {
+ sequence
+ }
+ }
+
+ /// Write one RPC message, ring the VF doorbell, and wait for its response.
+ #[expect(dead_code)]
+ pub(super) fn rpc_call(
+ &mut self,
+ dev: &device::Device<device::Bound>,
+ message_type: RpcMessage,
+ data: &[u8],
+ ) -> Result {
+ let sequence = self.next_sequence();
+ self.comm.submit(message_type, sequence, data)?;
+ self.message_sequence = sequence;
+
+ dev_dbg!(
+ dev,
+ "vGPU RPC: gfid={} type={} bytes={} sequence={}\n",
+ self.gfid.get(),
+ // CAST: `RpcMessage` has a `u32` representation.
+ message_type as u32,
+ data.len(),
+ sequence,
+ );
+
+ NV_VIRTUAL_FUNCTION_PRIV_DOORBELL::ring_gsp_plugin(self.bar0, self.gfid.get())?;
+ self.wait_response(dev, sequence)
+ }
+
+ fn wait_response(&self, dev: &device::Device<device::Bound>, expected_sequence: u32) -> Result {
+ let start = Instant::<Monotonic>::now();
+ let timeout = Delta::from_secs(120);
+
+ loop {
+ match self.comm.response(expected_sequence)? {
+ RpcResponse::Complete { status } => {
+ if status != 0 {
+ dev_dbg!(
+ dev,
+ "vGPU RPC: sequence {} failed with status {}\n",
+ expected_sequence,
+ status,
+ );
+ return Err(EIO);
+ }
+
+ dev_dbg!(
+ dev,
+ "vGPU RPC: sequence {} completed after {:?}\n",
+ expected_sequence,
+ start.elapsed(),
+ );
+ return Ok(());
+ }
+ RpcResponse::Pending { sequence } => {
+ if start.elapsed() >= timeout {
+ dev_dbg!(
+ dev,
+ "vGPU RPC: sequence {} timed out; last response was {}\n",
+ expected_sequence,
+ sequence,
+ );
+ return Err(ETIMEDOUT);
+ }
+ }
+ }
+ fsleep(Delta::from_millis(1));
+ }
+ }
+
+ /// Release the BAR1 mapping.
+ pub(super) fn unmap(&mut self) -> Result {
+ self.comm.unmap()
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index bf02595590e1..8a2d8e3970a3 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -20,6 +20,7 @@
};
use crate::{
+ driver::Bar0,
gpu::ChannelIdReservation,
gsp::{
cmdq::Cmdq,
@@ -41,6 +42,7 @@
ChannelMapEntry, //
},
gsp_plugin_comm::CommBufferRegion,
+ gsp_plugin_rpc::PluginRpc,
vram::{
VgpuVramLayout,
VgpuVramSlot,
@@ -141,7 +143,7 @@ struct VgpuInstance<'gpu> {
vgpu_type: VgpuType,
vm_pid: u32,
num_plugin_channels: u32,
- comm: CommBufferRegion<'gpu, 'gpu>,
+ plugin_rpc: PluginRpc<'gpu, 'gpu>,
// Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
chids: ChannelIdReservation<'gpu>,
@@ -170,7 +172,7 @@ fn teardown(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
send_cleanup(vgpu.dev, vgpu.cmdq, self.gfid)?;
self.needs_teardown = false;
}
- self.comm.unmap()
+ self.plugin_rpc.unmap()
})();
if let Err(error) = result {
self.failure = Some(error);
@@ -187,7 +189,7 @@ fn bootload(
) -> Result {
let fb = &self.vram_slot.fbmem;
let mgmt = &self.vram_slot.mgmt_heap;
- let logs = self.comm.plugin_logs();
+ let logs = self.plugin_rpc.comm().plugin_logs();
let payload = encode_vgpu_bootload(BootloadInfo {
dbdf: self.dbdf,
@@ -215,11 +217,11 @@ fn bootload(
payload.len() * size_of::<u64>(),
);
- self.comm.clear_plugin_ready()?;
+ self.plugin_rpc.comm().clear_plugin_ready()?;
self.needs_teardown = true;
send_bootload(dev, cmdq, &payload)?;
- wait_plugin_ready(dev, &self.comm)?;
+ wait_plugin_ready(dev, self.plugin_rpc.comm())?;
dev_dbg!(dev, "bootload: gfid={} plugin ready\n", self.gfid.get());
Ok(())
@@ -293,6 +295,7 @@ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<
fn allocate_instance<'a>(
&'a mut self,
vgpu: &'a VgpuManager<'gpu>,
+ bar0: Bar0<'gpu>,
info: InstanceInfo,
) -> Result<PendingInstance<'a, 'gpu>> {
let InstanceInfo {
@@ -351,7 +354,7 @@ fn allocate_instance<'a>(
vgpu_type,
vm_pid,
num_plugin_channels: PLUGIN_CHANNELS_PER_ENGINE,
- comm,
+ plugin_rpc: PluginRpc::new(comm, bar0, gfid),
vram_slot,
chids,
needs_teardown: false,
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 25/31] gpu: nova-core: vgpu: negotiate the GSP plugin RPC version
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (23 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 24/31] gpu: nova-core: vgpu: add GSP plugin RPC transactions Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 26/31] gpu: nova-core: vgpu: send GSP plugin configuration parameters Zhi Wang
` (5 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The host must negotiate the RPC protocol with the booted GSP plugin
before sending instance configuration.
After observing the plugin's ready marker, initialize the shared RPC
buffers and submit the existing typed version-negotiation request
through the instance's PluginRpc channel.
Co-developed-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/vgpu/commands.rs | 12 +++++++++++-
drivers/gpu/nova-core/vgpu/fw.rs | 4 ++--
drivers/gpu/nova-core/vgpu/fw/commands.rs | 1 -
drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs | 2 --
drivers/gpu/nova-core/vgpu/instance.rs | 3 +++
5 files changed, 16 insertions(+), 6 deletions(-)
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 52c6fc0ee335..268c3043d88b 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -27,13 +27,15 @@
commands::{
VgpuPropertiesSchema, //
},
+ RpcMessage, //
GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK,
GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES,
GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE,
GMCAPI_CMD_QUERY_VGPU_PROPERTIES,
GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK,
- GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE, //
+ GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE,
},
+ gsp_plugin_rpc::PluginRpc,
instance::Gfid, //
};
@@ -159,3 +161,11 @@ pub(super) fn send_cleanup(
dev_dbg!(dev, "cleanup: gfid={} done\n", gfid.get());
Ok(())
}
+
+/// Negotiate the host protocol with a bootloaded GSP plugin.
+pub(super) fn negotiate_plugin_version(
+ dev: &device::Device<device::Bound>,
+ rpc: &mut PluginRpc<'_, '_>,
+) -> Result {
+ rpc.rpc_call(dev, RpcMessage::VersionNegotiation, &[])
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index d80f36465296..6cec92c9b505 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -10,8 +10,6 @@
use crate::gsp::bindings;
-pub(super) use commands::RpcMessage;
-
pub(super) use bindings::{
GSP_PLUGIN_BOOTLOADED,
VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE,
@@ -29,6 +27,8 @@
VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE, //
};
+pub(super) use commands::RpcMessage;
+
pub(super) const GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE;
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index 91d43cf6dd85..915585706174 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -31,7 +31,6 @@
use super::bindings;
/// Message types supported by the nova-core plugin RPC channel.
-#[expect(dead_code)]
#[derive(Clone, Copy)]
#[repr(u32)]
pub(crate) enum RpcMessage {
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
index 0ee32f011f63..696127c7169f 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
@@ -65,7 +65,6 @@ pub(super) fn comm(&self) -> &CommBufferRegion<'map, 'gpu> {
}
/// Initialize the control and response buffers for the first RPC.
- #[expect(dead_code)]
pub(super) fn init_rpc(&mut self) -> Result {
self.comm.initialize()?;
self.message_sequence = 0;
@@ -82,7 +81,6 @@ fn next_sequence(&self) -> u32 {
}
/// Write one RPC message, ring the VF doorbell, and wait for its response.
- #[expect(dead_code)]
pub(super) fn rpc_call(
&mut self,
dev: &device::Device<device::Bound>,
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 8a2d8e3970a3..445d7f513fdc 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -31,6 +31,7 @@
use super::{
commands::{
+ negotiate_plugin_version,
send_bootload,
send_cleanup,
send_shutdown,
@@ -156,6 +157,8 @@ impl<'gpu> VgpuInstance<'gpu> {
fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
let dev = vgpu.dev;
self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
+ self.plugin_rpc.init_rpc()?;
+ negotiate_plugin_version(dev, &mut self.plugin_rpc)?;
Ok(())
}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 26/31] gpu: nova-core: vgpu: send GSP plugin configuration parameters
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (24 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 25/31] gpu: nova-core: vgpu: negotiate the GSP plugin RPC version Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 27/31] gpu: nova-core: vgpu: update the GSP plugin BME state Zhi Wang
` (4 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The GSP plugin needs instance identity, channel allocation and host
parameters before it can serve guest requests.
Encode the configuration as an NVKV stream, prefix it with its word
count and send SETUP_CONFIG_PARAMS_AND_INIT after protocol negotiation
through the instance's PluginRpc channel.
Derive the host's opaque device identity from the VF DBDF within the
supported 16-bit PCI segment namespace. This identifies the local device
without imposing a migration-persistent identity.
Describe the supported whole-GPU configuration and distinguish the
optional VMM capability bitmap from the separately reported migration
features. Encode the host page size with the kernel's infallible
conversion.
Co-developed-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/vgpu/commands.rs | 9 ++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 135 +++++++++++++++++++
drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs | 17 +++
drivers/gpu/nova-core/vgpu/instance.rs | 16 +++
4 files changed, 177 insertions(+)
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 268c3043d88b..ec40a960069b 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -169,3 +169,12 @@ pub(super) fn negotiate_plugin_version(
) -> Result {
rpc.rpc_call(dev, RpcMessage::VersionNegotiation, &[])
}
+
+/// Send an instance's encoded configuration to the GSP plugin.
+pub(super) fn send_plugin_config(
+ dev: &device::Device<device::Bound>,
+ rpc: &mut PluginRpc<'_, '_>,
+ config: &[u64],
+) -> Result {
+ rpc.rpc_call_nvkv(dev, RpcMessage::SetupConfigParamsAndInit, config)
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index 915585706174..e3b5198e9c80 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -9,6 +9,7 @@
use kernel::{
alloc::ArrayVec,
bitfield,
+ num::casts::usize_as_u64,
prelude::*, //
};
@@ -35,6 +36,7 @@
#[repr(u32)]
pub(crate) enum RpcMessage {
VersionNegotiation = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_VERSION_NEGOTIATION,
+ SetupConfigParamsAndInit = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_SETUP_CONFIG_PARAMS_AND_INIT,
}
bitfield! {
@@ -263,3 +265,136 @@ pub(crate) struct VgpuProperties {
impl VgpuProperties {
const STRING_LEN: usize = 64;
}
+
+#[derive(Clone, Copy)]
+#[repr(u32)]
+enum HypervisorType {
+ Unknown = 4,
+}
+
+impl From<HypervisorType> for u32 {
+ fn from(value: HypervisorType) -> Self {
+ // CAST: `HypervisorType` uses the wire field's `u32` representation.
+ value as u32
+ }
+}
+
+#[derive(Clone, Copy)]
+#[repr(u32)]
+enum CpuArch {
+ Aarch64 = 1,
+ X86_64 = 2,
+}
+
+impl CpuArch {
+ fn host() -> Result<Self> {
+ if cfg!(target_arch = "x86_64") {
+ Ok(Self::X86_64)
+ } else if cfg!(target_arch = "aarch64") {
+ Ok(Self::Aarch64)
+ } else {
+ Err(EOPNOTSUPP)
+ }
+ }
+}
+
+impl From<CpuArch> for u32 {
+ fn from(value: CpuArch) -> Self {
+ // CAST: `CpuArch` uses the wire field's `u32` representation.
+ value as u32
+ }
+}
+
+#[derive(Clone, Copy)]
+struct MigrationFeature(u32);
+
+impl MigrationFeature {
+ const PRESERVE_CTX_BUF: Self = Self(0x4000);
+}
+
+impl From<MigrationFeature> for u32 {
+ fn from(value: MigrationFeature) -> Self {
+ value.0
+ }
+}
+
+bitfield! {
+ struct FeatureFlags(u64) {
+ 3:3 enable_uvm => bool;
+ 5:5 vmm_migration => bool;
+ }
+}
+
+nvkv_encode! {
+ struct PluginConfigParamsRequest {
+ uuid: Key<[u8; 16], { Self::UUID_KEY }>,
+ dbdf: Key<Dbdf, { Self::DBDF_KEY }, u32>,
+ device_instance_id: Key<u32, { Self::DEVICE_INSTANCE_ID_KEY }>,
+ vgpu_type: Key<u32, { Self::VGPU_TYPE_KEY }>,
+ vm_pid: Key<u32, { Self::VM_PID_KEY }>,
+ swizz_id: Key<SwizzId, { Self::SWIZZ_ID_KEY }, u32>,
+ num_channels: Key<u32, { Self::NUM_CHANNELS_KEY }>,
+ num_plugin_channels: Key<u32, { Self::NUM_PLUGIN_CHANNELS_KEY }>,
+ vmm_cap: Key<u32, { Self::VMM_CAP_KEY }>,
+ migration_feature: Key<MigrationFeature, { Self::MIGRATION_FEATURE_KEY }, u32>,
+ hypervisor_type: Key<HypervisorType, { Self::HYPERVISOR_TYPE_KEY }, u32>,
+ cpu_arch: Key<CpuArch, { Self::CPU_ARCH_KEY }, u32>,
+ page_size: Key<u64, { Self::PAGE_SIZE_KEY }>,
+ feature_flags: Key<FeatureFlags, { Self::FEATURE_FLAGS_KEY }, u64>,
+ }
+}
+
+impl PluginConfigParamsRequest {
+ const UUID_KEY: KeyId = 0x0001;
+ const DBDF_KEY: KeyId = 0x0002;
+ const DEVICE_INSTANCE_ID_KEY: KeyId = 0x0004;
+ const VGPU_TYPE_KEY: KeyId = 0x0005;
+ const VM_PID_KEY: KeyId = 0x0006;
+ const SWIZZ_ID_KEY: KeyId = 0x0010;
+ const NUM_CHANNELS_KEY: KeyId = 0x0011;
+ const NUM_PLUGIN_CHANNELS_KEY: KeyId = 0x0012;
+ const VMM_CAP_KEY: KeyId = 0x0020;
+ const MIGRATION_FEATURE_KEY: KeyId = 0x0021;
+ const HYPERVISOR_TYPE_KEY: KeyId = 0x0022;
+ const CPU_ARCH_KEY: KeyId = 0x0023;
+ const PAGE_SIZE_KEY: KeyId = 0x0024;
+ const FEATURE_FLAGS_KEY: KeyId = 0x0030;
+}
+
+/// Encodes plugin configuration parameters using the typed NVKV schema.
+pub(crate) fn encode_plugin_config_params(
+ uuid: [u8; 16],
+ dbdf: Dbdf,
+ vgpu_type: u32,
+ vm_pid: u32,
+ num_channels: u32,
+ num_plugin_channels: u32,
+) -> Result<EncodedStream> {
+ let request = PluginConfigParamsRequest {
+ uuid: uuid.into(),
+ dbdf: dbdf.into(),
+ // The full host PCI address distinguishes VFs within a VM, including VFs
+ // from different PFs. This opaque identity lasts for the host device lifetime.
+ device_instance_id: dbdf.into_raw().into(),
+ vgpu_type: vgpu_type.into(),
+ vm_pid: vm_pid.into(),
+ swizz_id: SwizzId::WHOLE_GPU.into(),
+ num_channels: num_channels.into(),
+ num_plugin_channels: num_plugin_channels.into(),
+ // Advertise no optional VMM capabilities (vmioplugin.h VMM_CAP_*),
+ // independently of the plugin feature_flags and migration_feature below.
+ vmm_cap: 0.into(),
+ migration_feature: MigrationFeature::PRESERVE_CTX_BUF.into(),
+ hypervisor_type: HypervisorType::Unknown.into(),
+ cpu_arch: CpuArch::host()?.into(),
+ page_size: usize_as_u64(kernel::page::PAGE_SIZE).into(),
+ feature_flags: FeatureFlags::zeroed()
+ .with_enable_uvm(false)
+ .with_vmm_migration(true)
+ .into(),
+ };
+
+ let mut encoder = Encoder::new();
+ request.encode(&mut encoder)?;
+ Ok(encoder.finish())
+}
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
index 696127c7169f..48977cd67792 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
@@ -19,6 +19,7 @@
use kernel::{
device,
+ num::casts::usize_as_u64,
prelude::*,
time::{
delay::fsleep,
@@ -26,6 +27,7 @@
Instant,
Monotonic, //
},
+ transmute::AsBytes, //
};
use crate::{
@@ -105,6 +107,21 @@ pub(super) fn rpc_call(
self.wait_response(dev, sequence)
}
+ /// Send an NVKV stream prefixed by its word count.
+ pub(super) fn rpc_call_nvkv(
+ &mut self,
+ dev: &device::Device<device::Bound>,
+ message_type: RpcMessage,
+ encoded: &[u64],
+ ) -> Result {
+ let word_count = usize_as_u64(encoded.len());
+ let mut payload = KVec::new();
+ payload.extend_from_slice(&word_count.to_le_bytes(), GFP_KERNEL)?;
+ payload.extend_from_slice(AsBytes::as_bytes(encoded), GFP_KERNEL)?;
+
+ self.rpc_call(dev, message_type, &payload)
+ }
+
fn wait_response(&self, dev: &device::Device<device::Bound>, expected_sequence: u32) -> Result {
let start = Instant::<Monotonic>::now();
let timeout = Delta::from_secs(120);
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 445d7f513fdc..247e88bdc82f 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -34,10 +34,12 @@
negotiate_plugin_version,
send_bootload,
send_cleanup,
+ send_plugin_config,
send_shutdown,
Dbdf, //
},
fw::commands::{
+ encode_plugin_config_params,
encode_vgpu_bootload,
BootloadInfo,
ChannelMapEntry, //
@@ -159,6 +161,7 @@ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
self.plugin_rpc.init_rpc()?;
negotiate_plugin_version(dev, &mut self.plugin_rpc)?;
+ self.configure_plugin(dev)?;
Ok(())
}
@@ -230,6 +233,19 @@ fn bootload(
Ok(())
}
+ fn configure_plugin(&mut self, dev: &device::Device<device::Bound>) -> Result {
+ let config = encode_plugin_config_params(
+ [0; 16],
+ self.dbdf,
+ self.vgpu_type.vgpu_type_id,
+ self.vm_pid,
+ u32::try_from(self.chids.len()).map_err(|_| EOVERFLOW)?,
+ self.num_plugin_channels,
+ )?;
+
+ send_plugin_config(dev, &mut self.plugin_rpc, &config)
+ }
+
/// Stop the plugin when firmware may own instance resources.
fn shutdown(&mut self, dev: &device::Device<device::Bound>, cmdq: &Cmdq<'_>) -> Result {
if self.needs_teardown {
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 27/31] gpu: nova-core: vgpu: update the GSP plugin BME state
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (25 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 26/31] gpu: nova-core: vgpu: send GSP plugin configuration parameters Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 28/31] gpu: nova-core: vgpu: add CeUtils commands Zhi Wang
` (3 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
The vGPU manager reports the enabled bus-mastering state to the GSP plugin
after its initial configuration. The guest can then use the vGPU from now
on.
Add the BME state request and NVKV encoder, and send the enabled state
after configuration succeeds.
Co-developed-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Eliot Courtney <ecourtney@nvidia.com>
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/vgpu/commands.rs | 11 +++++++++++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 22 ++++++++++++++++++++++
drivers/gpu/nova-core/vgpu/instance.rs | 2 ++
3 files changed, 35 insertions(+)
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index ec40a960069b..bfb033e5eb37 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -25,6 +25,7 @@
use super::{
fw::{
commands::{
+ encode_plugin_set_bme,
VgpuPropertiesSchema, //
},
RpcMessage, //
@@ -178,3 +179,13 @@ pub(super) fn send_plugin_config(
) -> Result {
rpc.rpc_call_nvkv(dev, RpcMessage::SetupConfigParamsAndInit, config)
}
+
+/// Update the bus-mastering state reported to the GSP plugin.
+pub(super) fn set_plugin_bme(
+ dev: &device::Device<device::Bound>,
+ rpc: &mut PluginRpc<'_, '_>,
+ enable: bool,
+) -> Result {
+ let bme = encode_plugin_set_bme(enable)?;
+ rpc.rpc_call_nvkv(dev, RpcMessage::UpdateBmeState, &bme)
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index e3b5198e9c80..cc2298c3679c 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -37,6 +37,7 @@
pub(crate) enum RpcMessage {
VersionNegotiation = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_VERSION_NEGOTIATION,
SetupConfigParamsAndInit = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_SETUP_CONFIG_PARAMS_AND_INIT,
+ UpdateBmeState = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_UPDATE_BME_STATE,
}
bitfield! {
@@ -398,3 +399,24 @@ pub(crate) fn encode_plugin_config_params(
request.encode(&mut encoder)?;
Ok(encoder.finish())
}
+
+nvkv_encode! {
+ struct PluginSetBmeRequest {
+ bme_enable: Key<bool, { Self::BME_ENABLE_KEY }, u32>,
+ }
+}
+
+impl PluginSetBmeRequest {
+ const BME_ENABLE_KEY: KeyId = 0x0100;
+}
+
+/// Encodes a plugin BME state update using the typed NVKV schema.
+pub(crate) fn encode_plugin_set_bme(enable: bool) -> Result<EncodedStream> {
+ let request = PluginSetBmeRequest {
+ bme_enable: enable.into(),
+ };
+
+ let mut encoder = Encoder::new();
+ request.encode(&mut encoder)?;
+ Ok(encoder.finish())
+}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 247e88bdc82f..23c2b791aa25 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -36,6 +36,7 @@
send_cleanup,
send_plugin_config,
send_shutdown,
+ set_plugin_bme,
Dbdf, //
},
fw::commands::{
@@ -162,6 +163,7 @@ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
self.plugin_rpc.init_rpc()?;
negotiate_plugin_version(dev, &mut self.plugin_rpc)?;
self.configure_plugin(dev)?;
+ set_plugin_bme(dev, &mut self.plugin_rpc, true)?;
Ok(())
}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 28/31] gpu: nova-core: vgpu: add CeUtils commands
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (26 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 27/31] gpu: nova-core: vgpu: update the GSP plugin BME state Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 29/31] gpu: nova-core: vgpu: scrub guest VRAM with CeUtils Zhi Wang
` (2 subsequent siblings)
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Add Cmdq operations to allocate and free a CeUtils channel and submit
guest VRAM scrub requests. Decode the allocation and scrub replies, and
validate the returned semaphore address and aperture.
Distinguish an explicit allocation rejection from failures that may leave
firmware owning the channel, so callers can decide whether to retain its
host resources.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
.../gpu/nova-core/gsp/fw/r000_00/bindings.rs | 1 +
drivers/gpu/nova-core/vgpu/commands.rs | 143 +++++++++++++++++-
drivers/gpu/nova-core/vgpu/fw.rs | 11 ++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 52 ++++++-
4 files changed, 197 insertions(+), 10 deletions(-)
diff --git a/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs b/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
index 3cc718bb2ab8..333256ea8a04 100644
--- a/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
+++ b/drivers/gpu/nova-core/gsp/fw/r000_00/bindings.rs
@@ -851,6 +851,7 @@ fn default() -> Self {
}
}
}
+pub const NV_ADDR_FBMEM: u32 = 2;
pub const GSP_PLUGIN_BOOTLOADED: u32 = 1315261039;
pub const VGPU_CPU_GSP_CTRL_BUFF_VERSION: u32 = 2;
pub const VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE: u32 = 4096;
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index bfb033e5eb37..704d13259cf3 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -8,33 +8,49 @@
use kernel::{
device,
+ num::casts::{
+ usize_as_u64,
+ usize_into_u32, //
+ },
prelude::*,
time::Delta,
transmute::AsBytes, //
};
-use crate::gsp::{
- cmdq::Cmdq,
- nvkv::{
- nvkv_words,
- Decoder,
- UnknownKeyPolicy, //
- }, //
+use crate::{
+ gsp::{
+ cmdq::Cmdq,
+ nvkv::{
+ nvkv_words,
+ Decoder,
+ UnknownKeyPolicy, //
+ }, //
+ },
+ mm::PAGE_SIZE, //
};
use super::{
fw::{
commands::{
encode_plugin_set_bme,
+ AllocCeutilsRequest,
+ AllocCeutilsResponse,
+ FreeCeutilsRequest,
+ ScrubGuestFbRequest,
+ ScrubGuestFbResponse,
VgpuPropertiesSchema, //
},
- RpcMessage, //
+ RpcMessage,
GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK,
GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES,
GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE,
GMCAPI_CMD_QUERY_VGPU_PROPERTIES,
GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK,
GMCAPI_CMD_SHUTDOWN_GSP_VGPU_PLUGIN_TASK_COMPLETE,
+ GMCAPI_CMD_VGPU_MGR_ALLOC_GSP_CEUTILS,
+ GMCAPI_CMD_VGPU_MGR_FREE_GSP_CEUTILS,
+ GMCAPI_CMD_VGPU_MGR_SCRUB_GUEST_FB,
+ NV_ADDR_FBMEM, //
},
gsp_plugin_rpc::PluginRpc,
instance::Gfid, //
@@ -189,3 +205,114 @@ pub(super) fn set_plugin_bme(
let bme = encode_plugin_set_bme(enable)?;
rpc.rpc_call_nvkv(dev, RpcMessage::UpdateBmeState, &bme)
}
+
+/// Whether a failed allocation may still have transferred CHID ownership to firmware.
+#[expect(dead_code)]
+pub(super) enum CeUtilsAllocError {
+ /// A matching firmware response explicitly rejected the allocation.
+ NotOwned(Error),
+ /// The request may have completed despite a transport or response-validation error.
+ MayOwn(Error),
+}
+
+/// Allocate a CeUtils channel and validate its semaphore description.
+#[expect(dead_code)]
+pub(super) fn alloc_ceutils(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+ chid: u32,
+) -> core::result::Result<u64, CeUtilsAllocError> {
+ let request = AllocCeutilsRequest {
+ gfid: u32::from(gfid.get()).to_le(),
+ fixed_chid: chid.to_le(),
+ force_ceid: u32::MAX.to_le(),
+ swizz_id: 0,
+ };
+
+ dev_dbg!(dev, "alloc CeUtils: gfid={} chid={}\n", gfid.get(), chid,);
+
+ let command_id = GMCAPI_CMD_VGPU_MGR_ALLOC_GSP_CEUTILS;
+ let response = cmdq
+ .send_gmc_and_receive(
+ command_id,
+ IntoBytes::as_bytes(&request),
+ usize_into_u32::<{ size_of::<AllocCeutilsResponse>() }>(),
+ )
+ .map_err(CeUtilsAllocError::MayOwn)?;
+ check_status(dev, command_id, response.status).map_err(CeUtilsAllocError::NotOwned)?;
+ let (response, _) = AllocCeutilsResponse::read_from_prefix(&response.payload)
+ .map_err(|_| CeUtilsAllocError::MayOwn(EMSGSIZE))?;
+
+ let semaphore_address = u64::from_le(response.semaphore_address);
+ let semaphore_aperture = u32::from_le(response.semaphore_aperture);
+ let page_size = usize_as_u64(PAGE_SIZE);
+
+ if semaphore_address == 0
+ || !semaphore_address.is_multiple_of(page_size)
+ || semaphore_aperture != NV_ADDR_FBMEM
+ {
+ return Err(CeUtilsAllocError::MayOwn(EINVAL));
+ }
+
+ dev_dbg!(
+ dev,
+ "alloc CeUtils: gfid={} semaphore={:#x}\n",
+ gfid.get(),
+ semaphore_address,
+ );
+ Ok(semaphore_address)
+}
+
+/// Release a CeUtils allocation, including one whose allocation reply was lost.
+#[expect(dead_code)]
+pub(super) fn free_ceutils(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+) -> Result {
+ let request = FreeCeutilsRequest {
+ gfid: u32::from(gfid.get()).to_le(),
+ };
+
+ dev_dbg!(dev, "free CeUtils: gfid={}\n", gfid.get());
+ let command_id = GMCAPI_CMD_VGPU_MGR_FREE_GSP_CEUTILS;
+ let response = cmdq.send_gmc_and_receive(command_id, IntoBytes::as_bytes(&request), 0)?;
+ check_status(dev, command_id, response.status)
+}
+
+/// Submit an asynchronous guest FB scrub and return its work identifier.
+#[expect(dead_code)]
+pub(super) fn submit_ceutils_scrub(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+ fb_offset: u64,
+ fb_size: u64,
+) -> Result<u32> {
+ let request = ScrubGuestFbRequest {
+ gfid: u32::from(gfid.get()).to_le(),
+ reserved: 0,
+ fb_offset: fb_offset.to_le(),
+ fb_size: fb_size.to_le(),
+ };
+
+ dev_dbg!(
+ dev,
+ "submit scrub: gfid={} offset={:#x} size={:#x}\n",
+ gfid.get(),
+ fb_offset,
+ fb_size,
+ );
+
+ let command_id = GMCAPI_CMD_VGPU_MGR_SCRUB_GUEST_FB;
+ let response = cmdq.send_gmc_and_receive(
+ command_id,
+ IntoBytes::as_bytes(&request),
+ usize_into_u32::<{ size_of::<ScrubGuestFbResponse>() }>(),
+ )?;
+ check_status(dev, command_id, response.status)?;
+ let (response, _) =
+ ScrubGuestFbResponse::read_from_prefix(&response.payload).map_err(|_| EMSGSIZE)?;
+ u32::try_from(u64::from_le(response.work_id)).map_err(|_| EOVERFLOW)
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index 6cec92c9b505..795ce4855586 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -27,6 +27,8 @@
VGPU_CPU_GSP_VGPU_TASK_LOG_BUFF_REGION_SIZE, //
};
+pub(super) use bindings::NV_ADDR_FBMEM;
+
pub(super) use commands::RpcMessage;
pub(super) const GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE: u32 =
@@ -47,6 +49,15 @@
pub(super) const GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES;
+pub(super) const GMCAPI_CMD_VGPU_MGR_ALLOC_GSP_CEUTILS: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_VGPU_MGR_ALLOC_GSP_CEUTILS;
+
+pub(super) const GMCAPI_CMD_VGPU_MGR_FREE_GSP_CEUTILS: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_VGPU_MGR_FREE_GSP_CEUTILS;
+
+pub(super) const GMCAPI_CMD_VGPU_MGR_SCRUB_GUEST_FB: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_VGPU_MGR_SCRUB_GUEST_FB;
+
/// State observed in the response buffer for an expected RPC sequence.
pub(super) enum RpcResponse {
Pending {
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index cc2298c3679c..ad2b29340986 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -21,10 +21,10 @@
Encodable,
EncodedStream,
Encoder,
- Index, //
+ Index,
Key,
KeyId,
- Required,
+ Required, //
},
mm::vram::VramRegion, //
};
@@ -420,3 +420,51 @@ pub(crate) fn encode_plugin_set_bme(enable: bool) -> Result<EncodedStream> {
request.encode(&mut encoder)?;
Ok(encoder.finish())
}
+
+#[repr(C)]
+#[derive(IntoBytes, zerocopy_derive::Immutable)]
+pub(crate) struct AllocCeutilsRequest {
+ pub(crate) gfid: u32,
+ pub(crate) fixed_chid: u32,
+ pub(crate) force_ceid: u32,
+ pub(crate) swizz_id: u32,
+}
+
+static_assert!(size_of::<AllocCeutilsRequest>() == 16);
+
+#[repr(C)]
+#[derive(FromBytes)]
+pub(crate) struct AllocCeutilsResponse {
+ pub(crate) semaphore_address: u64,
+ pub(crate) semaphore_aperture: u32,
+ _reserved: u32,
+}
+
+static_assert!(size_of::<AllocCeutilsResponse>() == 16);
+
+#[repr(C)]
+#[derive(IntoBytes, zerocopy_derive::Immutable)]
+pub(crate) struct FreeCeutilsRequest {
+ pub(crate) gfid: u32,
+}
+
+static_assert!(size_of::<FreeCeutilsRequest>() == 4);
+
+#[repr(C)]
+#[derive(IntoBytes, zerocopy_derive::Immutable)]
+pub(crate) struct ScrubGuestFbRequest {
+ pub(crate) gfid: u32,
+ pub(crate) reserved: u32,
+ pub(crate) fb_offset: u64,
+ pub(crate) fb_size: u64,
+}
+
+static_assert!(size_of::<ScrubGuestFbRequest>() == 24);
+
+#[repr(C)]
+#[derive(FromBytes)]
+pub(crate) struct ScrubGuestFbResponse {
+ pub(crate) work_id: u64,
+}
+
+static_assert!(size_of::<ScrubGuestFbResponse>() == 8);
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 29/31] gpu: nova-core: vgpu: scrub guest VRAM with CeUtils
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (27 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 28/31] gpu: nova-core: vgpu: add CeUtils commands Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 30/31] gpu: nova-core: vgpu: export plugin log buffers via debugfs Zhi Wang
2026-09-28 10:28 ` [PATCH v3 31/31] gpu: nova-core: vgpu: introduce SR-IOV PF APIs Zhi Wang
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Scrub guest VRAM before plugin boot and after plugin shutdown, before its
slot can be reused. Reserve the last channel ID in each instance range
for CeUtils and give the remaining channels to the plugin.
Submit scrubs in chunks and wait for completion through a temporary BAR1
mapping of the CeUtils semaphore. Release the mapping on both success and
failure, preserving a polling error if unmapping also fails.
Keep allocation, scrubbing and release within the instance lifecycle.
Retain host resources until device removal when firmware ownership is
uncertain or a scrub fails, and do not retry a recorded teardown failure.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/vgpu.rs | 1 +
drivers/gpu/nova-core/vgpu/commands.rs | 4 -
drivers/gpu/nova-core/vgpu/instance.rs | 51 +++++++-
drivers/gpu/nova-core/vgpu/scrubber.rs | 173 +++++++++++++++++++++++++
4 files changed, 223 insertions(+), 6 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/scrubber.rs
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index d6bcdabafb42..f78a4621984a 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -38,6 +38,7 @@
mod gsp_plugin_rpc;
mod hal;
mod instance;
+mod scrubber;
mod vram;
/// vGPU state detected during GPU construction.
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 704d13259cf3..52e8339504c6 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -207,7 +207,6 @@ pub(super) fn set_plugin_bme(
}
/// Whether a failed allocation may still have transferred CHID ownership to firmware.
-#[expect(dead_code)]
pub(super) enum CeUtilsAllocError {
/// A matching firmware response explicitly rejected the allocation.
NotOwned(Error),
@@ -216,7 +215,6 @@ pub(super) enum CeUtilsAllocError {
}
/// Allocate a CeUtils channel and validate its semaphore description.
-#[expect(dead_code)]
pub(super) fn alloc_ceutils(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
@@ -265,7 +263,6 @@ pub(super) fn alloc_ceutils(
}
/// Release a CeUtils allocation, including one whose allocation reply was lost.
-#[expect(dead_code)]
pub(super) fn free_ceutils(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
@@ -282,7 +279,6 @@ pub(super) fn free_ceutils(
}
/// Submit an asynchronous guest FB scrub and return its work identifier.
-#[expect(dead_code)]
pub(super) fn submit_ceutils_scrub(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 23c2b791aa25..1ccf4309f2ed 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -31,12 +31,14 @@
use super::{
commands::{
+ free_ceutils,
negotiate_plugin_version,
send_bootload,
send_cleanup,
send_plugin_config,
send_shutdown,
set_plugin_bme,
+ CeUtilsAllocError,
Dbdf, //
},
fw::commands::{
@@ -47,6 +49,7 @@
},
gsp_plugin_comm::CommBufferRegion,
gsp_plugin_rpc::PluginRpc,
+ scrubber::CeUtils,
vram::{
VgpuVramLayout,
VgpuVramSlot,
@@ -151,12 +154,41 @@ struct VgpuInstance<'gpu> {
// Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
chids: ChannelIdReservation<'gpu>,
+ ceutils: Option<CeUtils>,
needs_teardown: bool,
/// An uncertain or failed operation retains resources until device removal.
failure: Option<Error>,
}
impl<'gpu> VgpuInstance<'gpu> {
+ fn initialize(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
+ let ceutils_chid =
+ u32::try_from(self.chids.end.checked_sub(1).ok_or(EINVAL)?).map_err(|_| EOVERFLOW)?;
+ let ceutils = match CeUtils::allocate(vgpu.dev, vgpu.cmdq, self.gfid, ceutils_chid) {
+ Ok(ceutils) => ceutils,
+ Err(CeUtilsAllocError::NotOwned(error)) => return Err(error),
+ Err(CeUtilsAllocError::MayOwn(error)) => {
+ self.failure = Some(error);
+ dev_err!(vgpu.dev, "CeUtils allocation failed: {:?}\n", error);
+ return Err(error);
+ }
+ };
+
+ let result = ceutils.scrub_guest_fb(
+ vgpu.dev,
+ vgpu.cmdq,
+ vgpu.bar_user,
+ vgpu.mm,
+ &self.vram_slot.fbmem,
+ );
+ self.ceutils = Some(ceutils);
+ if let Err(error) = result {
+ // A failed wait does not establish that the submitted scrub has stopped.
+ self.failure = Some(error);
+ }
+ result
+ }
+
fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
let dev = vgpu.dev;
self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
@@ -175,7 +207,17 @@ fn teardown(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
}
let result = (|| {
self.shutdown(vgpu.dev, vgpu.cmdq)?;
-
+ if let Some(ceutils) = self.ceutils.as_ref() {
+ ceutils.scrub_guest_fb(
+ vgpu.dev,
+ vgpu.cmdq,
+ vgpu.bar_user,
+ vgpu.mm,
+ &self.vram_slot.fbmem,
+ )?;
+ free_ceutils(vgpu.dev, vgpu.cmdq, self.gfid)?;
+ self.ceutils = None;
+ }
if self.needs_teardown {
send_cleanup(vgpu.dev, vgpu.cmdq, self.gfid)?;
self.needs_teardown = false;
@@ -241,7 +283,7 @@ fn configure_plugin(&mut self, dev: &device::Device<device::Bound>) -> Result {
self.dbdf,
self.vgpu_type.vgpu_type_id,
self.vm_pid,
- u32::try_from(self.chids.len()).map_err(|_| EOVERFLOW)?,
+ u32::try_from(self.chids.len().checked_sub(1).ok_or(EINVAL)?).map_err(|_| EOVERFLOW)?,
self.num_plugin_channels,
)?;
@@ -352,6 +394,9 @@ fn allocate_instance<'a>(
.total_channels
.checked_div(vgpu_type.max_instance)
.ok_or(EINVAL)?;
+ if channels_per_instance <= 1 {
+ return Err(EINVAL);
+ }
let channels_per_instance =
usize::try_from(channels_per_instance).map_err(|_| EOVERFLOW)?;
let channels_per_instance = NonZeroUsize::new(channels_per_instance).ok_or(EINVAL)?;
@@ -378,6 +423,7 @@ fn allocate_instance<'a>(
plugin_rpc: PluginRpc::new(comm, bar0, gfid),
vram_slot,
chids,
+ ceutils: None,
needs_teardown: false,
failure: None,
};
@@ -445,6 +491,7 @@ fn activate(mut self) -> Result {
.iter_mut()
.find(|instance| instance.gfid == self.gfid)
.ok_or(EIO)?;
+ instance.initialize(self.vgpu)?;
instance.activate(self.vgpu)?;
self.committed = true;
Ok(())
diff --git a/drivers/gpu/nova-core/vgpu/scrubber.rs b/drivers/gpu/nova-core/vgpu/scrubber.rs
new file mode 100644
index 000000000000..dc62ddc1ef01
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/scrubber.rs
@@ -0,0 +1,173 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! Per-VM CeUtils guest VRAM scrubbing.
+
+use kernel::{
+ device,
+ prelude::*,
+ sync::Mutex,
+ time::{
+ delay::fsleep,
+ Delta,
+ Instant,
+ Monotonic, //
+ },
+ types::ScopeGuard, //
+};
+
+use crate::{
+ gsp::cmdq::Cmdq,
+ mm::{
+ bar_user::BarUser,
+ vram::VramRegion,
+ GpuMm,
+ Pfn,
+ VramAddress, //
+ },
+ vgpu::instance::Gfid, //
+};
+
+use super::commands::{
+ self,
+ CeUtilsAllocError, //
+};
+
+// Semaphore layout from OpenRM `channel_utils.h`.
+const NV_CEUTILS_SEMA_PAGE_MAGIC: u32 = 0xce5e_5ea0;
+
+#[repr(C)]
+struct CeUtilsSemaphoreHeader {
+ magic: u32,
+ payload: u32,
+}
+
+static_assert!(size_of::<CeUtilsSemaphoreHeader>() == 8);
+
+const SEMA_PAGE_MAGIC_OFFSET: usize = core::mem::offset_of!(CeUtilsSemaphoreHeader, magic);
+const SEMA_PAGE_PAYLOAD_OFFSET: usize = core::mem::offset_of!(CeUtilsSemaphoreHeader, payload);
+
+const SCRUB_REQUEST_SIZE: u64 = 4 * 1024 * 1024 * 1024;
+
+// Host timeout policy; not a firmware ABI value.
+const SCRUB_TIMEOUT: Delta = Delta::from_secs(5);
+
+/// A firmware-owned per-VM CeUtils allocation.
+///
+/// Submitted scrubs must complete before releasing this allocation or its VRAM.
+pub(super) struct CeUtils {
+ gfid: Gfid,
+ semaphore_address: u64,
+}
+
+impl CeUtils {
+ pub(super) fn allocate(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+ chid: u32,
+ ) -> core::result::Result<Self, CeUtilsAllocError> {
+ let semaphore_address = commands::alloc_ceutils(dev, cmdq, gfid, chid)?;
+ Ok(Self {
+ gfid,
+ semaphore_address,
+ })
+ }
+
+ /// Scrub the complete guest VRAM and wait for semaphore completion.
+ pub(super) fn scrub_guest_fb(
+ &self,
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ bar_user: &BarUser<'_>,
+ mm: &Mutex<GpuMm<'_>>,
+ fb: &VramRegion,
+ ) -> Result {
+ let mut offset = fb.address();
+ let end = offset.checked_add(fb.size()).ok_or(EOVERFLOW)?;
+ while offset < end {
+ let size = core::cmp::min(SCRUB_REQUEST_SIZE, end - offset);
+ let work_id = commands::submit_ceutils_scrub(dev, cmdq, self.gfid, offset, size)?;
+ wait_scrub_complete(bar_user, mm, dev, self.semaphore_address, work_id)?;
+ offset = offset.checked_add(size).ok_or(EOVERFLOW)?;
+ }
+
+ Ok(())
+ }
+}
+
+/// Poll the GSP-owned CeUtils semaphore page through a temporary BAR1 map.
+fn wait_scrub_complete(
+ bar_user: &BarUser<'_>,
+ mm: &Mutex<GpuMm<'_>>,
+ dev: &device::Device<device::Bound>,
+ semaphore_address: u64,
+ work_id: u32,
+) -> Result {
+ let pfn = Pfn::from(VramAddress::from_raw(semaphore_address));
+ let semaphore_map = bar_user.map(&mut mm.lock(), &[pfn], false)?;
+ let semaphore_map = ScopeGuard::new_with_data(semaphore_map, |mapping| {
+ if let Err(error) = mapping.release(&mut mm.lock()) {
+ dev_err!(
+ dev,
+ "failed to release semaphore BAR1 mapping: {:?}\n",
+ error
+ );
+ }
+ });
+
+ let result = (|| {
+ let magic = semaphore_map.try_read32(SEMA_PAGE_MAGIC_OFFSET)?;
+ if magic != NV_CEUTILS_SEMA_PAGE_MAGIC {
+ dev_warn!(
+ dev,
+ "bad CeUtils semaphore magic {:#x}, expected {:#x}\n",
+ magic,
+ NV_CEUTILS_SEMA_PAGE_MAGIC,
+ );
+ return Err(EIO);
+ }
+
+ let start = Instant::<Monotonic>::now();
+ loop {
+ let value = semaphore_map.try_read32(SEMA_PAGE_PAYLOAD_OFFSET)?;
+ if value.wrapping_sub(work_id) < 0x8000_0000 {
+ dev_dbg!(
+ dev,
+ "scrub completed after {:?}: semaphore={:#x}, target={:#x}\n",
+ start.elapsed(),
+ value,
+ work_id,
+ );
+ return Ok(());
+ }
+
+ if start.elapsed() >= SCRUB_TIMEOUT {
+ dev_warn!(
+ dev,
+ "scrub timed out: semaphore={:#x}, target={:#x}\n",
+ value,
+ work_id,
+ );
+ return Err(ETIMEDOUT);
+ }
+ fsleep(Delta::from_millis(1));
+ }
+ })();
+
+ let cleanup = semaphore_map.dismiss().release(&mut mm.lock());
+ match result {
+ Ok(()) => cleanup,
+ Err(error) => {
+ if let Err(cleanup_error) = cleanup {
+ dev_err!(
+ dev,
+ "failed to release semaphore BAR1 mapping after error {:?}: {:?}\n",
+ error,
+ cleanup_error,
+ );
+ }
+ Err(error)
+ }
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 30/31] gpu: nova-core: vgpu: export plugin log buffers via debugfs
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (28 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 29/31] gpu: nova-core: vgpu: scrub guest VRAM with CeUtils Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
2026-09-28 10:28 ` [PATCH v3 31/31] gpu: nova-core: vgpu: introduce SR-IOV PF APIs Zhi Wang
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Expose each instance's init, vgpu and kernel plugin logs through debugfs
for nvlog_decoder.
Read the management-heap log buffers through BAR1 and reuse the existing
GSP log header when a firmware build ID is available. Use the INIT, VGPU
and KRNL task names expected by the decoder, and derive the instance
directory name from the typed DBDF fields.
Retain the GPU identity and firmware build ID in the manager for
per-instance log headers.
Retain the mapping while the files are accessible and revoke the debugfs
scope before explicitly unmapping it. Place the scope before the RPC
mapping in the instance's field order so device removal also drains
readers before the mapping is dropped. Preserve partial-read accounting
when a user copy fails.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 9 +-
drivers/gpu/nova-core/gsp.rs | 23 +++
drivers/gpu/nova-core/mm/bar_user.rs | 5 +-
drivers/gpu/nova-core/vgpu.rs | 11 +-
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 116 +++++++++++-
drivers/gpu/nova-core/vgpu/instance.rs | 60 +++++-
drivers/gpu/nova-core/vgpu/log.rs | 171 ++++++++++++++++++
7 files changed, 382 insertions(+), 13 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/log.rs
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index 012c7e6abca6..d5ac779e3b0b 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -558,9 +558,10 @@ pub(crate) fn new<'a>(
// SAFETY: These sibling fields are initialized at their final pinned
// addresses. The private manager cannot escape this `Gpu`, and is dropped
// before all its dependencies, both here on failure and on normal removal.
- let (cmdq, bar_user, mm, chid_pool) = unsafe {
+ // The GSP build ID is not modified after initialization.
+ let (gsp, bar_user, mm, chid_pool) = unsafe {
(
- &*core::ptr::from_ref(&gsp_resources.gsp.cmdq),
+ &*core::ptr::from_ref(&gsp_resources.gsp),
&*core::ptr::from_ref(bar_user.as_ref().get_ref()),
&*core::ptr::from_ref(mm.as_ref().get_ref()),
&*core::ptr::from_ref(chid_pool.as_ref().get_ref()),
@@ -568,7 +569,9 @@ pub(crate) fn new<'a>(
};
Some(KBox::pin_init(VgpuManager::new(
dev,
- cmdq,
+ &gsp.cmdq,
+ gsp_resources.spec,
+ gsp.build_id(),
bar_user,
mm,
chid_pool,
diff --git a/drivers/gpu/nova-core/gsp.rs b/drivers/gpu/nova-core/gsp.rs
index 45645bc7c9d8..ab3a7adbff69 100644
--- a/drivers/gpu/nova-core/gsp.rs
+++ b/drivers/gpu/nova-core/gsp.rs
@@ -180,6 +180,21 @@ fn new(spec: Spec, build_id: &BuildId, task_prefix: &str) -> Self {
}
}
+/// Size of the header prepended to debugfs log buffer dumps.
+pub(crate) const LOG_BUFFER_HEADER_SIZE: usize = size_of::<LogBufferHeader>();
+
+/// Builds a log header using the GPU implementation reported by the hardware.
+pub(crate) fn build_log_buffer_header(
+ spec: Spec,
+ build_id: &BuildId,
+ task_prefix: &str,
+) -> [u8; LOG_BUFFER_HEADER_SIZE] {
+ let header = LogBufferHeader::new(spec, build_id, task_prefix);
+ let mut bytes = [0; LOG_BUFFER_HEADER_SIZE];
+ bytes.copy_from_slice(header.as_bytes());
+ bytes
+}
+
/// The logging buffers are byte queues that contain encoded printf-like
/// messages from GSP-RM. They need to be decoded by a special application
/// that can parse the buffers.
@@ -368,6 +383,8 @@ fn register_debugfs<'data>(&'data self, dir: &debugfs::ScopedDir<'data, '_>) {
pub(crate) struct Gsp<'gsp> {
/// The GSP firmware's TLV.
gsp_tlv: firmware::Firmware,
+ /// Build identifier of the firmware whose log buffers are exposed.
+ build_id: Option<BuildId>,
/// Libos arguments.
pub(crate) libos: Coherent<'gsp, [LibosMemoryRegionInitArgument]>,
/// Log buffers, optionally exposed via debugfs.
@@ -383,6 +400,11 @@ pub(crate) struct Gsp<'gsp> {
}
impl<'gsp> Gsp<'gsp> {
+ /// Returns the GSP firmware build identifier, when available.
+ pub(crate) fn build_id(&self) -> Option<&BuildId> {
+ self.build_id.as_ref()
+ }
+
// Creates an in-place initializer for a `Gsp` manager for `pdev`.
pub(crate) fn new(
pdev: &'gsp pci::Device<device::Bound>,
@@ -405,6 +427,7 @@ pub(crate) fn new(
Ok(try_pin_init!(Self {
gsp_tlv,
+ build_id,
cmdq <- Cmdq::new(dev, bar),
rm_state_monitor: Coherent::zeroed(dev, GFP_KERNEL)?,
rmargs: Coherent::init(
diff --git a/drivers/gpu/nova-core/mm/bar_user.rs b/drivers/gpu/nova-core/mm/bar_user.rs
index bcbef1571fb9..72c65702c58b 100644
--- a/drivers/gpu/nova-core/mm/bar_user.rs
+++ b/drivers/gpu/nova-core/mm/bar_user.rs
@@ -197,7 +197,6 @@ pub(crate) struct BarMapping<'map, 'gpu> {
unmap_error: Option<Error>,
}
-#[expect(dead_code)]
impl<'map, 'gpu> BarMapping<'map, 'gpu> {
/// Maps the containing pages while restricting CPU access to the requested byte range.
pub(crate) fn new(
@@ -243,6 +242,10 @@ pub(crate) fn new(
})
}
+ pub(crate) fn bar1(&self) -> &'gpu Bar1<'gpu> {
+ self.access.bar_user.bar1
+ }
+
pub(crate) fn region(&self) -> &VramRegion {
&self.region
}
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index f78a4621984a..2dbf95222d4d 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -11,13 +11,15 @@
};
use crate::{
+ firmware::gsp::BuildId,
fsp::{
Fsp,
VgpuMode, //
},
gpu::{
ChannelIdPool,
- Chipset, //
+ Chipset,
+ Spec, //
},
gsp::{
cmdq::Cmdq,
@@ -38,6 +40,7 @@
mod gsp_plugin_rpc;
mod hal;
mod instance;
+mod log;
mod scrubber;
mod vram;
@@ -113,6 +116,8 @@ pub(crate) struct VgpuManager<'gpu> {
instances: Mutex<VgpuInstances<'gpu>>,
dev: &'gpu device::Device<device::Bound>,
cmdq: &'gpu Cmdq<'gpu>,
+ spec: Spec,
+ build_id: Option<&'gpu BuildId>,
bar_user: &'gpu BarUser<'gpu>,
mm: &'gpu Mutex<GpuMm<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
@@ -127,6 +132,8 @@ impl<'gpu> VgpuManager<'gpu> {
pub(crate) fn new(
dev: &'gpu device::Device<device::Bound>,
cmdq: &'gpu Cmdq<'gpu>,
+ spec: Spec,
+ build_id: Option<&'gpu BuildId>,
bar_user: &'gpu BarUser<'gpu>,
mm: &'gpu Mutex<GpuMm<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
@@ -139,6 +146,8 @@ pub(crate) fn new(
instances <- new_mutex!(VgpuInstances::new(), "nova-core::vgpu-instances"),
dev,
cmdq,
+ spec,
+ build_id,
bar_user,
mm,
chid_pool,
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index 3d2c69e8cd79..75739ddff58b 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -4,18 +4,22 @@
//! GSP plugin communication buffer mappings and access.
use kernel::{
+ io::Io,
num::casts::u32_as_usize,
prelude::*,
sync::Mutex, //
};
-use crate::mm::{
- bar_user::{
- BarMapping,
- BarUser, //
+use crate::{
+ driver::Bar1,
+ mm::{
+ bar_user::{
+ BarMapping,
+ BarUser, //
+ },
+ vram::VramRegion,
+ GpuMm, //
},
- vram::VramRegion,
- GpuMm, //
};
use super::fw::{
@@ -59,6 +63,97 @@ fn take_region(region: &VramRegion, cursor: &mut u64, size: u32) -> Result<VramR
Ok(subregion)
}
+/// BAR1 view of one vGPU plugin log buffer.
+pub(super) struct MappedPluginLogBuffer<'gpu> {
+ bar1: &'gpu Bar1<'gpu>,
+ gpu_va_addr: usize,
+ size: usize,
+}
+
+impl<'gpu> MappedPluginLogBuffer<'gpu> {
+ fn new(map: &BarMapping<'_, 'gpu>, region: &VramRegion) -> Result<Self> {
+ let start = region
+ .address()
+ .checked_sub(map.region().address())
+ .ok_or(EINVAL)
+ .and_then(|start| usize::try_from(start).map_err(|_| EOVERFLOW))?;
+ let size = usize::try_from(region.size()).map_err(|_| EOVERFLOW)?;
+ let end = start.checked_add(size).ok_or(EOVERFLOW)?;
+ if end > map.size() || !start.is_multiple_of(4) || !size.is_multiple_of(4) {
+ return Err(EINVAL);
+ }
+
+ let gpu_va_addr = usize::try_from(map.gpu_va_addr()?)
+ .map_err(|_| EOVERFLOW)?
+ .checked_add(start)
+ .ok_or(EOVERFLOW)?;
+ if !gpu_va_addr.is_multiple_of(4) {
+ return Err(EINVAL);
+ }
+
+ let bar1 = map.bar1();
+ if gpu_va_addr.checked_add(size).ok_or(EOVERFLOW)? > bar1.size() {
+ return Err(EINVAL);
+ }
+
+ Ok(Self {
+ bar1,
+ gpu_va_addr,
+ size,
+ })
+ }
+
+ pub(super) const fn size(&self) -> usize {
+ self.size
+ }
+
+ pub(super) fn read(&self, offset: usize, output: &mut [u8]) -> Result {
+ let end = offset.checked_add(output.len()).ok_or(EOVERFLOW)?;
+ if end > self.size {
+ return Err(EINVAL);
+ }
+
+ let mut source = offset;
+ let mut copied = 0usize;
+
+ while copied < output.len() {
+ let aligned_source = source & !3;
+ let within = source & 3;
+ let bar_offset = self
+ .gpu_va_addr
+ .checked_add(aligned_source)
+ .ok_or(EOVERFLOW)?;
+ let bytes = self.bar1.try_read32(bar_offset)?.to_le_bytes();
+ let chunk = (4 - within).min(output.len() - copied);
+
+ output[copied..copied + chunk].copy_from_slice(&bytes[within..within + chunk]);
+ source = source.checked_add(chunk).ok_or(EOVERFLOW)?;
+ copied += chunk;
+ }
+
+ Ok(())
+ }
+}
+
+/// BAR1 views of all vGPU plugin log buffers.
+pub(super) struct MappedPluginLogBuffers<'gpu> {
+ init: MappedPluginLogBuffer<'gpu>,
+ vgpu: MappedPluginLogBuffer<'gpu>,
+ kernel: MappedPluginLogBuffer<'gpu>,
+}
+
+impl<'gpu> MappedPluginLogBuffers<'gpu> {
+ pub(super) fn into_parts(
+ self,
+ ) -> (
+ MappedPluginLogBuffer<'gpu>,
+ MappedPluginLogBuffer<'gpu>,
+ MappedPluginLogBuffer<'gpu>,
+ ) {
+ (self.init, self.vgpu, self.kernel)
+ }
+}
+
/// BAR1 mapping of the plugin communication region in its management heap.
///
/// r000 layout, with byte offsets from the management heap (not to scale):
@@ -218,6 +313,15 @@ pub(super) fn plugin_logs(&self) -> PluginLogRegions {
}
}
+ /// Return BAR1 views that must stop being read before this mapping is destroyed.
+ pub(super) fn mapped_plugin_logs(&self) -> Result<MappedPluginLogBuffers<'gpu>> {
+ Ok(MappedPluginLogBuffers {
+ init: MappedPluginLogBuffer::new(&self.map, &self.init_log)?,
+ vgpu: MappedPluginLogBuffer::new(&self.map, &self.vgpu_log)?,
+ kernel: MappedPluginLogBuffer::new(&self.map, &self.kernel_log)?,
+ })
+ }
+
/// Clear a previous boot marker before starting the plugin.
pub(super) fn clear_plugin_ready(&self) -> Result {
let offset = self.io_offset(
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 1ccf4309f2ed..82a568ee3779 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -7,10 +7,12 @@
};
use kernel::{
+ debugfs,
device,
prelude::*,
ptr::Alignment,
sizes::SizeConstants,
+ str::CString,
time::{
delay::fsleep,
Delta,
@@ -21,7 +23,11 @@
use crate::{
driver::Bar0,
- gpu::ChannelIdReservation,
+ firmware::gsp::BuildId,
+ gpu::{
+ ChannelIdReservation,
+ Spec, //
+ },
gsp::{
cmdq::Cmdq,
commands::FifoEngineList, //
@@ -47,8 +53,12 @@
BootloadInfo,
ChannelMapEntry, //
},
- gsp_plugin_comm::CommBufferRegion,
+ gsp_plugin_comm::{
+ CommBufferRegion,
+ MappedPluginLogBuffers, //
+ },
gsp_plugin_rpc::PluginRpc,
+ log::VgpuLogBuffers,
scrubber::CeUtils,
vram::{
VgpuVramLayout,
@@ -150,6 +160,7 @@ struct VgpuInstance<'gpu> {
vgpu_type: VgpuType,
vm_pid: u32,
num_plugin_channels: u32,
+ debugfs_logs: Option<Pin<KBox<debugfs::Scope<VgpuLogBuffers<'gpu>>>>>,
plugin_rpc: PluginRpc<'gpu, 'gpu>,
// Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
@@ -197,6 +208,20 @@ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
self.configure_plugin(dev)?;
set_plugin_bme(dev, &mut self.plugin_rpc, true)?;
+ match self
+ .plugin_rpc
+ .comm()
+ .mapped_plugin_logs()
+ .and_then(|buffers| create_debugfs_logs(buffers, self.dbdf, vgpu.spec, vgpu.build_id))
+ {
+ Ok(logs) => self.debugfs_logs = Some(logs),
+ Err(error) => dev_warn!(
+ dev,
+ "debugfs logs unavailable for gfid={}: {:?}\n",
+ self.gfid.get(),
+ error,
+ ),
+ }
Ok(())
}
@@ -222,6 +247,8 @@ fn teardown(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
send_cleanup(vgpu.dev, vgpu.cmdq, self.gfid)?;
self.needs_teardown = false;
}
+ // Debugfs readers use BAR1 offsets directly and must finish before unmapping.
+ self.debugfs_logs = None;
self.plugin_rpc.unmap()
})();
if let Err(error) = result {
@@ -320,6 +347,34 @@ pub(super) const fn new(gfid: Gfid, dbdf: Dbdf, vgpu_type: VgpuType, vm_pid: u32
}
}
+fn create_debugfs_logs<'gpu>(
+ buffers: MappedPluginLogBuffers<'gpu>,
+ dbdf: Dbdf,
+ spec: Spec,
+ build_id: Option<&BuildId>,
+) -> Result<Pin<KBox<debugfs::Scope<VgpuLogBuffers<'gpu>>>>> {
+ let logs = VgpuLogBuffers::new(buffers, spec, build_id);
+ let directory = CString::try_from_fmt(fmt!(
+ "{:04x}:{:02x}:{:02x}.{:x}-vgpu",
+ dbdf.domain(),
+ dbdf.bus(),
+ dbdf.device(),
+ dbdf.function(),
+ ))?;
+
+ #[allow(static_mut_refs)]
+ // SAFETY: The root is initialized before driver registration and cleared
+ // only after driver unregistration has drained all users.
+ let root = unsafe { crate::DEBUGFS_ROOT.as_ref() }.ok_or(ENODEV)?;
+
+ KBox::pin_init(
+ root.scope(logs, &directory, |logs, directory| {
+ VgpuLogBuffers::register_debugfs(logs, directory);
+ }),
+ GFP_KERNEL,
+ )
+}
+
/// Registry of live vGPU instances.
pub(super) struct VgpuInstances<'gpu> {
instances: KVec<VgpuInstance<'gpu>>,
@@ -420,6 +475,7 @@ fn allocate_instance<'a>(
vgpu_type,
vm_pid,
num_plugin_channels: PLUGIN_CHANNELS_PER_ENGINE,
+ debugfs_logs: None,
plugin_rpc: PluginRpc::new(comm, bar0, gfid),
vram_slot,
chids,
diff --git a/drivers/gpu/nova-core/vgpu/log.rs b/drivers/gpu/nova-core/vgpu/log.rs
new file mode 100644
index 000000000000..30a33bd6723e
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/log.rs
@@ -0,0 +1,171 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! GSP plugin logs exposed through debugfs.
+//!
+//! With debugfs mounted at `/sys/kernel/debug`, the directory uses the VF's
+//! PCI domain:bus:device.function address:
+//!
+//! ```text
+//! /sys/kernel/debug/nova-core/<VF-DBDF>-vgpu/
+//! |-- init_log
+//! |-- vgpu_log
+//! `-- kernel_log
+//! ```
+
+use kernel::{
+ debugfs,
+ fs::file,
+ prelude::*,
+ uaccess::UserSliceWriter, //
+};
+
+use crate::{
+ firmware::gsp::BuildId,
+ gpu::Spec,
+ gsp::{
+ build_log_buffer_header,
+ LOG_BUFFER_HEADER_SIZE, //
+ },
+ vgpu::gsp_plugin_comm::{
+ MappedPluginLogBuffer,
+ MappedPluginLogBuffers, //
+ },
+};
+
+const LOG_READ_CHUNK_SIZE: usize = 4096;
+
+/// A vGPU plugin log buffer backed by VRAM, read via BAR1 MMIO.
+///
+/// An optional header lets `nvlog_decoder` identify the GPU architecture and firmware build.
+struct VgpuLogBuffer<'gpu> {
+ buffer: MappedPluginLogBuffer<'gpu>,
+ header: [u8; LOG_BUFFER_HEADER_SIZE],
+ header_len: usize,
+}
+
+impl<'gpu> VgpuLogBuffer<'gpu> {
+ fn new(
+ buffer: MappedPluginLogBuffer<'gpu>,
+ spec: Spec,
+ build_id: Option<&BuildId>,
+ task_prefix: &str,
+ ) -> Self {
+ let (header, header_len) = match build_id {
+ Some(bid) => (
+ build_log_buffer_header(spec, bid, task_prefix),
+ LOG_BUFFER_HEADER_SIZE,
+ ),
+ None => ([0u8; LOG_BUFFER_HEADER_SIZE], 0),
+ };
+
+ Self {
+ buffer,
+ header,
+ header_len,
+ }
+ }
+}
+
+impl debugfs::BinaryWriter for VgpuLogBuffer<'_> {
+ fn write_to_slice(
+ &self,
+ writer: &mut UserSliceWriter,
+ offset: &mut file::Offset,
+ ) -> Result<usize> {
+ if offset.is_negative() {
+ return Err(EINVAL);
+ }
+
+ let offset_val: usize = (*offset).try_into().map_err(|_| EINVAL)?;
+ let total_len = self
+ .header_len
+ .checked_add(self.buffer.size())
+ .ok_or(EOVERFLOW)?;
+
+ if offset_val >= total_len {
+ return Ok(0);
+ }
+
+ let count = (total_len - offset_val).min(writer.len());
+ if count == 0 {
+ return Ok(0);
+ }
+
+ // Keep the staging buffer on the heap to avoid a page-sized kernel stack object.
+ let staging_size = count.min(LOG_READ_CHUNK_SIZE);
+ let mut staging = KVec::new();
+ staging.resize(staging_size, 0, GFP_KERNEL)?;
+
+ let mut written = 0usize;
+ let result: Result = (|| {
+ while written < count {
+ let chunk_len = (count - written).min(staging.len());
+ let chunk = &mut staging[..chunk_len];
+ let chunk_offset = offset_val.checked_add(written).ok_or(EOVERFLOW)?;
+ let mut filled = 0usize;
+
+ if chunk_offset < self.header_len {
+ let header_len = (self.header_len - chunk_offset).min(chunk_len);
+ chunk[..header_len]
+ .copy_from_slice(&self.header[chunk_offset..chunk_offset + header_len]);
+ filled = header_len;
+ }
+
+ if filled < chunk_len {
+ let log_offset = chunk_offset
+ .checked_add(filled)
+ .ok_or(EOVERFLOW)?
+ .checked_sub(self.header_len)
+ .ok_or(EINVAL)?;
+
+ self.buffer.read(log_offset, &mut chunk[filled..])?;
+ }
+
+ writer.write_slice(chunk)?;
+ written = written.checked_add(chunk_len).ok_or(EOVERFLOW)?;
+ }
+ Ok(())
+ })();
+ if written == 0 {
+ result?;
+ }
+
+ *offset = (*offset)
+ .checked_add(i64::try_from(written).map_err(|_| EOVERFLOW)?)
+ .ok_or(EOVERFLOW)?;
+ Ok(written)
+ }
+}
+
+/// The three plugin log streams for one vGPU instance.
+pub(super) struct VgpuLogBuffers<'gpu> {
+ init_log: VgpuLogBuffer<'gpu>,
+ vgpu_log: VgpuLogBuffer<'gpu>,
+ kernel_log: VgpuLogBuffer<'gpu>,
+}
+
+impl<'gpu> VgpuLogBuffers<'gpu> {
+ pub(super) fn new(
+ buffers: MappedPluginLogBuffers<'gpu>,
+ spec: Spec,
+ build_id: Option<&BuildId>,
+ ) -> Self {
+ let (init, vgpu, kernel) = buffers.into_parts();
+
+ Self {
+ init_log: VgpuLogBuffer::new(init, spec, build_id, "INIT"),
+ vgpu_log: VgpuLogBuffer::new(vgpu, spec, build_id, "VGPU"),
+ kernel_log: VgpuLogBuffer::new(kernel, spec, build_id, "KRNL"),
+ }
+ }
+
+ pub(super) fn register_debugfs<'data, 'dir>(
+ logs: &'data Self,
+ dir: &'dir debugfs::ScopedDir<'data, 'dir>,
+ ) {
+ dir.read_binary_file(c"init_log", &logs.init_log);
+ dir.read_binary_file(c"vgpu_log", &logs.vgpu_log);
+ dir.read_binary_file(c"kernel_log", &logs.kernel_log);
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread* [PATCH v3 31/31] gpu: nova-core: vgpu: introduce SR-IOV PF APIs
2026-09-28 10:28 [PATCH v3 00/31] Introduce NVIDIA vGPU manager Zhi Wang
` (29 preceding siblings ...)
2026-09-28 10:28 ` [PATCH v3 30/31] gpu: nova-core: vgpu: export plugin log buffers via debugfs Zhi Wang
@ 2026-09-28 10:28 ` Zhi Wang
30 siblings, 0 replies; 32+ messages in thread
From: Zhi Wang @ 2026-09-28 10:28 UTC (permalink / raw)
To: dakr, acourbot
Cc: alex, jgg, yishaih, skolothumtho, kevin.tian, airlied, simona,
ojeda, alex.gaynor, boqun.feng, gary, bjorn3_gh, lossin,
a.hindborg, aliceryhl, tmgross, jhubbard, ecourtney, cjia,
smitra, kjaju, alkumar, ankita, aniketa, kwankhede, targupta,
nova-gpu, linux-kernel, zhiwang, Zhi Wang
Provide PF operations for querying vGPU availability and creating,
closing and resetting VF instances.
Validate VF identifiers against the PF's VF count and route lifecycle
operations through the manager. Retain the GPU context needed to create
plugin RPC channels and return guest PCI properties after activation.
Signed-off-by: Zhi Wang <zhiw@nvidia.com>
---
drivers/gpu/nova-core/gpu.rs | 26 +++-
drivers/gpu/nova-core/nova_core.rs | 6 +
drivers/gpu/nova-core/vgpu.rs | 2 +
drivers/gpu/nova-core/vgpu/commands.rs | 10 +-
drivers/gpu/nova-core/vgpu/fw/commands.rs | 1 +
drivers/gpu/nova-core/vgpu/instance.rs | 93 +++++++++++--
drivers/gpu/nova-core/vgpu/vgpu_api.rs | 158 ++++++++++++++++++++++
7 files changed, 281 insertions(+), 15 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/vgpu_api.rs
diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index d5ac779e3b0b..df9037d8965c 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -1,6 +1,9 @@
// SPDX-License-Identifier: GPL-2.0
-use core::ops::Range;
+use core::{
+ num::NonZero,
+ ops::Range, //
+};
use kernel::{
device,
@@ -35,6 +38,7 @@
fsp::Fsp,
gsp::{
self,
+ cmdq::Cmdq,
Gsp,
GspBootContext, //
},
@@ -397,6 +401,25 @@ fn drop(self: Pin<&mut Self>) {
}
impl<'gpu> Gpu<'gpu> {
+ pub(crate) fn cmdq(&self) -> &Cmdq<'gpu> {
+ &self.gsp_resources.gsp.cmdq
+ }
+
+ pub(crate) fn vgpu_manager(&self) -> Option<&VgpuManager<'gpu>> {
+ self.vgpu.as_ref().map(|vgpu| vgpu.as_ref().get_ref())
+ }
+
+ pub(crate) fn vgpu_total_vfs(&self) -> Option<NonZero<u16>> {
+ match self.gsp_resources.vgpu_state {
+ VgpuState::Disabled => None,
+ VgpuState::Enabled { total_vfs } => Some(total_vfs),
+ }
+ }
+
+ pub(crate) fn bar0(&self) -> Bar0<'gpu> {
+ self.gsp_resources.bar
+ }
+
pub(crate) fn new<'a>(
pdev: &'gpu pci::Device<device::Core<'a>>,
bar: Bar0<'gpu>,
@@ -468,7 +491,6 @@ pub(crate) fn new<'a>(
}),
-
// GSP boot left the SWGEN0 latch set and pending bits in the tree.
_: {
irq::gsp::quiesce(bar, gsp_resources.spec.chipset, vectors_ref)?;
diff --git a/drivers/gpu/nova-core/nova_core.rs b/drivers/gpu/nova-core/nova_core.rs
index cbaef6d3d9f1..4e074e014f8b 100644
--- a/drivers/gpu/nova-core/nova_core.rs
+++ b/drivers/gpu/nova-core/nova_core.rs
@@ -29,6 +29,12 @@
mod vbios;
mod vgpu;
+#[cfg(CONFIG_PCI_IOV)]
+pub use vgpu::vgpu_api::{
+ NovaCoreVfApi,
+ VgpuTypeInfo, //
+};
+
pub(crate) const MODULE_NAME: &core::ffi::CStr = <LocalModule as kernel::ModuleMetadata>::NAME;
// TODO: Move this into per-module data once that exists.
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 2dbf95222d4d..47a27236a086 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -42,6 +42,8 @@
mod instance;
mod log;
mod scrubber;
+#[cfg_attr(not(CONFIG_PCI_IOV), expect(dead_code, unreachable_pub))]
+pub(crate) mod vgpu_api;
mod vram;
/// vGPU state detected during GPU construction.
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 52e8339504c6..ba8848211e44 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -77,7 +77,6 @@ fn check_status(dev: &device::Device<device::Bound>, command_id: u32, status: u3
}
/// Query the vGPU type assigned to a VF by its DBDF.
-#[expect(dead_code)]
pub(super) fn query_assigned_vf_type(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
@@ -94,7 +93,6 @@ pub(super) fn query_assigned_vf_type(
}
/// Query and decode the firmware properties of one vGPU type.
-#[expect(dead_code)]
pub(super) fn query_vgpu_properties(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
@@ -206,6 +204,14 @@ pub(super) fn set_plugin_bme(
rpc.rpc_call_nvkv(dev, RpcMessage::UpdateBmeState, &bme)
}
+/// Reset an active GSP plugin.
+pub(super) fn reset_plugin(
+ dev: &device::Device<device::Bound>,
+ rpc: &mut PluginRpc<'_, '_>,
+) -> Result {
+ rpc.rpc_call(dev, RpcMessage::Reset, &[])
+}
+
/// Whether a failed allocation may still have transferred CHID ownership to firmware.
pub(super) enum CeUtilsAllocError {
/// A matching firmware response explicitly rejected the allocation.
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index ad2b29340986..3738beea0463 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -37,6 +37,7 @@
pub(crate) enum RpcMessage {
VersionNegotiation = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_VERSION_NEGOTIATION,
SetupConfigParamsAndInit = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_SETUP_CONFIG_PARAMS_AND_INIT,
+ Reset = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_RESET,
UpdateBmeState = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_UPDATE_BME_STATE,
}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 82a568ee3779..c887ccd7db7e 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -39,13 +39,16 @@
commands::{
free_ceutils,
negotiate_plugin_version,
+ query_vgpu_properties,
+ reset_plugin,
send_bootload,
send_cleanup,
send_plugin_config,
send_shutdown,
set_plugin_bme,
CeUtilsAllocError,
- Dbdf, //
+ Dbdf,
+ VgpuProperties, //
},
fw::commands::{
encode_plugin_config_params,
@@ -124,7 +127,6 @@ fn wait_plugin_ready(
impl Gfid {
/// Validates an external GFID for a device supporting `total_vfs` VFs.
- #[expect(dead_code)]
pub(super) fn new(gfid: u32, total_vfs: NonZero<u16>) -> Result<Self> {
let gfid = u16::try_from(gfid).map_err(|_| EINVAL)?;
let gfid = NonZero::new(gfid).ok_or(EINVAL)?;
@@ -142,17 +144,30 @@ pub(super) const fn get(self) -> u16 {
}
/// Resource requirements and device identity for one vGPU type.
-#[expect(dead_code)]
pub(super) struct VgpuType {
- vgpu_type_id: u32,
- bar1_length: u64,
+ pub(super) vgpu_type_id: u32,
+ pub(super) bar1_length: u64,
max_instance: u32,
- pci_dev_id: u32,
- pci_subsys_id: u32,
- fb_length: u64,
+ pub(super) pci_dev_id: u32,
+ pub(super) pci_subsys_id: u32,
+ pub(super) fb_length: u64,
gsp_heap_size: u64,
}
+impl VgpuType {
+ fn from_properties(properties: &VgpuProperties) -> Self {
+ Self {
+ vgpu_type_id: properties.type_id,
+ bar1_length: properties.bar1_length,
+ max_instance: properties.max_instance,
+ pci_dev_id: properties.dev_id,
+ pci_subsys_id: properties.subsystem_id,
+ fb_length: properties.fb_length,
+ gsp_heap_size: properties.gsp_heap_size,
+ }
+ }
+}
+
/// A vGPU instance and the resources reserved for it.
struct VgpuInstance<'gpu> {
gfid: Gfid,
@@ -166,6 +181,7 @@ struct VgpuInstance<'gpu> {
vram_slot: VgpuVramSlot,
chids: ChannelIdReservation<'gpu>,
ceutils: Option<CeUtils>,
+ active: bool,
needs_teardown: bool,
/// An uncertain or failed operation retains resources until device removal.
failure: Option<Error>,
@@ -222,6 +238,7 @@ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
error,
),
}
+ self.active = true;
Ok(())
}
@@ -323,6 +340,7 @@ fn shutdown(&mut self, dev: &device::Device<device::Bound>, cmdq: &Cmdq<'_>) ->
send_shutdown(dev, cmdq, self.gfid)?;
dev_dbg!(dev, "shutdown: gfid={} stopped\n", self.gfid.get());
}
+ self.active = false;
Ok(())
}
}
@@ -335,7 +353,6 @@ pub(super) struct InstanceInfo {
vm_pid: u32,
}
-#[expect(dead_code)]
impl InstanceInfo {
pub(super) const fn new(gfid: Gfid, dbdf: Dbdf, vgpu_type: VgpuType, vm_pid: u32) -> Self {
Self {
@@ -381,7 +398,6 @@ pub(super) struct VgpuInstances<'gpu> {
vram_slots: Option<VgpuVramSlotAllocator>,
}
-#[expect(dead_code)]
impl<'gpu> VgpuInstances<'gpu> {
pub(super) const fn new() -> Self {
Self {
@@ -480,6 +496,7 @@ fn allocate_instance<'a>(
vram_slot,
chids,
ceutils: None,
+ active: false,
needs_teardown: false,
failure: None,
};
@@ -538,7 +555,6 @@ struct PendingInstance<'a, 'gpu> {
committed: bool,
}
-#[expect(dead_code)]
impl PendingInstance<'_, '_> {
fn activate(mut self) -> Result {
let instance = self
@@ -569,3 +585,58 @@ fn drop(&mut self) {
}
}
}
+
+/// Query and decode one vGPU type using the typed NVKV schema.
+pub(super) fn query_vgpu_type(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ type_id: u32,
+) -> Result<VgpuType> {
+ let properties = query_vgpu_properties(dev, cmdq, type_id)?;
+ Ok(VgpuType::from_properties(&properties))
+}
+
+impl<'gpu> VgpuManager<'gpu> {
+ /// Allocate, register, and activate a vGPU instance.
+ ///
+ /// Keep the registry locked through creation or rollback so duplicate checks and
+ /// type limits remain stable. MM and BAR-user locks are acquired only within
+ /// individual memory operations, in that order.
+ pub(super) fn create_instance(&self, bar: Bar0<'gpu>, info: InstanceInfo) -> Result {
+ let mut instances = self.instances.lock();
+ instances.allocate_instance(self, bar, info)?.activate()?;
+ Ok(())
+ }
+
+ pub(super) fn close_instance(&self, gfid: Gfid) -> Result {
+ self.instances.lock().destroy_instance(self, gfid)
+ }
+
+ pub(super) fn reset_instance(&self, gfid: Gfid) -> Result {
+ let mut instances = self.instances.lock();
+ let instance = instances
+ .instances
+ .iter_mut()
+ .find(|instance| instance.gfid == gfid)
+ .ok_or(ENOENT)?;
+ if instance.failure.is_some() || !instance.active {
+ return Err(EBUSY);
+ }
+
+ let result = (|| {
+ reset_plugin(self.dev, &mut instance.plugin_rpc)?;
+ instance.ceutils.as_ref().ok_or(EINVAL)?.scrub_guest_fb(
+ self.dev,
+ self.cmdq,
+ self.bar_user,
+ self.mm,
+ &instance.vram_slot.fbmem,
+ )
+ })();
+ if let Err(error) = result {
+ instance.failure = Some(error);
+ instance.active = false;
+ }
+ result
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/vgpu_api.rs b/drivers/gpu/nova-core/vgpu/vgpu_api.rs
new file mode 100644
index 000000000000..f9f3e527bdc0
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/vgpu_api.rs
@@ -0,0 +1,158 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! PF-owned lifecycle operations for Nova VF drivers.
+
+use core::num::NonZero;
+
+use kernel::{
+ device,
+ pci,
+ prelude::*, //
+};
+
+use crate::{
+ driver::Bar0,
+ gpu::Gpu,
+ gsp::cmdq::Cmdq,
+ vgpu::instance::{
+ query_vgpu_type,
+ Gfid,
+ InstanceInfo,
+ VgpuType, //
+ },
+};
+
+use super::{
+ commands::{
+ query_assigned_vf_type,
+ Dbdf, //
+ },
+ VgpuManager, //
+};
+
+/// Guest PCI properties of an activated vGPU instance.
+pub struct VgpuTypeInfo {
+ /// PCI device ID to present to the guest.
+ pub pci_dev_id: u32,
+ /// PCI subsystem ID to present to the guest.
+ pub pci_subsys_id: u32,
+ /// BAR1 aperture size in MiB.
+ pub bar1_length: u64,
+}
+
+impl VgpuTypeInfo {
+ fn from_vgpu_type(vgpu_type: &VgpuType) -> Self {
+ Self {
+ pci_dev_id: vgpu_type.pci_dev_id,
+ pci_subsys_id: vgpu_type.pci_subsys_id,
+ bar1_length: vgpu_type.bar1_length,
+ }
+ }
+}
+
+/// PF-owned operations available while a VF driver is bound.
+pub struct NovaCoreVfApi<'gpu> {
+ pdev: &'gpu pci::Device<device::Bound>,
+ cmdq: &'gpu Cmdq<'gpu>,
+ bar: Bar0<'gpu>,
+ vgpu: Option<&'gpu VgpuManager<'gpu>>,
+ total_vfs: Option<NonZero<u16>>,
+}
+
+impl<'gpu> NovaCoreVfApi<'gpu> {
+ #[expect(dead_code)]
+ pub(crate) fn new(gpu: &'gpu Gpu<'gpu>, pdev: &'gpu pci::Device<device::Bound>) -> Self {
+ Self {
+ pdev,
+ cmdq: gpu.cmdq(),
+ bar: gpu.bar0(),
+ vgpu: gpu.vgpu_manager(),
+ total_vfs: gpu.vgpu_total_vfs(),
+ }
+ }
+
+ /// Returns whether the PF booted with vGPU support enabled.
+ pub fn is_available(&self) -> bool {
+ self.vgpu.is_some()
+ }
+
+ fn gfid(&self, gfid: u32) -> Result<Gfid> {
+ let total_vfs = self.total_vfs.ok_or(ENODEV)?;
+ Gfid::new(gfid, total_vfs)
+ }
+}
+
+impl NovaCoreVfApi<'_> {
+ /// Creates and boots an instance for a one-based VF ID.
+ ///
+ /// `sbdf` encodes the VF address as `(segment << 16) | (bus << 8) | devfn`.
+ /// `vm_pid` identifies the VM process's thread group. This call may sleep.
+ pub fn open_instance(&self, gfid: u32, sbdf: u32, vm_pid: u32) -> Result<VgpuTypeInfo> {
+ let dev = self.pdev.as_ref();
+ let gfid = self.gfid(gfid)?;
+ let dbdf = Dbdf::from_raw(sbdf);
+
+ dev_dbg!(
+ dev,
+ "vgpu_open: gfid={} sbdf={:#x}\n",
+ gfid.get(),
+ dbdf.into_raw()
+ );
+
+ let bar = self.bar;
+ let cmdq = self.cmdq;
+ let vgpu = self.vgpu.ok_or(ENODEV)?;
+
+ let type_id = query_assigned_vf_type(dev, cmdq, dbdf)?;
+ dev_dbg!(
+ dev,
+ "vgpu_open: gfid={} assigned type_id={}\n",
+ gfid.get(),
+ type_id
+ );
+
+ let vgpu_type = query_vgpu_type(dev, cmdq, type_id)?;
+ dev_dbg!(
+ dev,
+ "vgpu_open: gfid={} vgpu_type={} fb_length={:#x}\n",
+ gfid.get(),
+ vgpu_type.vgpu_type_id,
+ vgpu_type.fb_length
+ );
+
+ let type_info = VgpuTypeInfo::from_vgpu_type(&vgpu_type);
+ vgpu.create_instance(bar, InstanceInfo::new(gfid, dbdf, vgpu_type, vm_pid))?;
+
+ Ok(type_info)
+ }
+
+ /// Tears down an instance, retaining resources if firmware cleanup fails.
+ ///
+ /// This call may sleep.
+ pub fn close_instance(&self, gfid: u32) -> Result {
+ let dev = self.pdev.as_ref();
+ let gfid = self.gfid(gfid)?;
+
+ dev_dbg!(dev, "vgpu_close: gfid={}\n", gfid.get());
+
+ let result = self.vgpu.ok_or(ENODEV)?.close_instance(gfid);
+ if let Err(error) = result {
+ dev_err!(dev, "vgpu_close: gfid={} failed: {:?}\n", gfid.get(), error);
+ }
+ result
+ }
+
+ /// Resets an instance and scrubs its guest VRAM. This call may sleep.
+ pub fn reset_instance(&self, gfid: u32) -> Result {
+ let dev = self.pdev.as_ref();
+ let gfid = self.gfid(gfid)?;
+
+ dev_dbg!(dev, "vgpu_reset: gfid={}\n", gfid.get());
+
+ self.vgpu.ok_or(ENODEV)?.reset_instance(gfid)?;
+
+ dev_dbg!(dev, "vgpu_reset: gfid={} done\n", gfid.get());
+ Ok(())
+ }
+}
^ permalink raw reply [flat|nested] 32+ messages in thread