[PATCH v3 21/31] gpu: nova-core: vgpu: add instance boot

From: Zhi Wang

Date: Mon Sep 28 2026 - 06:37:17 EST


Start an allocated instance's GSP plugin with its channel map, VRAM,
plugin heap and log regions. Clear the old ready marker before BOOTLOAD
and wait for firmware to publish a new one.

Retain the command queue and memory dependencies needed by the instance,
and map its communication buffers before boot.

Co-developed-by: Alok Kumar <alkumar@xxxxxxxxxx>
Signed-off-by: Alok Kumar <alkumar@xxxxxxxxxx>
Signed-off-by: Zhi Wang <zhiw@xxxxxxxxxx>
---
drivers/gpu/nova-core/gpu.rs | 71 +++++---
drivers/gpu/nova-core/gsp/fw/commands.rs | 1 -
drivers/gpu/nova-core/vgpu.rs | 30 +++-
drivers/gpu/nova-core/vgpu/commands.rs | 21 ++-
drivers/gpu/nova-core/vgpu/fw.rs | 3 +
drivers/gpu/nova-core/vgpu/fw/commands.rs | 2 -
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 16 +-
drivers/gpu/nova-core/vgpu/instance.rs | 152 ++++++++++++++++--
drivers/gpu/nova-core/vgpu/vram.rs | 1 -
9 files changed, 245 insertions(+), 52 deletions(-)

diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index b41a9921381d..012c7e6abca6 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -8,6 +8,7 @@
fmt,
gpu::buddy::GpuBuddyParams,
io::Io,
+ new_mutex,
num::Bounded,
pci,
prelude::*,
@@ -16,6 +17,7 @@
SizeConstants,
SZ_4K, //
},
+ sync::Mutex,
};

use crate::{
@@ -327,6 +329,7 @@ fn static_info(&self) -> &gsp::commands::GspStaticInfo {
#[pin_data]
pub(crate) struct Gpu<'gpu> {
spec: Spec,
+ /// Drops before the MM, BAR1 mappings and GSP needed for instance teardown.
vgpu: Option<Pin<KBox<VgpuManager<'gpu>>>>,
/// GSP event interrupt registration.
///
@@ -339,7 +342,8 @@ pub(crate) struct Gpu<'gpu> {
///
/// Must be kept declared *before* `gsp_resources`, so that its components are dropped while
/// the GSP is still operational.
- mm: GpuMm<'gpu>,
+ #[pin]
+ mm: Mutex<GpuMm<'gpu>>,
/// BAR1 user interface for CPU access to GPU virtual memory.
#[pin]
bar_user: BarUser<'gpu>,
@@ -463,22 +467,7 @@ pub(crate) fn new<'a>(
})?,
}),

- vgpu: {
- let info = &gsp_resources.boot_result.static_info;
- match gsp_resources.vgpu_state {
- VgpuState::Disabled => None,
- VgpuState::Enabled { .. } => Some(KBox::pin_init(VgpuManager::new(
- // SAFETY: `chid_pool` is initialized above at its final pinned address.
- // The private manager and its pool borrow cannot escape this `Gpu`.
- // Completed field drop order drops the manager before the pool; on failure,
- // pin-init drops it before the earlier-initialized pool.
- unsafe { &*core::ptr::from_ref(chid_pool.as_ref().get_ref()) },
- &info.fifo_engine_list(),
- info.vmmu_segment_size,
- TOTAL_CHANNELS,
- ), GFP_KERNEL)?),
- }
- },
+

// GSP boot left the SWGEN0 latch set and pending bits in the tree.
_: {
@@ -529,7 +518,7 @@ pub(crate) fn new<'a>(
},

// Create GPU memory manager owning memory management resources.
- mm: {
+ mm <- {
let info = gsp_resources.static_info();
let usable_vram = info.usable_fb_regions().next().ok_or(ENODEV)?;
let buddy_params = GpuBuddyParams {
@@ -538,12 +527,15 @@ pub(crate) fn new<'a>(
chunk_size: Alignment::new::<SZ_4K>(),
};

- GpuMm::new(
- bar,
- gsp_resources.spec.chipset,
- buddy_params,
- VramAddress::from_raw(info.total_fb_end().ok_or(ENODEV)?),
- )?
+ new_mutex!(
+ GpuMm::new(
+ bar,
+ gsp_resources.spec.chipset,
+ buddy_params,
+ VramAddress::from_raw(info.total_fb_end().ok_or(ENODEV)?),
+ )?,
+ "nova-core::gpu-mm",
+ )
},

// Create BAR1 user interface for CPU access to GPU virtual memory.
@@ -558,6 +550,34 @@ pub(crate) fn new<'a>(
bar1,
)?
},
+
+ vgpu: {
+ match gsp_resources.vgpu_state {
+ VgpuState::Disabled => None,
+ VgpuState::Enabled { .. } => {
+ // SAFETY: These sibling fields are initialized at their final pinned
+ // addresses. The private manager cannot escape this `Gpu`, and is dropped
+ // before all its dependencies, both here on failure and on normal removal.
+ let (cmdq, bar_user, mm, chid_pool) = unsafe {
+ (
+ &*core::ptr::from_ref(&gsp_resources.gsp.cmdq),
+ &*core::ptr::from_ref(bar_user.as_ref().get_ref()),
+ &*core::ptr::from_ref(mm.as_ref().get_ref()),
+ &*core::ptr::from_ref(chid_pool.as_ref().get_ref()),
+ )
+ };
+ Some(KBox::pin_init(VgpuManager::new(
+ dev,
+ cmdq,
+ bar_user,
+ mm,
+ chid_pool,
+ gsp_resources.static_info(),
+ TOTAL_CHANNELS,
+ ), GFP_KERNEL)?)
+ }
+ }
+ },
})
}

@@ -567,10 +587,11 @@ pub(crate) fn run_selftests(self: Pin<&mut Self>, pdev: &pci::Device<device::Bou
let this = self.project();
let dev = pdev.as_ref();
let info = this.gsp_resources.static_info();
+ let mut mm = this.mm.lock();

if let Err(err) = crate::mm::selftest::run(
dev,
- this.mm,
+ &mut mm,
info.usable_fb_regions(),
this.bar_user.as_ref().get_ref(),
info.bar1_pde_base(),
diff --git a/drivers/gpu/nova-core/gsp/fw/commands.rs b/drivers/gpu/nova-core/gsp/fw/commands.rs
index 7381b30f3680..b1087d950b30 100644
--- a/drivers/gpu/nova-core/gsp/fw/commands.rs
+++ b/drivers/gpu/nova-core/gsp/fw/commands.rs
@@ -335,7 +335,6 @@ pub(crate) struct FifoEngineList {
}

impl FifoEngineList {
- #[expect(dead_code)]
pub(crate) fn gmc_ids(&self) -> &[u32] {
// PANIC: The type invariant bounds `count` by the array capacity.
&self.gmc_ids[..self.count]
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index 024cb13694f4..84c4d0b3839d 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -19,7 +19,17 @@
ChannelIdPool,
Chipset, //
},
- gsp::commands::FifoEngineList, //
+ gsp::{
+ cmdq::Cmdq,
+ commands::{
+ FifoEngineList,
+ GspStaticInfo, //
+ }, //
+ },
+ mm::{
+ bar_user::BarUser,
+ GpuMm, //
+ }, //
};

mod commands;
@@ -99,6 +109,10 @@ fn query_state(
pub(crate) struct VgpuManager<'gpu> {
#[pin]
instances: Mutex<VgpuInstances<'gpu>>,
+ dev: &'gpu device::Device<device::Bound>,
+ cmdq: &'gpu Cmdq<'gpu>,
+ bar_user: &'gpu BarUser<'gpu>,
+ mm: &'gpu Mutex<GpuMm<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
/// VMMU segment size in bytes, or zero if GSP-RM omitted it.
vmmu_segment_size: u64,
@@ -109,14 +123,22 @@ pub(crate) struct VgpuManager<'gpu> {
impl<'gpu> VgpuManager<'gpu> {
/// Retains runtime parameters from a completed vGPU-enabled GSP boot.
pub(crate) fn new(
+ dev: &'gpu device::Device<device::Bound>,
+ cmdq: &'gpu Cmdq<'gpu>,
+ bar_user: &'gpu BarUser<'gpu>,
+ mm: &'gpu Mutex<GpuMm<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
- fifo_engine_list: &FifoEngineList,
- vmmu_segment_size: u64,
+ info: &GspStaticInfo,
total_channels: u32,
) -> impl PinInit<Self> + use<'gpu> {
- let fifo_engine_list = *fifo_engine_list;
+ let fifo_engine_list = info.fifo_engine_list();
+ let vmmu_segment_size = info.vmmu_segment_size;
pin_init!(Self {
instances <- new_mutex!(VgpuInstances::new(), "nova-core::vgpu-instances"),
+ dev,
+ cmdq,
+ bar_user,
+ mm,
chid_pool,
vmmu_segment_size,
total_channels,
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 679c22b0381a..2e9c250b65ab 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -8,7 +8,9 @@

use kernel::{
device,
- prelude::*, //
+ prelude::*,
+ time::Delta,
+ transmute::AsBytes, //
};

use crate::gsp::{
@@ -25,6 +27,7 @@
commands::{
VgpuPropertiesSchema, //
},
+ GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK,
GMCAPI_CMD_QUERY_ASSIGNED_VF_VGPU_TYPE,
GMCAPI_CMD_QUERY_VGPU_PROPERTIES, //
}, //
@@ -89,3 +92,19 @@ pub(super) fn query_vgpu_properties(
Ok(properties)
}
}
+
+/// Send BOOTLOAD and check its firmware status.
+pub(super) fn send_bootload(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ payload: &[u64],
+) -> Result {
+ let command_id = GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK;
+ let response = cmdq.send_gmc_and_receive_timeout(
+ command_id,
+ AsBytes::as_bytes(payload),
+ 0,
+ Delta::from_secs(10),
+ )?;
+ check_status(dev, command_id, response.status)
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index aabe114b55c2..82c29dfb467a 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -30,3 +30,6 @@

pub(super) const GMCAPI_CMD_QUERY_VGPU_PROPERTIES: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_QUERY_VGPU_PROPERTIES;
+
+pub(super) const GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK: u32 =
+ bindings::GMCAPI_COMMANDS_GMCAPI_CMD_BOOTLOAD_GSP_VGPU_PLUGIN_TASK;
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index 83048473d35f..bf8d44819e13 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -61,7 +61,6 @@ pub(crate) struct ChannelMapEntry(u64) {
impl ChannelMapEntry {
const KEY: KeyId = 0x1001;

- #[expect(dead_code)]
pub(crate) fn new(engine_type: u16, index: u16, chid_offset: u32) -> Self {
Self::zeroed()
.with_engine_type(engine_type)
@@ -159,7 +158,6 @@ pub(crate) struct BootloadInfo<'a> {
}

/// Encodes a `VGPU_BOOTLOAD` request using the typed NVKV schema.
-#[expect(dead_code)]
pub(crate) fn encode_vgpu_bootload(info: BootloadInfo<'_>) -> Result<EncodedStream> {
let request = VgpuBootloadRequest {
dbdf: info.dbdf.into(),
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index b945e216a1d6..0b4d03c5f0fc 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -39,7 +39,6 @@
size_of::<RawControlRegion>() == u32_as_usize(fw::VGPU_CPU_GSP_CTRL_BUFF_REGION_SIZE)
);
/// Physical VRAM regions containing the vGPU plugin logs.
-#[expect(dead_code)]
pub(super) struct PluginLogRegions {
pub(super) init: VramRegion,
pub(super) vgpu: VramRegion,
@@ -86,7 +85,6 @@ pub(super) struct CommBufferRegion<'map, 'gpu> {
kernel_log: VramRegion,
}

-#[expect(dead_code)]
impl<'map, 'gpu> CommBufferRegion<'map, 'gpu> {
/// Map the communication portion of a plugin management heap.
pub(super) fn new(
@@ -188,6 +186,19 @@ pub(super) fn plugin_logs(&self) -> PluginLogRegions {
}
}

+ /// Clear a previous boot marker before starting the plugin.
+ pub(super) fn clear_plugin_ready(&self) -> Result {
+ let offset = self.io_offset(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_seq_num),
+ size_of::<u32>(),
+ )?;
+ self.map.try_write32(0, offset)?;
+ // Complete the posted clear before firmware can publish its new marker.
+ self.map.try_read32(offset)?;
+ Ok(())
+ }
+
/// Return whether firmware has published the plugin boot marker.
pub(super) fn is_plugin_ready(&self) -> Result<bool> {
let value = self.read_u32(
@@ -199,6 +210,7 @@ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
}

/// Invalidate the PTEs and release the communication mapping.
+ #[expect(dead_code)]
pub(super) fn unmap(&mut self) -> Result {
self.map.unmap()
}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index a90aaef28b35..b2a0b25c0e15 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -7,18 +7,38 @@
};

use kernel::{
+ device,
prelude::*,
ptr::Alignment,
- sizes::SizeConstants, //
+ sizes::SizeConstants,
+ time::{
+ delay::fsleep,
+ Delta,
+ Instant,
+ Monotonic, //
+ }, //
};

use crate::{
gpu::ChannelIdReservation,
- mm::GpuMm, //
+ gsp::{
+ cmdq::Cmdq,
+ commands::FifoEngineList, //
+ },
+ mm::GpuMm,
};

use super::{
- commands::Dbdf,
+ commands::{
+ send_bootload,
+ Dbdf, //
+ },
+ fw::commands::{
+ encode_vgpu_bootload,
+ BootloadInfo,
+ ChannelMapEntry, //
+ },
+ gsp_plugin_comm::CommBufferRegion,
vram::{
VgpuVramLayout,
VgpuVramSlot,
@@ -27,6 +47,52 @@
VgpuManager, //
};

+/// Ready limit used by `vmiopd_negotiate_cpu_gsp_version()` for the same marker.
+const PLUGIN_READY_TIMEOUT: Delta = Delta::from_secs(10);
+
+/// Per-engine channel budget reserved by the full SR-IOV plugin.
+///
+/// The supported GB20x path keeps firmware's non-heavy default, which uses
+/// `PLUGIN_ALLOCATED_CHANNELS_PER_ENGINE` rather than the heavy-mode budget.
+const PLUGIN_CHANNELS_PER_ENGINE: u32 = 3;
+
+/// Build the typed channel mapping from the GSP FIFO engine list.
+fn channel_mapping(
+ fifo_engine_list: &FifoEngineList,
+ chid_offset: u32,
+) -> Result<KVVec<ChannelMapEntry>> {
+ let mut mapping = KVVec::new();
+ for &gmc_id in fifo_engine_list.gmc_ids() {
+ // CAST: The mask leaves only the low 16 bits of the GMC engine ID.
+ let engine_type = (gmc_id & u32::from(u16::MAX)) as u16;
+ // CAST: Shifting a `u32` by 16 leaves at most 16 bits.
+ let index = (gmc_id >> u16::BITS) as u16;
+ mapping.push(
+ ChannelMapEntry::new(engine_type, index, chid_offset),
+ GFP_KERNEL,
+ )?;
+ }
+ Ok(mapping)
+}
+
+fn wait_plugin_ready(
+ dev: &device::Device<device::Bound>,
+ comm: &CommBufferRegion<'_, '_>,
+) -> Result {
+ let start = Instant::<Monotonic>::now();
+
+ loop {
+ if comm.is_plugin_ready()? {
+ dev_dbg!(dev, "vGPU plugin ready after {:?}\n", start.elapsed());
+ return Ok(());
+ }
+ if start.elapsed() >= PLUGIN_READY_TIMEOUT {
+ return Err(ETIMEDOUT);
+ }
+ fsleep(Delta::from_millis(1));
+ }
+}
+
/// Guest Function ID validated against one device's total number of VFs.
///
/// GFID 0 is reserved for the PF; VFs start at 1. Keeping the nonzero `u16`
@@ -49,7 +115,6 @@ pub(super) fn new(gfid: u32, total_vfs: NonZero<u16>) -> Result<Self> {
}
}

- #[expect(dead_code)]
pub(super) const fn get(self) -> u16 {
self.0.get()
}
@@ -68,14 +133,72 @@ pub(super) struct VgpuType {
}

/// A vGPU instance and the resources reserved for it.
-#[expect(dead_code)]
struct VgpuInstance<'gpu> {
gfid: Gfid,
dbdf: Dbdf,
vgpu_type: VgpuType,
vm_pid: u32,
- chids: ChannelIdReservation<'gpu>,
+ num_plugin_channels: u32,
+ comm: CommBufferRegion<'gpu, 'gpu>,
+ // Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
+ chids: ChannelIdReservation<'gpu>,
+}
+
+impl<'gpu> VgpuInstance<'gpu> {
+ #[expect(dead_code)]
+ fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
+ let dev = vgpu.dev;
+ self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
+
+ Ok(())
+ }
+
+ /// Bootload the GSP vGPU plugin and wait for its BAR1 ready indication.
+ fn bootload(
+ &mut self,
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ fifo_engine_list: &FifoEngineList,
+ ) -> Result {
+ let fb = &self.vram_slot.fbmem;
+ let mgmt = &self.vram_slot.mgmt_heap;
+ let logs = self.comm.plugin_logs();
+
+ let payload = encode_vgpu_bootload(BootloadInfo {
+ dbdf: self.dbdf,
+ gfid: u32::from(self.gfid.get()),
+ vgpu_type: self.vgpu_type.vgpu_type_id,
+ vm_pid: self.vm_pid,
+ num_channels: u32::try_from(self.chids.len()).map_err(|_| EOVERFLOW)?,
+ num_plugin_channels: self.num_plugin_channels,
+ channel_mapping: channel_mapping(
+ fifo_engine_list,
+ u32::try_from(self.chids.start).map_err(|_| EOVERFLOW)?,
+ )?,
+ guest_fb: fb,
+ plugin_heap: mgmt,
+ ctrl_buffer_offset: 0,
+ init_log: &logs.init,
+ vgpu_log: &logs.vgpu,
+ kernel_log: &logs.kernel,
+ })?;
+
+ dev_dbg!(
+ dev,
+ "bootload: gfid={} sending {} typed NVKV bytes\n",
+ self.gfid.get(),
+ payload.len() * size_of::<u64>(),
+ );
+
+ self.comm.clear_plugin_ready()?;
+ send_bootload(dev, cmdq, &payload)?;
+
+ wait_plugin_ready(dev, &self.comm)?;
+
+ dev_dbg!(dev, "bootload: gfid={} plugin ready\n", self.gfid.get());
+ Ok(())
+ }
}

/// Identity and firmware profile used to allocate an instance.
@@ -133,12 +256,7 @@ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<
}

/// Allocate resources and register a new inactive vGPU instance.
- fn allocate_instance(
- &mut self,
- mm: &GpuMm<'_>,
- vgpu: &VgpuManager<'gpu>,
- info: InstanceInfo,
- ) -> Result<Gfid> {
+ fn allocate_instance(&mut self, vgpu: &VgpuManager<'gpu>, info: InstanceInfo) -> Result<Gfid> {
let InstanceInfo {
gfid,
dbdf,
@@ -165,8 +283,7 @@ fn allocate_instance(
return Err(ENOSPC);
}

- // Reserve registry capacity before acquiring resources so publishing
- // the completed instance cannot fail due to memory pressure.
+ // Reserve capacity before acquiring resources so registration cannot allocate.
self.instances.reserve(1, GFP_KERNEL)?;

let channels_per_instance = vgpu
@@ -187,15 +304,18 @@ fn allocate_instance(
heap_size: vgpu_type.gsp_heap_size,
fb_align: vgpu.vmmu_segment_size,
};
- let vram_slot = self.alloc_vram_slot(mm, vram_layout)?;
+ let vram_slot = self.alloc_vram_slot(&vgpu.mm.lock(), vram_layout)?;
+ let comm = CommBufferRegion::new(vgpu.bar_user, vgpu.mm, &vram_slot.mgmt_heap)?;

let instance = VgpuInstance {
gfid,
dbdf,
vgpu_type,
vm_pid,
- chids,
+ num_plugin_channels: PLUGIN_CHANNELS_PER_ENGINE,
+ comm,
vram_slot,
+ chids,
};
self.instances
.push_within_capacity(instance)
diff --git a/drivers/gpu/nova-core/vgpu/vram.rs b/drivers/gpu/nova-core/vgpu/vram.rs
index faaeea106009..bd7d1b4bf039 100644
--- a/drivers/gpu/nova-core/vgpu/vram.rs
+++ b/drivers/gpu/nova-core/vgpu/vram.rs
@@ -102,7 +102,6 @@ fn drop(&mut self) {
/// All device accesses and mappings of its regions must end before dropping the slot.
/// Clearing the entry locks a sleeping mutex, so dropping the slot may sleep.
#[must_use]
-#[expect(dead_code)]
pub(super) struct VgpuVramSlot {
pub(super) fbmem: VramRegion,
pub(super) mgmt_heap: VramRegion,