[PATCH v3 18/31] gpu: nova-core: vgpu: add instance create/destroy

From: Zhi Wang

Date: Mon Sep 28 2026 - 06:36:41 EST


A vGPU instance owns the host resources reserved for a VF. Its assigned
type determines the channel range and VRAM and management-heap
reservation.

Add a manager-owned instance registry, reject duplicate GFID or DBDF
entries and enforce the profile's instance limit. Reserve registry
capacity before acquiring resources, then move the channel and VRAM
reservations into the instance. Removing an instance or dropping the
registry releases these host reservations through their owners.

Signed-off-by: Zhi Wang <zhiw@xxxxxxxxxx>
---
drivers/gpu/nova-core/gpu.rs | 12 +-
drivers/gpu/nova-core/vgpu.rs | 24 ++-
drivers/gpu/nova-core/vgpu/instance.rs | 218 +++++++++++++++++++++++++
drivers/gpu/nova-core/vgpu/vram.rs | 1 -
4 files changed, 240 insertions(+), 15 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/instance.rs

diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs
index 97b33d4b2631..b41a9921381d 100644
--- a/drivers/gpu/nova-core/gpu.rs
+++ b/drivers/gpu/nova-core/gpu.rs
@@ -54,14 +54,16 @@
},
};

-#[cfg_attr(not(CONFIG_KUNIT = "y"), expect(dead_code))]
mod channel;
mod hal;
mod regs;

use self::channel::TOTAL_CHANNELS;

-pub(crate) use self::channel::ChannelIdPool;
+pub(crate) use self::channel::{
+ ChannelIdPool,
+ ChannelIdReservation, //
+};

macro_rules! define_chipset {
({ $($variant:ident = $value:expr),* $(,)* }) =>
@@ -325,7 +327,7 @@ fn static_info(&self) -> &gsp::commands::GspStaticInfo {
#[pin_data]
pub(crate) struct Gpu<'gpu> {
spec: Spec,
- vgpu: Option<VgpuManager<'gpu>>,
+ vgpu: Option<Pin<KBox<VgpuManager<'gpu>>>>,
/// GSP event interrupt registration.
///
/// Must be kept declared *before* `gsp_resources`, so that the handler is unregistered, and
@@ -465,7 +467,7 @@ pub(crate) fn new<'a>(
let info = &gsp_resources.boot_result.static_info;
match gsp_resources.vgpu_state {
VgpuState::Disabled => None,
- VgpuState::Enabled { .. } => Some(VgpuManager::new(
+ VgpuState::Enabled { .. } => Some(KBox::pin_init(VgpuManager::new(
// SAFETY: `chid_pool` is initialized above at its final pinned address.
// The private manager and its pool borrow cannot escape this `Gpu`.
// Completed field drop order drops the manager before the pool; on failure,
@@ -474,7 +476,7 @@ pub(crate) fn new<'a>(
&info.fifo_engine_list(),
info.vmmu_segment_size,
TOTAL_CHANNELS,
- )),
+ ), GFP_KERNEL)?),
}
},

diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index d910445dd2b0..fb4f1b5f7754 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -4,8 +4,10 @@

use kernel::{
device,
+ new_mutex,
pci,
- prelude::*, //
+ prelude::*,
+ sync::Mutex, //
};

use crate::{
@@ -23,6 +25,7 @@
mod commands;
mod fw;
mod hal;
+mod instance;
mod vram;

/// vGPU state detected during GPU construction.
@@ -88,16 +91,17 @@ fn query_state(
}
}

+use self::instance::VgpuInstances;
+
/// Runtime resources for an enabled vGPU boot.
+#[pin_data]
pub(crate) struct VgpuManager<'gpu> {
- #[expect(dead_code)]
+ #[pin]
+ instances: Mutex<VgpuInstances<'gpu>>,
chid_pool: &'gpu ChannelIdPool,
/// VMMU segment size in bytes, or zero if GSP-RM omitted it.
- #[expect(dead_code)]
vmmu_segment_size: u64,
- #[expect(dead_code)]
total_channels: u32,
- #[expect(dead_code)]
fifo_engine_list: FifoEngineList,
}

@@ -108,12 +112,14 @@ pub(crate) fn new(
fifo_engine_list: &FifoEngineList,
vmmu_segment_size: u64,
total_channels: u32,
- ) -> Self {
- Self {
+ ) -> impl PinInit<Self> + use<'gpu> {
+ let fifo_engine_list = *fifo_engine_list;
+ pin_init!(Self {
+ instances <- new_mutex!(VgpuInstances::new(), "nova-core::vgpu-instances"),
chid_pool,
vmmu_segment_size,
total_channels,
- fifo_engine_list: *fifo_engine_list,
- }
+ fifo_engine_list,
+ })
}
}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
new file mode 100644
index 000000000000..a90aaef28b35
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -0,0 +1,218 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+use core::num::{
+ NonZero,
+ NonZeroUsize, //
+};
+
+use kernel::{
+ prelude::*,
+ ptr::Alignment,
+ sizes::SizeConstants, //
+};
+
+use crate::{
+ gpu::ChannelIdReservation,
+ mm::GpuMm, //
+};
+
+use super::{
+ commands::Dbdf,
+ vram::{
+ VgpuVramLayout,
+ VgpuVramSlot,
+ VgpuVramSlotAllocator, //
+ },
+ VgpuManager, //
+};
+
+/// Guest Function ID validated against one device's total number of VFs.
+///
+/// GFID 0 is reserved for the PF; VFs start at 1. Keeping the nonzero `u16`
+/// preserves the PCI SR-IOV range when forming a plugin doorbell handle.
+#[repr(transparent)]
+#[derive(Clone, Copy, PartialEq, Eq)]
+pub(super) struct Gfid(NonZero<u16>);
+
+impl Gfid {
+ /// Validates an external GFID for a device supporting `total_vfs` VFs.
+ #[expect(dead_code)]
+ pub(super) fn new(gfid: u32, total_vfs: NonZero<u16>) -> Result<Self> {
+ let gfid = u16::try_from(gfid).map_err(|_| EINVAL)?;
+ let gfid = NonZero::new(gfid).ok_or(EINVAL)?;
+
+ if gfid > total_vfs {
+ Err(EINVAL)
+ } else {
+ Ok(Self(gfid))
+ }
+ }
+
+ #[expect(dead_code)]
+ pub(super) const fn get(self) -> u16 {
+ self.0.get()
+ }
+}
+
+/// Resource requirements and device identity for one vGPU type.
+#[expect(dead_code)]
+pub(super) struct VgpuType {
+ vgpu_type_id: u32,
+ bar1_length: u64,
+ max_instance: u32,
+ pci_dev_id: u32,
+ pci_subsys_id: u32,
+ fb_length: u64,
+ gsp_heap_size: u64,
+}
+
+/// A vGPU instance and the resources reserved for it.
+#[expect(dead_code)]
+struct VgpuInstance<'gpu> {
+ gfid: Gfid,
+ dbdf: Dbdf,
+ vgpu_type: VgpuType,
+ vm_pid: u32,
+ chids: ChannelIdReservation<'gpu>,
+ vram_slot: VgpuVramSlot,
+}
+
+/// Identity and firmware profile used to allocate an instance.
+pub(super) struct InstanceInfo {
+ gfid: Gfid,
+ dbdf: Dbdf,
+ vgpu_type: VgpuType,
+ vm_pid: u32,
+}
+
+#[expect(dead_code)]
+impl InstanceInfo {
+ pub(super) const fn new(gfid: Gfid, dbdf: Dbdf, vgpu_type: VgpuType, vm_pid: u32) -> Self {
+ Self {
+ gfid,
+ dbdf,
+ vgpu_type,
+ vm_pid,
+ }
+ }
+}
+
+/// Registry of live vGPU instances.
+pub(super) struct VgpuInstances<'gpu> {
+ instances: KVec<VgpuInstance<'gpu>>,
+ vram_slots: Option<VgpuVramSlotAllocator>,
+}
+
+#[expect(dead_code)]
+impl<'gpu> VgpuInstances<'gpu> {
+ pub(super) const fn new() -> Self {
+ Self {
+ instances: KVec::new(),
+ vram_slots: None,
+ }
+ }
+
+ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<VgpuVramSlot> {
+ let replace_empty_pool = match self.vram_slots.as_ref() {
+ Some(allocator) if allocator.is_empty() => !allocator.matches_layout(layout)?,
+ _ => false,
+ };
+ if replace_empty_pool {
+ self.vram_slots = None;
+ }
+
+ if let Some(allocator) = self.vram_slots.as_mut() {
+ return allocator.alloc(layout);
+ }
+
+ let mut allocator = VgpuVramSlotAllocator::new(mm, layout)?;
+ let slot = allocator.alloc(layout)?;
+ self.vram_slots = Some(allocator);
+ Ok(slot)
+ }
+
+ /// Allocate resources and register a new inactive vGPU instance.
+ fn allocate_instance(
+ &mut self,
+ mm: &GpuMm<'_>,
+ vgpu: &VgpuManager<'gpu>,
+ info: InstanceInfo,
+ ) -> Result<Gfid> {
+ let InstanceInfo {
+ gfid,
+ dbdf,
+ vgpu_type,
+ vm_pid,
+ } = info;
+
+ let instance_exists = self
+ .instances
+ .iter()
+ .any(|instance| instance.gfid == gfid || instance.dbdf == dbdf);
+ if instance_exists {
+ return Err(EEXIST);
+ }
+
+ let type_id = vgpu_type.vgpu_type_id;
+ let num_type_instances = self
+ .instances
+ .iter()
+ .filter(|instance| instance.vgpu_type.vgpu_type_id == type_id)
+ .count();
+ let max_instances = usize::try_from(vgpu_type.max_instance).map_err(|_| EOVERFLOW)?;
+ if max_instances == 0 || num_type_instances >= max_instances {
+ return Err(ENOSPC);
+ }
+
+ // Reserve registry capacity before acquiring resources so publishing
+ // the completed instance cannot fail due to memory pressure.
+ self.instances.reserve(1, GFP_KERNEL)?;
+
+ let channels_per_instance = vgpu
+ .total_channels
+ .checked_div(vgpu_type.max_instance)
+ .ok_or(EINVAL)?;
+ let channels_per_instance =
+ usize::try_from(channels_per_instance).map_err(|_| EOVERFLOW)?;
+ let channels_per_instance = NonZeroUsize::new(channels_per_instance).ok_or(EINVAL)?;
+ let chids = vgpu
+ .chid_pool
+ .reserve_ids(channels_per_instance, Alignment::SZ_1)?;
+
+ let vram_layout = VgpuVramLayout {
+ type_id,
+ max_slots: vgpu_type.max_instance,
+ fb_size: vgpu_type.fb_length,
+ heap_size: vgpu_type.gsp_heap_size,
+ fb_align: vgpu.vmmu_segment_size,
+ };
+ let vram_slot = self.alloc_vram_slot(mm, vram_layout)?;
+
+ let instance = VgpuInstance {
+ gfid,
+ dbdf,
+ vgpu_type,
+ vm_pid,
+ chids,
+ vram_slot,
+ };
+ self.instances
+ .push_within_capacity(instance)
+ .map_err(|_| EIO)?;
+
+ Ok(gfid)
+ }
+
+ /// Remove an instance and release its channel and VRAM reservations.
+ fn destroy_instance(&mut self, gfid: Gfid) -> Result {
+ let instance_index = self
+ .instances
+ .iter()
+ .position(|instance| instance.gfid == gfid)
+ .ok_or(ENOENT)?;
+ let instance = self.instances.remove(instance_index).map_err(|_| EIO)?;
+ drop(instance);
+ Ok(())
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/vram.rs b/drivers/gpu/nova-core/vgpu/vram.rs
index 65e6a945ea6f..faaeea106009 100644
--- a/drivers/gpu/nova-core/vgpu/vram.rs
+++ b/drivers/gpu/nova-core/vgpu/vram.rs
@@ -116,7 +116,6 @@ pub(super) struct VgpuVramSlotAllocator {
slot_bitmap: Arc<SlotBitmap>,
}

-#[expect(dead_code)]
impl VgpuVramSlotAllocator {
pub(super) fn new(mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<Self> {
let layout = layout.validated()?;