[PATCH v3 29/31] gpu: nova-core: vgpu: scrub guest VRAM with CeUtils

From: Zhi Wang

Date: Mon Sep 28 2026 - 06:41:37 EST


Scrub guest VRAM before plugin boot and after plugin shutdown, before its
slot can be reused. Reserve the last channel ID in each instance range
for CeUtils and give the remaining channels to the plugin.

Submit scrubs in chunks and wait for completion through a temporary BAR1
mapping of the CeUtils semaphore. Release the mapping on both success and
failure, preserving a polling error if unmapping also fails.

Keep allocation, scrubbing and release within the instance lifecycle.
Retain host resources until device removal when firmware ownership is
uncertain or a scrub fails, and do not retry a recorded teardown failure.

Signed-off-by: Zhi Wang <zhiw@xxxxxxxxxx>
---
drivers/gpu/nova-core/vgpu.rs | 1 +
drivers/gpu/nova-core/vgpu/commands.rs | 4 -
drivers/gpu/nova-core/vgpu/instance.rs | 51 +++++++-
drivers/gpu/nova-core/vgpu/scrubber.rs | 173 +++++++++++++++++++++++++
4 files changed, 223 insertions(+), 6 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/scrubber.rs

diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index d6bcdabafb42..f78a4621984a 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -38,6 +38,7 @@
mod gsp_plugin_rpc;
mod hal;
mod instance;
+mod scrubber;
mod vram;

/// vGPU state detected during GPU construction.
diff --git a/drivers/gpu/nova-core/vgpu/commands.rs b/drivers/gpu/nova-core/vgpu/commands.rs
index 704d13259cf3..52e8339504c6 100644
--- a/drivers/gpu/nova-core/vgpu/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/commands.rs
@@ -207,7 +207,6 @@ pub(super) fn set_plugin_bme(
}

/// Whether a failed allocation may still have transferred CHID ownership to firmware.
-#[expect(dead_code)]
pub(super) enum CeUtilsAllocError {
/// A matching firmware response explicitly rejected the allocation.
NotOwned(Error),
@@ -216,7 +215,6 @@ pub(super) enum CeUtilsAllocError {
}

/// Allocate a CeUtils channel and validate its semaphore description.
-#[expect(dead_code)]
pub(super) fn alloc_ceutils(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
@@ -265,7 +263,6 @@ pub(super) fn alloc_ceutils(
}

/// Release a CeUtils allocation, including one whose allocation reply was lost.
-#[expect(dead_code)]
pub(super) fn free_ceutils(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
@@ -282,7 +279,6 @@ pub(super) fn free_ceutils(
}

/// Submit an asynchronous guest FB scrub and return its work identifier.
-#[expect(dead_code)]
pub(super) fn submit_ceutils_scrub(
dev: &device::Device<device::Bound>,
cmdq: &Cmdq<'_>,
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index 23c2b791aa25..1ccf4309f2ed 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -31,12 +31,14 @@

use super::{
commands::{
+ free_ceutils,
negotiate_plugin_version,
send_bootload,
send_cleanup,
send_plugin_config,
send_shutdown,
set_plugin_bme,
+ CeUtilsAllocError,
Dbdf, //
},
fw::commands::{
@@ -47,6 +49,7 @@
},
gsp_plugin_comm::CommBufferRegion,
gsp_plugin_rpc::PluginRpc,
+ scrubber::CeUtils,
vram::{
VgpuVramLayout,
VgpuVramSlot,
@@ -151,12 +154,41 @@ struct VgpuInstance<'gpu> {
// Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
chids: ChannelIdReservation<'gpu>,
+ ceutils: Option<CeUtils>,
needs_teardown: bool,
/// An uncertain or failed operation retains resources until device removal.
failure: Option<Error>,
}

impl<'gpu> VgpuInstance<'gpu> {
+ fn initialize(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
+ let ceutils_chid =
+ u32::try_from(self.chids.end.checked_sub(1).ok_or(EINVAL)?).map_err(|_| EOVERFLOW)?;
+ let ceutils = match CeUtils::allocate(vgpu.dev, vgpu.cmdq, self.gfid, ceutils_chid) {
+ Ok(ceutils) => ceutils,
+ Err(CeUtilsAllocError::NotOwned(error)) => return Err(error),
+ Err(CeUtilsAllocError::MayOwn(error)) => {
+ self.failure = Some(error);
+ dev_err!(vgpu.dev, "CeUtils allocation failed: {:?}\n", error);
+ return Err(error);
+ }
+ };
+
+ let result = ceutils.scrub_guest_fb(
+ vgpu.dev,
+ vgpu.cmdq,
+ vgpu.bar_user,
+ vgpu.mm,
+ &self.vram_slot.fbmem,
+ );
+ self.ceutils = Some(ceutils);
+ if let Err(error) = result {
+ // A failed wait does not establish that the submitted scrub has stopped.
+ self.failure = Some(error);
+ }
+ result
+ }
+
fn activate(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
let dev = vgpu.dev;
self.bootload(dev, vgpu.cmdq, &vgpu.fifo_engine_list)?;
@@ -175,7 +207,17 @@ fn teardown(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
}
let result = (|| {
self.shutdown(vgpu.dev, vgpu.cmdq)?;
-
+ if let Some(ceutils) = self.ceutils.as_ref() {
+ ceutils.scrub_guest_fb(
+ vgpu.dev,
+ vgpu.cmdq,
+ vgpu.bar_user,
+ vgpu.mm,
+ &self.vram_slot.fbmem,
+ )?;
+ free_ceutils(vgpu.dev, vgpu.cmdq, self.gfid)?;
+ self.ceutils = None;
+ }
if self.needs_teardown {
send_cleanup(vgpu.dev, vgpu.cmdq, self.gfid)?;
self.needs_teardown = false;
@@ -241,7 +283,7 @@ fn configure_plugin(&mut self, dev: &device::Device<device::Bound>) -> Result {
self.dbdf,
self.vgpu_type.vgpu_type_id,
self.vm_pid,
- u32::try_from(self.chids.len()).map_err(|_| EOVERFLOW)?,
+ u32::try_from(self.chids.len().checked_sub(1).ok_or(EINVAL)?).map_err(|_| EOVERFLOW)?,
self.num_plugin_channels,
)?;

@@ -352,6 +394,9 @@ fn allocate_instance<'a>(
.total_channels
.checked_div(vgpu_type.max_instance)
.ok_or(EINVAL)?;
+ if channels_per_instance <= 1 {
+ return Err(EINVAL);
+ }
let channels_per_instance =
usize::try_from(channels_per_instance).map_err(|_| EOVERFLOW)?;
let channels_per_instance = NonZeroUsize::new(channels_per_instance).ok_or(EINVAL)?;
@@ -378,6 +423,7 @@ fn allocate_instance<'a>(
plugin_rpc: PluginRpc::new(comm, bar0, gfid),
vram_slot,
chids,
+ ceutils: None,
needs_teardown: false,
failure: None,
};
@@ -445,6 +491,7 @@ fn activate(mut self) -> Result {
.iter_mut()
.find(|instance| instance.gfid == self.gfid)
.ok_or(EIO)?;
+ instance.initialize(self.vgpu)?;
instance.activate(self.vgpu)?;
self.committed = true;
Ok(())
diff --git a/drivers/gpu/nova-core/vgpu/scrubber.rs b/drivers/gpu/nova-core/vgpu/scrubber.rs
new file mode 100644
index 000000000000..dc62ddc1ef01
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/scrubber.rs
@@ -0,0 +1,173 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! Per-VM CeUtils guest VRAM scrubbing.
+
+use kernel::{
+ device,
+ prelude::*,
+ sync::Mutex,
+ time::{
+ delay::fsleep,
+ Delta,
+ Instant,
+ Monotonic, //
+ },
+ types::ScopeGuard, //
+};
+
+use crate::{
+ gsp::cmdq::Cmdq,
+ mm::{
+ bar_user::BarUser,
+ vram::VramRegion,
+ GpuMm,
+ Pfn,
+ VramAddress, //
+ },
+ vgpu::instance::Gfid, //
+};
+
+use super::commands::{
+ self,
+ CeUtilsAllocError, //
+};
+
+// Semaphore layout from OpenRM `channel_utils.h`.
+const NV_CEUTILS_SEMA_PAGE_MAGIC: u32 = 0xce5e_5ea0;
+
+#[repr(C)]
+struct CeUtilsSemaphoreHeader {
+ magic: u32,
+ payload: u32,
+}
+
+static_assert!(size_of::<CeUtilsSemaphoreHeader>() == 8);
+
+const SEMA_PAGE_MAGIC_OFFSET: usize = core::mem::offset_of!(CeUtilsSemaphoreHeader, magic);
+const SEMA_PAGE_PAYLOAD_OFFSET: usize = core::mem::offset_of!(CeUtilsSemaphoreHeader, payload);
+
+const SCRUB_REQUEST_SIZE: u64 = 4 * 1024 * 1024 * 1024;
+
+// Host timeout policy; not a firmware ABI value.
+const SCRUB_TIMEOUT: Delta = Delta::from_secs(5);
+
+/// A firmware-owned per-VM CeUtils allocation.
+///
+/// Submitted scrubs must complete before releasing this allocation or its VRAM.
+pub(super) struct CeUtils {
+ gfid: Gfid,
+ semaphore_address: u64,
+}
+
+impl CeUtils {
+ pub(super) fn allocate(
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ gfid: Gfid,
+ chid: u32,
+ ) -> core::result::Result<Self, CeUtilsAllocError> {
+ let semaphore_address = commands::alloc_ceutils(dev, cmdq, gfid, chid)?;
+ Ok(Self {
+ gfid,
+ semaphore_address,
+ })
+ }
+
+ /// Scrub the complete guest VRAM and wait for semaphore completion.
+ pub(super) fn scrub_guest_fb(
+ &self,
+ dev: &device::Device<device::Bound>,
+ cmdq: &Cmdq<'_>,
+ bar_user: &BarUser<'_>,
+ mm: &Mutex<GpuMm<'_>>,
+ fb: &VramRegion,
+ ) -> Result {
+ let mut offset = fb.address();
+ let end = offset.checked_add(fb.size()).ok_or(EOVERFLOW)?;
+ while offset < end {
+ let size = core::cmp::min(SCRUB_REQUEST_SIZE, end - offset);
+ let work_id = commands::submit_ceutils_scrub(dev, cmdq, self.gfid, offset, size)?;
+ wait_scrub_complete(bar_user, mm, dev, self.semaphore_address, work_id)?;
+ offset = offset.checked_add(size).ok_or(EOVERFLOW)?;
+ }
+
+ Ok(())
+ }
+}
+
+/// Poll the GSP-owned CeUtils semaphore page through a temporary BAR1 map.
+fn wait_scrub_complete(
+ bar_user: &BarUser<'_>,
+ mm: &Mutex<GpuMm<'_>>,
+ dev: &device::Device<device::Bound>,
+ semaphore_address: u64,
+ work_id: u32,
+) -> Result {
+ let pfn = Pfn::from(VramAddress::from_raw(semaphore_address));
+ let semaphore_map = bar_user.map(&mut mm.lock(), &[pfn], false)?;
+ let semaphore_map = ScopeGuard::new_with_data(semaphore_map, |mapping| {
+ if let Err(error) = mapping.release(&mut mm.lock()) {
+ dev_err!(
+ dev,
+ "failed to release semaphore BAR1 mapping: {:?}\n",
+ error
+ );
+ }
+ });
+
+ let result = (|| {
+ let magic = semaphore_map.try_read32(SEMA_PAGE_MAGIC_OFFSET)?;
+ if magic != NV_CEUTILS_SEMA_PAGE_MAGIC {
+ dev_warn!(
+ dev,
+ "bad CeUtils semaphore magic {:#x}, expected {:#x}\n",
+ magic,
+ NV_CEUTILS_SEMA_PAGE_MAGIC,
+ );
+ return Err(EIO);
+ }
+
+ let start = Instant::<Monotonic>::now();
+ loop {
+ let value = semaphore_map.try_read32(SEMA_PAGE_PAYLOAD_OFFSET)?;
+ if value.wrapping_sub(work_id) < 0x8000_0000 {
+ dev_dbg!(
+ dev,
+ "scrub completed after {:?}: semaphore={:#x}, target={:#x}\n",
+ start.elapsed(),
+ value,
+ work_id,
+ );
+ return Ok(());
+ }
+
+ if start.elapsed() >= SCRUB_TIMEOUT {
+ dev_warn!(
+ dev,
+ "scrub timed out: semaphore={:#x}, target={:#x}\n",
+ value,
+ work_id,
+ );
+ return Err(ETIMEDOUT);
+ }
+ fsleep(Delta::from_millis(1));
+ }
+ })();
+
+ let cleanup = semaphore_map.dismiss().release(&mut mm.lock());
+ match result {
+ Ok(()) => cleanup,
+ Err(error) => {
+ if let Err(cleanup_error) = cleanup {
+ dev_err!(
+ dev,
+ "failed to release semaphore BAR1 mapping after error {:?}: {:?}\n",
+ error,
+ cleanup_error,
+ );
+ }
+ Err(error)
+ }
+ }
+}