[PATCH v3 24/31] gpu: nova-core: vgpu: add GSP plugin RPC transactions
From: Zhi Wang
Date: Mon Sep 28 2026 - 06:33:18 EST
A GSP plugin consumes a shared-buffer request after its VF doorbell is
rung and publishes completion in the response region.
Add PluginRpc to own the communication mapping together with the
instance's stable BAR0 and validated GFID. Accept typed RPC message IDs,
copy the payload before publishing its sequence, advance the local
sequence, ring and read back the doorbell, then wait for a matching
completion and check its firmware status.
Keep one sequence identity for each published request, including a
request whose doorbell or wait later fails, and bound completion polling
with the RPC timeout. Delegate explicit unmap to the owned communication
region so teardown errors retain its owner; ordinary drop releases the
mapping through BarMapping.
Signed-off-by: Zhi Wang <zhiw@xxxxxxxxxx>
---
drivers/gpu/nova-core/regs.rs | 31 +++-
drivers/gpu/nova-core/vgpu.rs | 1 +
drivers/gpu/nova-core/vgpu/fw.rs | 13 ++
drivers/gpu/nova-core/vgpu/fw/commands.rs | 10 ++
drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs | 51 +++++-
drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs | 155 ++++++++++++++++++
drivers/gpu/nova-core/vgpu/instance.rs | 15 +-
7 files changed, 267 insertions(+), 9 deletions(-)
create mode 100644 drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
diff --git a/drivers/gpu/nova-core/regs.rs b/drivers/gpu/nova-core/regs.rs
index ec8e05dc3351..cadaa3495b4e 100644
--- a/drivers/gpu/nova-core/regs.rs
+++ b/drivers/gpu/nova-core/regs.rs
@@ -7,13 +7,17 @@
Io,
Mmio, //
},
+ prelude::*,
sizes::SizeConstants,
time, //
};
use pin_init::Zeroable;
use crate::{
- driver::NovaRegisters,
+ driver::{
+ Bar0,
+ NovaRegisters, //
+ },
falcon::{
DmaTrfCmdSize,
FalconCoreRev,
@@ -39,6 +43,31 @@
pub(crate) NV_PBUS_SW_SCRATCH(u32)[64] @ 0x00001400 {}
}
+// VIRTUAL_FUNCTION
+
+register! {
+ base: NovaRegisters;
+
+ // PF BAR0 exposes the virtual-function register window at 0x00b8_0000.
+ pub(crate) NV_VIRTUAL_FUNCTION_PRIV_DOORBELL(u32) @ 0x00b8_2200 {
+ 31:0 handle;
+ }
+}
+
+impl NV_VIRTUAL_FUNCTION_PRIV_DOORBELL {
+ const DOORBELL_STRIDE: u32 = 32;
+ const DOORBELL_VECTOR: u32 = 17;
+
+ /// Notify the GSP plugin for the given guest function and read back the doorbell.
+ pub(crate) fn ring_gsp_plugin(bar0: Bar0<'_>, gfid: u16) -> Result {
+ // A `u16` GFID produces a handle of at most 0x1f_fff1, which fits in `u32`.
+ let value = u32::from(gfid) * Self::DOORBELL_STRIDE + Self::DOORBELL_VECTOR;
+ bar0.try_write_reg(Self::zeroed().with_handle(value))?;
+ bar0.try_read(NV_VIRTUAL_FUNCTION_PRIV_DOORBELL)?;
+ Ok(())
+ }
+}
+
// PGC6 register space.
//
// `GC6` is a GPU low-power state where VRAM is in self-refresh and the GPU is powered down (except
diff --git a/drivers/gpu/nova-core/vgpu.rs b/drivers/gpu/nova-core/vgpu.rs
index f7184e7009a5..d6bcdabafb42 100644
--- a/drivers/gpu/nova-core/vgpu.rs
+++ b/drivers/gpu/nova-core/vgpu.rs
@@ -35,6 +35,7 @@
mod commands;
mod fw;
mod gsp_plugin_comm;
+mod gsp_plugin_rpc;
mod hal;
mod instance;
mod vram;
diff --git a/drivers/gpu/nova-core/vgpu/fw.rs b/drivers/gpu/nova-core/vgpu/fw.rs
index 03225b9833c3..d80f36465296 100644
--- a/drivers/gpu/nova-core/vgpu/fw.rs
+++ b/drivers/gpu/nova-core/vgpu/fw.rs
@@ -10,6 +10,8 @@
use crate::gsp::bindings;
+pub(super) use commands::RpcMessage;
+
pub(super) use bindings::{
GSP_PLUGIN_BOOTLOADED,
VGPU_CPU_GSP_COMMUNICATION_BUFF_TOTAL_SIZE,
@@ -44,3 +46,14 @@
pub(super) const GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES: u32 =
bindings::GMCAPI_COMMANDS_GMCAPI_CMD_CLEANUP_GSP_VGPU_PLUGIN_RESOURCES;
+
+/// State observed in the response buffer for an expected RPC sequence.
+pub(super) enum RpcResponse {
+ Pending {
+ /// Last sequence completed by firmware.
+ sequence: u32,
+ },
+ Complete {
+ status: u32,
+ },
+}
diff --git a/drivers/gpu/nova-core/vgpu/fw/commands.rs b/drivers/gpu/nova-core/vgpu/fw/commands.rs
index bf8d44819e13..91d43cf6dd85 100644
--- a/drivers/gpu/nova-core/vgpu/fw/commands.rs
+++ b/drivers/gpu/nova-core/vgpu/fw/commands.rs
@@ -28,6 +28,16 @@
mm::vram::VramRegion, //
};
+use super::bindings;
+
+/// Message types supported by the nova-core plugin RPC channel.
+#[expect(dead_code)]
+#[derive(Clone, Copy)]
+#[repr(u32)]
+pub(crate) enum RpcMessage {
+ VersionNegotiation = bindings::MESSAGE_NV_VGPU_CPU_RPC_MSG_VERSION_NEGOTIATION,
+}
+
bitfield! {
pub(crate) struct Dbdf(u32) {
2:0 function;
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
index 31da31de4fd2..3d2c69e8cd79 100644
--- a/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_comm.rs
@@ -21,7 +21,9 @@
use super::fw::{
self,
RawControlRegion,
- RawResponseRegion, //
+ RawResponseRegion,
+ RpcMessage,
+ RpcResponse, //
};
static_assert!(
@@ -240,7 +242,6 @@ pub(super) fn is_plugin_ready(&self) -> Result<bool> {
}
/// Initialize the shared control and response buffers for plugin RPC.
- #[expect(dead_code)]
pub(super) fn initialize(&self) -> Result {
self.write_u64(
&self.control,
@@ -338,6 +339,52 @@ pub(super) fn initialize(&self) -> Result {
)
}
+ /// Copy and publish one RPC request to firmware.
+ pub(super) fn submit(&self, message: RpcMessage, sequence: u32, data: &[u8]) -> Result {
+ if u64::try_from(data.len()).map_err(|_| EOVERFLOW)? > self.message.size() {
+ return Err(E2BIG);
+ }
+
+ for (index, chunk) in data.chunks(size_of::<u32>()).enumerate() {
+ let mut bytes = [0u8; size_of::<u32>()];
+ bytes[..chunk.len()].copy_from_slice(chunk);
+ let field = index.checked_mul(size_of::<u32>()).ok_or(EOVERFLOW)?;
+ self.write_u32(&self.message, field, u32::from_le_bytes(bytes))?;
+ }
+
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_type),
+ // CAST: `RpcMessage` has a `u32` representation.
+ message as u32,
+ )?;
+ self.write_u32(
+ &self.control,
+ core::mem::offset_of!(RawControlRegion, __bindgen_anon_1.message_seq_num),
+ sequence,
+ )
+ }
+
+ /// Read firmware's response for an expected RPC sequence.
+ pub(super) fn response(&self, expected_sequence: u32) -> Result<RpcResponse> {
+ let sequence = self.read_u32(
+ &self.response,
+ core::mem::offset_of!(
+ RawResponseRegion,
+ __bindgen_anon_1.message_seq_num_processed
+ ),
+ )?;
+ if sequence != expected_sequence {
+ return Ok(RpcResponse::Pending { sequence });
+ }
+
+ let status = self.read_u32(
+ &self.response,
+ core::mem::offset_of!(RawResponseRegion, __bindgen_anon_1.result_code),
+ )?;
+ Ok(RpcResponse::Complete { status })
+ }
+
/// Invalidate the PTEs and release the communication mapping.
pub(super) fn unmap(&mut self) -> Result {
self.map.unmap()
diff --git a/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
new file mode 100644
index 000000000000..0ee32f011f63
--- /dev/null
+++ b/drivers/gpu/nova-core/vgpu/gsp_plugin_rpc.rs
@@ -0,0 +1,155 @@
+// SPDX-License-Identifier: GPL-2.0
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+//! GSP plugin RPC.
+//!
+//! ```text
+//! Host (PluginRpc) Shared RPC buffer (VRAM) GSP plugin
+//! | | |
+//! |-- BAR1: payload --------->| Message |
+//! |-- BAR1: type, sequence -->| Control |
+//! | | |
+//! |-- BAR0: VF doorbell ----------------------------->|
+//! | |<-- read request ------|
+//! | | | process RPC
+//! | |<-- completion --------|
+//! |-- poll processed seq ---->| Response |
+//! |<-- matching seq, status --| |
+//! ```
+
+use kernel::{
+ device,
+ prelude::*,
+ time::{
+ delay::fsleep,
+ Delta,
+ Instant,
+ Monotonic, //
+ },
+};
+
+use crate::{
+ driver::Bar0,
+ regs::NV_VIRTUAL_FUNCTION_PRIV_DOORBELL, //
+};
+
+use super::{
+ fw::{
+ RpcMessage,
+ RpcResponse, //
+ },
+ gsp_plugin_comm::CommBufferRegion,
+ instance::Gfid, //
+};
+
+/// BAR1-backed channel used to communicate with one GSP plugin.
+pub(super) struct PluginRpc<'map, 'gpu> {
+ comm: CommBufferRegion<'map, 'gpu>,
+ bar0: Bar0<'gpu>,
+ gfid: Gfid,
+ message_sequence: u32,
+}
+
+impl<'map, 'gpu> PluginRpc<'map, 'gpu> {
+ pub(super) fn new(comm: CommBufferRegion<'map, 'gpu>, bar0: Bar0<'gpu>, gfid: Gfid) -> Self {
+ Self {
+ comm,
+ bar0,
+ gfid,
+ message_sequence: 0,
+ }
+ }
+
+ pub(super) fn comm(&self) -> &CommBufferRegion<'map, 'gpu> {
+ &self.comm
+ }
+
+ /// Initialize the control and response buffers for the first RPC.
+ #[expect(dead_code)]
+ pub(super) fn init_rpc(&mut self) -> Result {
+ self.comm.initialize()?;
+ self.message_sequence = 0;
+ Ok(())
+ }
+
+ fn next_sequence(&self) -> u32 {
+ let sequence = self.message_sequence.wrapping_add(1);
+ if sequence == 0 {
+ 1
+ } else {
+ sequence
+ }
+ }
+
+ /// Write one RPC message, ring the VF doorbell, and wait for its response.
+ #[expect(dead_code)]
+ pub(super) fn rpc_call(
+ &mut self,
+ dev: &device::Device<device::Bound>,
+ message_type: RpcMessage,
+ data: &[u8],
+ ) -> Result {
+ let sequence = self.next_sequence();
+ self.comm.submit(message_type, sequence, data)?;
+ self.message_sequence = sequence;
+
+ dev_dbg!(
+ dev,
+ "vGPU RPC: gfid={} type={} bytes={} sequence={}\n",
+ self.gfid.get(),
+ // CAST: `RpcMessage` has a `u32` representation.
+ message_type as u32,
+ data.len(),
+ sequence,
+ );
+
+ NV_VIRTUAL_FUNCTION_PRIV_DOORBELL::ring_gsp_plugin(self.bar0, self.gfid.get())?;
+ self.wait_response(dev, sequence)
+ }
+
+ fn wait_response(&self, dev: &device::Device<device::Bound>, expected_sequence: u32) -> Result {
+ let start = Instant::<Monotonic>::now();
+ let timeout = Delta::from_secs(120);
+
+ loop {
+ match self.comm.response(expected_sequence)? {
+ RpcResponse::Complete { status } => {
+ if status != 0 {
+ dev_dbg!(
+ dev,
+ "vGPU RPC: sequence {} failed with status {}\n",
+ expected_sequence,
+ status,
+ );
+ return Err(EIO);
+ }
+
+ dev_dbg!(
+ dev,
+ "vGPU RPC: sequence {} completed after {:?}\n",
+ expected_sequence,
+ start.elapsed(),
+ );
+ return Ok(());
+ }
+ RpcResponse::Pending { sequence } => {
+ if start.elapsed() >= timeout {
+ dev_dbg!(
+ dev,
+ "vGPU RPC: sequence {} timed out; last response was {}\n",
+ expected_sequence,
+ sequence,
+ );
+ return Err(ETIMEDOUT);
+ }
+ }
+ }
+ fsleep(Delta::from_millis(1));
+ }
+ }
+
+ /// Release the BAR1 mapping.
+ pub(super) fn unmap(&mut self) -> Result {
+ self.comm.unmap()
+ }
+}
diff --git a/drivers/gpu/nova-core/vgpu/instance.rs b/drivers/gpu/nova-core/vgpu/instance.rs
index bf02595590e1..8a2d8e3970a3 100644
--- a/drivers/gpu/nova-core/vgpu/instance.rs
+++ b/drivers/gpu/nova-core/vgpu/instance.rs
@@ -20,6 +20,7 @@
};
use crate::{
+ driver::Bar0,
gpu::ChannelIdReservation,
gsp::{
cmdq::Cmdq,
@@ -41,6 +42,7 @@
ChannelMapEntry, //
},
gsp_plugin_comm::CommBufferRegion,
+ gsp_plugin_rpc::PluginRpc,
vram::{
VgpuVramLayout,
VgpuVramSlot,
@@ -141,7 +143,7 @@ struct VgpuInstance<'gpu> {
vgpu_type: VgpuType,
vm_pid: u32,
num_plugin_channels: u32,
- comm: CommBufferRegion<'gpu, 'gpu>,
+ plugin_rpc: PluginRpc<'gpu, 'gpu>,
// Unmap the communication region before returning its slot and channel IDs.
vram_slot: VgpuVramSlot,
chids: ChannelIdReservation<'gpu>,
@@ -170,7 +172,7 @@ fn teardown(&mut self, vgpu: &VgpuManager<'gpu>) -> Result {
send_cleanup(vgpu.dev, vgpu.cmdq, self.gfid)?;
self.needs_teardown = false;
}
- self.comm.unmap()
+ self.plugin_rpc.unmap()
})();
if let Err(error) = result {
self.failure = Some(error);
@@ -187,7 +189,7 @@ fn bootload(
) -> Result {
let fb = &self.vram_slot.fbmem;
let mgmt = &self.vram_slot.mgmt_heap;
- let logs = self.comm.plugin_logs();
+ let logs = self.plugin_rpc.comm().plugin_logs();
let payload = encode_vgpu_bootload(BootloadInfo {
dbdf: self.dbdf,
@@ -215,11 +217,11 @@ fn bootload(
payload.len() * size_of::<u64>(),
);
- self.comm.clear_plugin_ready()?;
+ self.plugin_rpc.comm().clear_plugin_ready()?;
self.needs_teardown = true;
send_bootload(dev, cmdq, &payload)?;
- wait_plugin_ready(dev, &self.comm)?;
+ wait_plugin_ready(dev, self.plugin_rpc.comm())?;
dev_dbg!(dev, "bootload: gfid={} plugin ready\n", self.gfid.get());
Ok(())
@@ -293,6 +295,7 @@ fn alloc_vram_slot(&mut self, mm: &GpuMm<'_>, layout: VgpuVramLayout) -> Result<
fn allocate_instance<'a>(
&'a mut self,
vgpu: &'a VgpuManager<'gpu>,
+ bar0: Bar0<'gpu>,
info: InstanceInfo,
) -> Result<PendingInstance<'a, 'gpu>> {
let InstanceInfo {
@@ -351,7 +354,7 @@ fn allocate_instance<'a>(
vgpu_type,
vm_pid,
num_plugin_channels: PLUGIN_CHANNELS_PER_ENGINE,
- comm,
+ plugin_rpc: PluginRpc::new(comm, bar0, gfid),
vram_slot,
chids,
needs_teardown: false,