From: Danilo Krummrich NVIDIA vGPU VFs need nova-core to manage their instances when assigned to userspace through VFIO. Add a Rust VFIO PCI variant driver that uses the typed SR-IOV PF APIs to create an instance on first open and close it on last close. Expose the assigned device and subsystem IDs, enforce the assigned BAR1 aperture, and coordinate firmware reset with PCI reset. Match NVIDIA devices through a VFIO override-only entry. Devices without an available Nova vGPU interface use ordinary PCI passthrough. Use the VFIO PCI adapter to keep callback data available throughout registration and removal. Extend the GPU build rules to make nova-core crate metadata available to the VFIO driver and export the Rust symbols it references. Signed-off-by: Danilo Krummrich Signed-off-by: Zhi Wang --- drivers/gpu/Makefile | 15 +- drivers/vfio/pci/Kconfig | 2 + drivers/vfio/pci/Makefile | 2 + drivers/vfio/pci/nvidia-vgpu/Kconfig | 20 ++ drivers/vfio/pci/nvidia-vgpu/Makefile | 3 + drivers/vfio/pci/nvidia-vgpu/nvidia_vgpu.rs | 366 ++++++++++++++++++++ 6 files changed, 405 insertions(+), 3 deletions(-) create mode 100644 drivers/vfio/pci/nvidia-vgpu/Kconfig create mode 100644 drivers/vfio/pci/nvidia-vgpu/Makefile create mode 100644 drivers/vfio/pci/nvidia-vgpu/nvidia_vgpu.rs diff --git a/drivers/gpu/Makefile b/drivers/gpu/Makefile index e372fc02139f..1b8a907e716b 100644 --- a/drivers/gpu/Makefile +++ b/drivers/gpu/Makefile @@ -19,8 +19,9 @@ nova-core-y := nova-core/nova_core.o nova-core/nova_core_exports.o obj-$(CONFIG_DRM_NOVA) += nova-drm.o nova-drm-y := drm/nova/nova.o -# Export Rust symbols from nova-core only if nova-drm actually references them. -nova-core-export-deps := $(if $(CONFIG_DRM_NOVA),$(obj)/drm/nova/nova.o) +# Export Rust symbols from nova-core only if dependent modules reference them. +nova-core-export-deps := $(if $(CONFIG_DRM_NOVA),$(obj)/drm/nova/nova.o) \ + $(if $(CONFIG_NVIDIA_VGPU_VFIO_PCI),$(obj)/../vfio/pci/nvidia-vgpu/nvidia_vgpu.o) rust_needed_exports = \ { $(if $(strip $(2)),$(NM) -u $(2);,) echo "__DEFINED_RUST_SYMBOLS__"; \ @@ -57,10 +58,18 @@ $(obj)/nova-core/nova_core_exports.o: private cmd_gensymtypes_c = \ $(obj)/nova-core/nova_core.o endif -# Output nova-core's crate metadata for use by nova-drm at compile time. +# Output nova-core's crate metadata for its Rust consumers at compile time. RUSTFLAGS_nova-core/nova_core.o += \ --emit=metadata=$(objtree)/$(obj)/nova-core/libnova_core.rmeta # Allow nova-drm to import nova-core's types. $(obj)/drm/nova/nova.o: $(obj)/nova-core/nova_core.o RUSTFLAGS_drm/nova/nova.o := -L $(objtree)/$(obj)/nova-core --extern nova_core + +# Build the NVIDIA VFIO driver here so metadata and generated exports share +# the same dependency graph as nova-drm. +obj-$(CONFIG_NVIDIA_VGPU_VFIO_PCI) += nvidia-vgpu-vfio-pci.o +nvidia-vgpu-vfio-pci-y := ../vfio/pci/nvidia-vgpu/nvidia_vgpu.o + +$(obj)/../vfio/pci/nvidia-vgpu/nvidia_vgpu.o: $(obj)/nova-core/nova_core.o +RUSTFLAGS_../vfio/pci/nvidia-vgpu/nvidia_vgpu.o := -L $(objtree)/$(obj)/nova-core --extern nova_core diff --git a/drivers/vfio/pci/Kconfig b/drivers/vfio/pci/Kconfig index 296bf01e185e..b48d8d1af42a 100644 --- a/drivers/vfio/pci/Kconfig +++ b/drivers/vfio/pci/Kconfig @@ -74,4 +74,6 @@ source "drivers/vfio/pci/qat/Kconfig" source "drivers/vfio/pci/xe/Kconfig" +source "drivers/vfio/pci/nvidia-vgpu/Kconfig" + endmenu diff --git a/drivers/vfio/pci/Makefile b/drivers/vfio/pci/Makefile index 6138f1bf241d..58c88304a185 100644 --- a/drivers/vfio/pci/Makefile +++ b/drivers/vfio/pci/Makefile @@ -24,3 +24,5 @@ obj-$(CONFIG_NVGRACE_GPU_VFIO_PCI) += nvgrace-gpu/ obj-$(CONFIG_QAT_VFIO_PCI) += qat/ obj-$(CONFIG_XE_VFIO_PCI) += xe/ + +# nvidia-vgpu is built from drivers/gpu/Makefile for nova-core crate linkage. diff --git a/drivers/vfio/pci/nvidia-vgpu/Kconfig b/drivers/vfio/pci/nvidia-vgpu/Kconfig new file mode 100644 index 000000000000..1363f84163e2 --- /dev/null +++ b/drivers/vfio/pci/nvidia-vgpu/Kconfig @@ -0,0 +1,20 @@ +# SPDX-License-Identifier: GPL-2.0-only +config NVIDIA_VGPU_VFIO_PCI + tristate "VFIO support for the NVIDIA vGPU" + depends on NOVA_CORE && PCI_IOV && RUST + select VFIO_PCI_CORE + help + This option enables VFIO (Virtual Function I/O) support for + NVIDIA virtual GPUs (vGPU). It allows the assignment of a virtual + GPU instance to userspace applications via VFIO, typically used + with hypervisors such as KVM and device emulators like QEMU. + + Devices without an available Nova vGPU interface use ordinary PCI + passthrough. Both paths use the VFIO PCI core defaults; vfio-pci + module options do not apply. + + The NVIDIA vGPU allows a physical GPU to be partitioned into + multiple virtual GPUs, each of which can be passed to a virtual + machine as a PCI device using the standard VFIO infrastructure. + + If you don't know what to do here, say N. diff --git a/drivers/vfio/pci/nvidia-vgpu/Makefile b/drivers/vfio/pci/nvidia-vgpu/Makefile new file mode 100644 index 000000000000..bce0133562f3 --- /dev/null +++ b/drivers/vfio/pci/nvidia-vgpu/Makefile @@ -0,0 +1,3 @@ +# SPDX-License-Identifier: GPL-2.0 +# nvidia-vgpu is built from drivers/gpu/Makefile for nova-core crate linkage. +# nvidia_vgpu.o (rust-analyzer marker - DO NOT REMOVE). diff --git a/drivers/vfio/pci/nvidia-vgpu/nvidia_vgpu.rs b/drivers/vfio/pci/nvidia-vgpu/nvidia_vgpu.rs new file mode 100644 index 000000000000..e3f1cd0abf5f --- /dev/null +++ b/drivers/vfio/pci/nvidia-vgpu/nvidia_vgpu.rs @@ -0,0 +1,366 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! NVIDIA vGPU VFIO variant driver. +//! +//! Nova VFs own an instance between the first VFIO open and last close. Other +//! matching devices use ordinary PCI passthrough. + +use kernel::{ + bindings, + device::{ + Bound, + Core, // + }, + io::resource::Flags, + pci, + prelude::*, + sync::{ + CondVar, + Mutex, + MutexGuard, // + }, + types::CovariantForLt, + vfio::{ + self, + pci::{ + GetRegionInfo, + Ioctl, + Mapping, + Mmap, + Open, + Position, + Read, + Write, // + }, // + }, // +}; +use nova_core::{ + NovaCoreVfApi, + NovaCoreVfApiHandle, + VgpuInstance, // +}; + +struct NvidiaVgpuOps; + +struct VgpuRegistration<'a> { + api: NovaCoreVfApiHandle<'a>, + gfid: u32, +} + +#[derive(Default)] +struct InstanceState { + active: bool, + resetting: bool, + reset_error: Option, +} + +#[pin_data] +struct NvidiaVgpuRegData<'a> { + pdev: &'a pci::Device, + vgpu: Option>, + fb_bar: u32, + #[pin] + state: Mutex, + #[pin] + reset_done: CondVar, +} + +impl NvidiaVgpuRegData<'_> { + fn lock_instance(&self) -> MutexGuard<'_, InstanceState> { + let mut state = self.state.lock(); + while state.resetting { + self.reset_done.wait(&mut state); + } + state + } +} + +struct NvidiaVgpuOpenData<'a> { + registration: &'a NvidiaVgpuRegData<'a>, + instance: Option>, +} + +impl<'a> NvidiaVgpuOpenData<'a> { + fn new( + dev: &vfio::pci::Device, + rd: &'a NvidiaVgpuRegData<'a>, + ) -> Result { + let mut state = rd.lock_instance(); + let instance = match &rd.vgpu { + Some(vgpu) => { + // The firmware DBDF format has a 16-bit PCI segment field. + let segment = u16::try_from(rd.pdev.domain_nr()).map_err(|_| EOVERFLOW)?; + let dbdf = (u32::from(segment) << 16) | u32::from(rd.pdev.dev_id()); + let vm_pid = kernel::current!().tgid().try_into()?; + let instance = vgpu.api.open(vgpu.gfid, dbdf, vm_pid)?; + let info = instance.type_info(); + dev.set_device_id(info.pci_dev_id as u16); + state.active = true; + state.reset_error = None; + Some(instance) + } + None => None, + }; + Ok(Self { + registration: rd, + instance, + }) + } + + fn bar1_size(&self) -> Result> { + let Some(instance) = &self.instance else { + return Ok(None); + }; + let physical = self + .registration + .pdev + .resource_len(self.registration.fb_bar)?; + let size = instance + .type_info() + .bar1_length + .checked_mul(1 << 20) + .ok_or(EOVERFLOW)?; + Ok(Some(if size == 0 { + physical + } else { + size.min(physical) + })) + } + + fn limit_bar1(&self, buf: &mut vfio::UserBuf, position: &Position<'_>) -> Result { + if buf.is_empty() || position.region_index() != self.registration.fb_bar { + return Ok(()); + } + if let Some(size) = self.bar1_size()? { + let offset = position.region_offset(); + if offset >= size { + return Err(EINVAL); + } + if size - offset < buf.len() as u64 { + buf.truncate((size - offset) as usize); + } + } + Ok(()) + } +} + +impl Drop for NvidiaVgpuOpenData<'_> { + fn drop(&mut self) { + let mut state = self.registration.lock_instance(); + state.active = false; + drop(self.instance.take()); + } +} + +impl vfio::pci::Operations for NvidiaVgpuOps { + const NAME: &'static CStr = c"nvidia-vgpu-vfio-pci"; + type RegistrationData = CovariantForLt!(NvidiaVgpuRegData<'_>); + type OpenData<'a> = NvidiaVgpuOpenData<'a>; + + fn open_device<'a>( + dev: &'a vfio::pci::Device, + rd: &'a NvidiaVgpuRegData<'a>, + ) -> impl PinInit, Error> + 'a { + NvidiaVgpuOpenData::new(dev, rd) + } + + fn ioctl<'a>( + dev: &vfio::pci::Device, + rd: &NvidiaVgpuRegData<'a>, + _open_data: Pin<&Self::OpenData<'a>>, + cmd: u32, + arg: usize, + ) -> Result { + let reset = cmd == vfio::DEVICE_RESET || cmd == vfio::pci::DEVICE_PCI_HOT_RESET; + if reset { + if let Some(error) = rd.state.lock().reset_error { + return Err(error); + } + } + let result = dev.core_ioctl(cmd, arg)?; + if reset { + if let Some(error) = rd.state.lock().reset_error { + return Err(error); + } + } + Ok(result) + } + + fn read<'a>( + dev: &vfio::pci::Device, + _rd: &NvidiaVgpuRegData<'a>, + open_data: Pin<&Self::OpenData<'a>>, + buf: &mut vfio::UserBuf, + position: &mut Position<'_>, + ) -> Result { + if let Some(instance) = &open_data.instance { + if position.region_index() == vfio::pci::CONFIG_REGION_INDEX { + let offset = position.region_offset(); + return position.with_temporary(|next| { + let count = dev.core_read(buf, next)?; + buf.write_overlapping( + offset, + count.try_into()?, + u64::from(bindings::PCI_SUBSYSTEM_ID), + &(instance.type_info().pci_subsys_id as u16).to_le_bytes(), + )?; + Ok(count) + }); + } + } + open_data.limit_bar1(buf, position)?; + dev.core_read(buf, position) + } + + fn write<'a>( + dev: &vfio::pci::Device, + _rd: &NvidiaVgpuRegData<'a>, + open_data: Pin<&Self::OpenData<'a>>, + buf: &mut vfio::UserBuf, + position: &mut Position<'_>, + ) -> Result { + open_data.limit_bar1(buf, position)?; + dev.core_write(buf, position) + } + + fn mmap<'a>( + dev: &vfio::pci::Device, + rd: &NvidiaVgpuRegData<'a>, + open_data: Pin<&Self::OpenData<'a>>, + mapping: &mut Mapping<'_>, + ) -> Result { + if mapping.region_index() == rd.fb_bar { + if let Some(size) = open_data.bar1_size()? { + if mapping.region_end()? > size { + return Err(EINVAL); + } + } + } + dev.core_mmap(mapping) + } + + fn get_region_info<'a>( + dev: &vfio::pci::Device, + rd: &NvidiaVgpuRegData<'a>, + open_data: Pin<&Self::OpenData<'a>>, + info: &mut bindings::vfio_region_info, + caps: &mut vfio::InfoCap<'_>, + ) -> Result { + dev.core_get_region_info(info, caps)?; + if info.index == rd.fb_bar && info.size != 0 { + if let Some(size) = open_data.bar1_size()? { + info.size = info.size.min(size); + } + } + Ok(()) + } + + fn reset_prepare(rd: &NvidiaVgpuRegData<'_>) { + // PCI holds the device lock, and VFIO may hold its memory lock. Open and + // close release this mutex before entering the VFIO core. + let mut state = rd.state.lock(); + state.resetting = true; + if state.active { + if let Some(vgpu) = &rd.vgpu { + if let Err(error) = vgpu.api.reset(vgpu.gfid) { + state.reset_error.get_or_insert(error); + } + } + } + } + + fn reset_done(rd: &NvidiaVgpuRegData<'_>) -> Result { + let mut state = rd.state.lock(); + let error = state.reset_error; + state.resetting = false; + rd.reset_done.notify_all(); + error.map_or(Ok(()), Err) + } +} + +struct NvidiaVgpuDriver; + +kernel::pci_device_table!( + PCI_TABLE, + ::IdInfo, + [( + pci::DeviceId::from_id_vfio_override(pci::Vendor::NVIDIA, bindings::PCI_ANY_ID as u32), + (), + ),] +); + +#[pin_data] +struct NvidiaVgpuData<'bound> { + reg: vfio::pci::Registration<'bound, NvidiaVgpuOps>, +} + +// SAFETY: The selector only borrows this device's sole registration and cannot sleep. This driver +// uses the VFIO adapter below and does not enable SR-IOV from probe or unbind. +unsafe impl vfio::pci::Driver for NvidiaVgpuDriver { + type Operations = NvidiaVgpuOps; + + fn registration<'a, 'bound>( + data: Pin<&'a Self::Data<'bound>>, + ) -> &'a vfio::pci::Registration<'bound, NvidiaVgpuOps> { + &data.get_ref().reg + } +} + +#[vtable] +impl pci::Driver for NvidiaVgpuDriver { + type IdInfo = (); + type Data<'bound> = NvidiaVgpuData<'bound>; + const ID_TABLE: pci::IdTable = &PCI_TABLE; + const DRIVER_MANAGED_DMA: bool = true; + + fn probe<'bound>( + pdev: &'bound pci::Device>, + _info: Option<&'bound Self::IdInfo>, + ) -> impl PinInit, Error> + 'bound { + try_pin_init!(NvidiaVgpuData { + reg: { + let vgpu = match NovaCoreVfApi::handle(pdev) { + Ok(api) => Some(VgpuRegistration { + api, + gfid: pdev.vf_id()?.checked_add(1).ok_or(EOVERFLOW)?, + }), + Err(_) => None, + }; + let dev = if vgpu.is_some() { + vfio::pci::Device::::new(pdev)? + } else { + vfio::pci::Device::::new_passthrough(pdev)? + }; + let fb_bar = if pdev.resource_flags(0)?.contains(Flags::IORESOURCE_MEM_64) { + 2 + } else { + 1 + }; + // SAFETY: The device was allocated for this probe and has not been + // registered. Private data owns it, and the VFIO adapter handles its registration. + unsafe { + vfio::pci::Registration::new( + pdev, + &dev, + try_pin_init!(NvidiaVgpuRegData { + pdev, + vgpu, + fb_bar, + state <- kernel::new_mutex!(InstanceState::default()), + reset_done <- kernel::new_condvar!(), + }), + )? + } + }, + }) + } +} + +kernel::module_driver!(, vfio::pci::Adapter, { + type: NvidiaVgpuDriver, + name: "nvidia-vgpu-vfio-pci", + authors: ["NVIDIA"], + description: "NVIDIA vGPU VFIO variant driver", + license: "GPL v2", +}); -- 2.53.0