diff options
83 files changed, 6708 insertions, 1344 deletions
diff --git a/Documentation/gpu/nova/core/pramin.rst b/Documentation/gpu/nova/core/pramin.rst new file mode 100644 index 000000000000..f50b052d73ba --- /dev/null +++ b/Documentation/gpu/nova/core/pramin.rst @@ -0,0 +1,128 @@ +.. SPDX-License-Identifier: GPL-2.0 + +========================= +PRAMIN aperture mechanism +========================= + +.. note:: + The following description is approximate and current as of the Ampere + family. It may change for future generations and is intended to assist in + understanding the driver code. + +Introduction +============ + +PRAMIN is a hardware aperture mechanism that provides CPU access to GPU Video +RAM (VRAM) before the GPU's Memory Management Unit (MMU) and page tables are +initialized. This 1 MiB sliding window, located at a fixed offset within BAR0, +is essential for setting up page tables and other critical GPU data structures +without relying on the GPU's MMU. + +Architecture Overview +===================== + +The PRAMIN aperture mechanism is logically implemented by the GPU's PBUS (PCIe +Bus Controller Unit) and provides a CPU-accessible window into VRAM through the +PCIe interface:: + + +-----------------+ PCIe +------------------------------+ + | CPU |<----------->| GPU | + +-----------------+ | | + | +----------------------+ | + | | PBUS | | + | | (Bus Controller) | | + | | | | + | | +--------------+ <------------ [1] + | | | PRAMIN | | | + | | | Window | | | + | | | (1 MiB) | | | + | | +--------------+ | | + | | | | | + | +---------|------------+ | + | | | + | v | + | +----------------------+ <------- [2] + | | VRAM | | + | | (Several GiB) | | + | | | | + | | FB[0x0000000000] | | + | | ... | | + | | FB[0xFFFFFFFFFF] | | + | +----------------------+ | + +------------------------------+ + + [1] Window starts at BAR0 + 0x700000. + [2] Program PRAMIN to any 64 KiB-aligned VRAM boundary. + +PBUS is responsible for, among other things, handling MMIO +accesses to the BAR registers. + +PRAMIN Window Operation +======================= + +The PRAMIN window provides a 1 MiB sliding aperture that can be repositioned +over the entire VRAM address space using the ``NV_PBUS_BAR0_WINDOW`` register. + +Window Control Mechanism +------------------------- + +:: + + NV_PBUS_BAR0_WINDOW Register (0x1700): + +-------+--------+--------------------------------------+ + | 31:26 | 25:24 | 23:0 | + | RSVD | TARGET | BASE_ADDR | + | | | (bits 39:16 of VRAM address) | + +-------+--------+--------------------------------------+ + + The 24-bit BASE_ADDR field encodes bits [39:16] of the target VRAM address, + providing 40-bit (1 TiB) address space coverage with 64 KiB alignment. + + TARGET field (bits 25:24): + - 0x0: VRAM (Video Memory) + - 0x1: Reserved (unused) + - 0x2: SYS_MEM_COH (Coherent System Memory) + - 0x3: SYS_MEM_NONCOH (Non-coherent System Memory) + +.. note:: + Nova only uses TARGET=VRAM (0x0) for video memory access. The SYS_MEM + target values are documented here for hardware completeness but are + not used by the driver. + +64 KiB Alignment Requirement +---------------------------- + +The PRAMIN window must be aligned to 64 KiB boundaries in VRAM. This is enforced +by the ``BASE_ADDR`` field representing bits [39:16] of the target address:: + + VRAM Address Calculation: + actual_vram_addr = (BASE_ADDR << 16) + pramin_offset + Where: + - BASE_ADDR: 24-bit value from NV_PBUS_BAR0_WINDOW[23:0] + - pramin_offset: 20-bit offset within the PRAMIN window [0x00000-0xFFFFF] + + Example Window Positioning: + +---------------------------------------------------------+ + | VRAM Space | + | | + | 0x0000000000 +-----------------+ <-- 64 KiB aligned | + | | PRAMIN Window | | + | | (1 MiB) | | + | 0x00000FFFFF +-----------------+ | + | | + | | ^ | + | | | Window can slide | + | v | to any 64 KiB-aligned boundary | + | | + | 0x0123400000 +-----------------+ <-- 64 KiB aligned | + | | PRAMIN Window | | + | | (1 MiB) | | + | 0x01234FFFFF +-----------------+ | + | | + | ... | + | | + | 0xFFFFF00000 +-----------------+ <-- 64 KiB aligned | + | | PRAMIN Window | | + | | (1 MiB) | | + | 0xFFFFFFFFFF +-----------------+ | + +---------------------------------------------------------+ diff --git a/Documentation/gpu/nova/core/todo.rst b/Documentation/gpu/nova/core/todo.rst index d5130b2b08fb..a01c362b1be0 100644 --- a/Documentation/gpu/nova/core/todo.rst +++ b/Documentation/gpu/nova/core/todo.rst @@ -33,7 +33,7 @@ A good example from nova-core would be the ``Chipset`` enum type, which defines the value ``AD102``. When probing the GPU the value ``0x192`` can be read from a certain register indication the chipset AD102. Hence, the enum value ``AD102`` should be derived from the number ``0x192``. Currently, nova-core uses a custom -implementation (``Chipset::from_u32`` for this. +implementation (``Chipset::from_u32``) for this. Instead, it would be desirable to have something like the ``FromPrimitive`` trait [1] from the num crate. diff --git a/Documentation/gpu/nova/index.rst b/Documentation/gpu/nova/index.rst index 2afa58e8f08d..59b206238498 100644 --- a/Documentation/gpu/nova/index.rst +++ b/Documentation/gpu/nova/index.rst @@ -34,3 +34,4 @@ vGPU manager VFIO driver and the nova-drm driver. core/fwsec core/falcon core/tlv + core/pramin diff --git a/MAINTAINERS b/MAINTAINERS index 841df4364f8e..5bf38ba77b15 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -7499,6 +7499,7 @@ F: rust/kernel/io.rs F: rust/kernel/io/ F: rust/kernel/irq.rs F: rust/kernel/irq/ +F: rust/macros/io/ DEVICE RESOURCE MANAGEMENT HELPERS M: Hans de Goede <hansg@kernel.org> @@ -7685,6 +7686,7 @@ F: fs/dlm/ DMA BUFFER SHARING FRAMEWORK M: Sumit Semwal <sumit.semwal@linaro.org> M: Christian König <christian.koenig@amd.com> +M: Philipp Stanner <phasta@kernel.org> L: linux-media@vger.kernel.org L: dri-devel@lists.freedesktop.org L: linaro-mm-sig@lists.linaro.org (moderated for non-subscribers) @@ -7698,6 +7700,8 @@ F: include/linux/dma-buf.h F: include/linux/dma-buf/ F: include/linux/dma-resv.h F: rust/helpers/dma-resv.c +F: rust/helpers/dma_fence.c +F: rust/kernel/dma_buf/ K: \bdma_(?:buf|fence|resv)\b DMA GENERIC OFFLOAD ENGINE SUBSYSTEM @@ -8711,7 +8715,9 @@ T: git https://gitlab.freedesktop.org/drm/rust/kernel.git F: drivers/gpu/drm/nova/ F: drivers/gpu/drm/tyr/ F: drivers/gpu/nova-core/ +F: rust/helpers/dma_fence.c F: rust/helpers/gpu.c +F: rust/kernel/dma_buf/ F: rust/kernel/drm/ F: rust/kernel/gpu.rs F: rust/kernel/gpu/ diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs index bfb0ba19caff..730b84e37a54 100644 --- a/drivers/gpu/drm/tyr/driver.rs +++ b/drivers/gpu/drm/tyr/driver.rs @@ -46,6 +46,7 @@ use crate::{ }; pub(crate) type IoMem<'a> = kernel::io::mem::IoMem<'a, SZ_2M>; +pub(crate) type TyrRegisters = kernel::io::Region<SZ_2M>; pub(crate) struct TyrDrmDriver; diff --git a/drivers/gpu/drm/tyr/fw.rs b/drivers/gpu/drm/tyr/fw.rs index 47d25c901bd0..7edb5eff1707 100644 --- a/drivers/gpu/drm/tyr/fw.rs +++ b/drivers/gpu/drm/tyr/fw.rs @@ -39,7 +39,8 @@ use kernel::{ use crate::{ driver::{ IoMem, - TyrDrmDevice, // + TyrDrmDevice, + TyrRegisters, // }, fw::parser::{ FwParser, @@ -101,6 +102,8 @@ impl From<CacheMode> for Bounded<u32, 2> { } register! { + base: TyrRegisters; + #[allow(non_upper_case_globals)] pub(super) SectionFlags(u32) @ 0x0 { 0:0 read => bool; diff --git a/drivers/gpu/drm/tyr/regs.rs b/drivers/gpu/drm/tyr/regs.rs index a62724378ced..0c419c4e1186 100644 --- a/drivers/gpu/drm/tyr/regs.rs +++ b/drivers/gpu/drm/tyr/regs.rs @@ -57,7 +57,11 @@ pub(crate) mod gpu_control { uapi, // }; + use crate::driver::TyrRegisters; + register! { + base: TyrRegisters; + /// GPU identification register. pub(crate) GPU_ID(u32) @ 0x0 { /// Status of the GPU release. @@ -315,6 +319,8 @@ pub(crate) mod gpu_control { } register! { + base: TyrRegisters; + /// GPU command register. /// /// Use the constructor methods to create commands: @@ -380,6 +386,8 @@ pub(crate) mod gpu_control { } register! { + base: TyrRegisters; + /// GPU status register. Read only. pub(crate) GPU_STATUS(u32) @ 0x34 { /// GPU active, a 1-bit boolean flag. @@ -463,6 +471,8 @@ pub(crate) mod gpu_control { } register! { + base: TyrRegisters; + /// GPU fault status register. Read only. pub(crate) GPU_FAULTSTATUS(u32) @ 0x3c { /// Exception type. @@ -768,6 +778,8 @@ pub(crate) mod gpu_control { } register! { + base: TyrRegisters; + /// Coherency enable. An index of which coherency protocols should be used. /// This register only selects the protocol for coherency messages on the /// interconnect. This is not to enable or disable coherency controlled by MMU. @@ -808,6 +820,8 @@ pub(crate) mod gpu_control { } register! { + base: TyrRegisters; + /// MCU control. pub(crate) MCU_CONTROL(u32) @ 0x700 { /// Request MCU state change. @@ -849,6 +863,8 @@ pub(crate) mod gpu_control { } register! { + base: TyrRegisters; + /// MCU status. Read only. pub(crate) MCU_STATUS(u32) @ 0x704 { /// Read current state of MCU. @@ -862,7 +878,11 @@ pub(crate) mod gpu_control { pub(crate) mod job_control { use kernel::register; + use crate::driver::TyrRegisters; + register! { + base: TyrRegisters; + /// Raw status of job interrupts. /// /// Write to this register to trigger these interrupts. @@ -912,7 +932,11 @@ pub(crate) mod job_control { pub(crate) mod mmu_control { use kernel::register; + use crate::driver::TyrRegisters; + register! { + base: TyrRegisters; + /// IRQ sources raw status. /// /// This register contains the raw unmasked interrupt sources for MMU status and exception @@ -966,9 +990,10 @@ pub(crate) mod mmu_control { prelude::*, register, // }; - use pin_init::Zeroable; + use crate::driver::TyrRegisters; + /// Maximum number of hardware address space slots. /// The actual number of slots available is usually lower. pub(crate) const MAX_AS: usize = 16; @@ -977,6 +1002,8 @@ pub(crate) mod mmu_control { const STRIDE: usize = 0x40; register! { + base: TyrRegisters; + /// Translation table base address. A 64-bit pointer. /// /// This field contains the address of the top level of a translation table structure. @@ -1104,6 +1131,8 @@ pub(crate) mod mmu_control { } register! { + base: TyrRegisters; + /// Stage 1 memory attributes (8-bit bitfield). /// /// This is not an actual register, but a bitfield definition used by the MEMATTR @@ -1137,6 +1166,8 @@ pub(crate) mod mmu_control { } register! { + base: TyrRegisters; + /// Memory attributes. /// /// Each address space can configure up to 8 different memory attribute profiles. @@ -1292,6 +1323,8 @@ pub(crate) mod mmu_control { } register! { + base: TyrRegisters; + /// Lock region address for each address space. pub(crate) LOCKADDR(u64)[MAX_AS, stride = STRIDE] @ 0x2410 { /// Lock region size. @@ -1353,6 +1386,8 @@ pub(crate) mod mmu_control { } register! { + base: TyrRegisters; + /// MMU command register for each address space. Write only. pub(crate) COMMAND(u32)[MAX_AS, stride = STRIDE] @ 0x2418 { 7:0 command ?=> MmuCommand; @@ -1480,6 +1515,8 @@ pub(crate) mod mmu_control { } register! { + base: TyrRegisters; + /// Fault status register for each address space. Read only. pub(crate) FAULTSTATUS(u32)[MAX_AS, stride = STRIDE] @ 0x241c { /// Exception type. @@ -1705,6 +1742,8 @@ pub(crate) mod mmu_control { } register! { + base: TyrRegisters; + /// Translation configuration and control. pub(crate) TRANSCFG(u64)[MAX_AS, stride = STRIDE] @ 0x2430 { /// Address space mode. @@ -1760,6 +1799,8 @@ pub(crate) mod mmu_control { pub(crate) mod doorbell_block { use kernel::register; + use crate::driver::TyrRegisters; + /// Number of doorbells available. pub(crate) const NUM_DOORBELLS: usize = 64; @@ -1770,6 +1811,8 @@ pub(crate) mod doorbell_block { const STRIDE: usize = 0x10000; register! { + base: TyrRegisters; + /// Doorbell request register. Write-only. pub(crate) DOORBELL(u32)[NUM_DOORBELLS, stride = STRIDE] @ 0x80000 { /// Doorbell set. Writing 1 triggers the doorbell. diff --git a/drivers/gpu/nova-core/Kconfig b/drivers/gpu/nova-core/Kconfig index f918f69e0599..1934f17baa8b 100644 --- a/drivers/gpu/nova-core/Kconfig +++ b/drivers/gpu/nova-core/Kconfig @@ -5,6 +5,7 @@ config NOVA_CORE depends on RUST depends on !CPU_BIG_ENDIAN select AUXILIARY_BUS + select GPU_BUDDY select RUST_FW_LOADER_ABSTRACTIONS default n help @@ -15,3 +16,12 @@ config NOVA_CORE This driver is work in progress and may not be functional. If M is selected, the module will be called nova-core. + +config NOVA_CORE_SELFTESTS + bool "Nova Core driver self-tests" + depends on NOVA_CORE + default n + help + Build the driver self-tests and run them when the GPU is probed. + + If unsure, say N. diff --git a/drivers/gpu/nova-core/driver.rs b/drivers/gpu/nova-core/driver.rs index bbd93959e0b2..0672a0707a71 100644 --- a/drivers/gpu/nova-core/driver.rs +++ b/drivers/gpu/nova-core/driver.rs @@ -2,7 +2,11 @@ use kernel::{ auxiliary, - device::Core, + device::{ + Bound, + Core, // + }, + io::resource, pci, pci::{ Class, @@ -28,6 +32,7 @@ pub(crate) struct NovaCore<'bound> { #[pin] pub(crate) gpu: Gpu<'bound>, bar: pci::Bar<'bound, BAR0_SIZE>, + bar1: Bar1<'bound>, #[allow(clippy::type_complexity)] _reg: auxiliary::Registration<'bound, CovariantForLt!(())>, } @@ -37,6 +42,27 @@ pub(crate) struct NovaCoreDriver; const BAR0_SIZE: usize = SZ_16M; pub(crate) type Bar0<'a> = &'a pci::Bar<'a, BAR0_SIZE>; +pub(crate) type NovaRegisters = kernel::io::Region<BAR0_SIZE>; +pub(crate) type Bar1<'a> = pci::Bar<'a>; + +/// Returns the Linux PCI resource index that holds BAR1 for an NVIDIA GPU. +/// +/// On Maxwell through Ada, BAR0 is a 32-bit memory BAR occupying a single +/// Linux PCI resource slot, so BAR1 lives at index 1. Starting with Blackwell +/// (and on some Ampere GA100 / Hopper SKUs) BAR0 is a 64-bit memory BAR that +/// consumes two consecutive resource slots: index 0 holds the low 32 bits and +/// index 1 holds the high 32 bits (with no `flags` / or size of its own), +/// shifting BAR1 to index 2. +pub(crate) fn bar1_resource_index(pdev: &pci::Device<Bound>) -> Result<u32> { + // Probe the `IORESOURCE_MEM_64` flag of BAR0 as a robust way of exposing + // if BAR0 and hence BAR1 is 64-bit. + let flags0 = pdev.resource_flags(0)?; + if flags0.contains(resource::Flags::IORESOURCE_MEM_64) { + Ok(2) + } else { + Ok(1) + } +} kernel::pci_device_table!( PCI_TABLE, @@ -79,12 +105,21 @@ impl pci::Driver for NovaCoreDriver { Ok(try_pin_init!(NovaCore { bar: pdev.iomap_region_sized::<BAR0_SIZE>(0, c"nova-core/bar0")?, - // TODO: Use `&bar` self-referential pin-init syntax once available. - // - // SAFETY: `bar` is initialized before this expression is evaluated - // (`try_pin_init!()` initializes fields in declaration order), lives at a pinned - // stable address, and is dropped after `gpu` (struct field drop order). - gpu <- Gpu::new(pdev, unsafe { &*core::ptr::from_ref(bar) }), + bar1: { + let bar1_idx = bar1_resource_index(pdev)?; + pdev.iomap_region(bar1_idx, c"nova-core/bar1")? + }, + // TODO: Use self-referential pin-init syntax once available. + gpu <- Gpu::new( + pdev, + // SAFETY: `bar` is initialized above, pinned, and outlives `gpu`. + unsafe { &*core::ptr::from_ref(bar) }, + // SAFETY: `bar1` is initialized above, pinned, and outlives `gpu`. + unsafe { &*core::ptr::from_ref(bar1) }, + ), + // Run optional GPU selftests. + #[cfg(CONFIG_NOVA_CORE_SELFTESTS)] + _: { gpu.run_selftests(pdev) }, _reg: auxiliary::Registration::new( pdev.as_ref(), c"nova-drm", diff --git a/drivers/gpu/nova-core/falcon.rs b/drivers/gpu/nova-core/falcon.rs index 65cb12d26e2b..9015de965a53 100644 --- a/drivers/gpu/nova-core/falcon.rs +++ b/drivers/gpu/nova-core/falcon.rs @@ -14,13 +14,12 @@ use kernel::{ io::{ io_project, poll::read_poll_timeout, - register::{ - RegisterBase, - WithBase, // - }, + register::Array, Io, + Mmio, // }, prelude::*, + sizes::SZ_4K, time::Delta, }; @@ -165,18 +164,25 @@ bounded_enum! { } } -/// Type used to represent the `PFALCON` registers address base for a given falcon engine. -pub(crate) struct PFalconBase(()); +const PFALCON_REGION_SIZE: usize = SZ_4K; +const PFALCON2_REGION_SIZE: usize = SZ_4K; -/// Type used to represent the `PFALCON2` registers address base for a given falcon engine. -pub(crate) struct PFalcon2Base(()); +/// Type used to represent the `PFALCON` registers. +#[repr(align(4))] +#[derive(FromBytes, IntoBytes)] +pub(crate) struct PFalconRegisters([u8; PFALCON_REGION_SIZE]); + +/// Type used to represent the `PFALCON2` registers. +#[repr(align(4))] +#[derive(FromBytes, IntoBytes)] +pub(crate) struct PFalcon2Registers([u8; PFALCON2_REGION_SIZE]); /// Trait defining the parameters of a given Falcon engine. /// /// Each engine provides one base for `PFALCON` and `PFALCON2` registers. -pub(crate) trait FalconEngine: - Send + Sync + RegisterBase<PFalconBase> + RegisterBase<PFalcon2Base> + Sized -{ +pub(crate) trait FalconEngine: Send + Sync + Sized { + fn pfalcon(io: Bar0<'_>) -> Mmio<'_, PFalconRegisters>; + fn pfalcon2(io: Bar0<'_>) -> Mmio<'_, PFalcon2Registers>; } /// Represents a portion of the firmware to be loaded into a particular memory (e.g. IMEM or DMEM) @@ -358,6 +364,9 @@ pub(crate) struct Falcon<'a, E: FalconEngine> { hal: KBox<dyn FalconHal<E>>, dev: &'a device::Device<device::Bound>, bar: Bar0<'a>, + // TODO: make private + pub(crate) pfalcon: Mmio<'a, PFalconRegisters>, + pfalcon2: Mmio<'a, PFalcon2Registers>, } impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { @@ -371,19 +380,19 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { hal: hal::falcon_hal(chipset)?, dev, bar, + pfalcon: E::pfalcon(bar), + pfalcon2: E::pfalcon2(bar), }) } /// Resets DMA-related registers. pub(crate) fn dma_reset(&self) { - self.bar.update(regs::NV_PFALCON_FBIF_CTL::of::<E>(), |v| { + self.pfalcon.update(regs::NV_PFALCON_FBIF_CTL, |v| { v.with_allow_phys_no_ctx(true) }); - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_DMACTL::zeroed(), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_DMACTL::zeroed()); } /// Reset the controller, select the falcon core, and wait for memory scrubbing to complete. @@ -392,10 +401,10 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { self.hal.select_core(self)?; self.hal.reset_wait_mem_scrubbing(self)?; - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_RM::from(self.bar.read(regs::NV_PMC_BOOT_0).into_raw()), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_RM::from(crate::gpu::boot_0_raw( + self.bar, + ))); Ok(()) } @@ -413,8 +422,8 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { return Err(EINVAL); } - self.bar.write( - WithBase::of::<E>().at(Self::PIO_PORT), + self.pfalcon.write( + Array::at(Self::PIO_PORT), regs::NV_PFALCON_FALCON_IMEMC::zeroed() .with_secure(load_offsets.secure) .with_aincw(true) @@ -424,14 +433,14 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { for (n, block) in load_offsets.data.chunks(MEM_BLOCK_ALIGNMENT).enumerate() { let n = u16::try_from(n)?; let tag: u16 = load_offsets.start_tag.checked_add(n).ok_or(ERANGE)?; - self.bar.write( - WithBase::of::<E>().at(Self::PIO_PORT), + self.pfalcon.write( + Array::at(Self::PIO_PORT), regs::NV_PFALCON_FALCON_IMEMT::zeroed().with_tag(tag), ); for word in block.chunks_exact(4) { let w = [word[0], word[1], word[2], word[3]]; - self.bar.write( - WithBase::of::<E>().at(Self::PIO_PORT), + self.pfalcon.write( + Array::at(Self::PIO_PORT), regs::NV_PFALCON_FALCON_IMEMD::zeroed().with_data(u32::from_le_bytes(w)), ); } @@ -450,8 +459,8 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { return Err(EINVAL); } - self.bar.write( - WithBase::of::<E>().at(Self::PIO_PORT), + self.pfalcon.write( + Array::at(Self::PIO_PORT), regs::NV_PFALCON_FALCON_DMEMC::zeroed() .with_aincw(true) .with_offs(load_offsets.dst_start), @@ -459,8 +468,8 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { for word in load_offsets.data.chunks_exact(4) { let w = [word[0], word[1], word[2], word[3]]; - self.bar.write( - WithBase::of::<E>().at(Self::PIO_PORT), + self.pfalcon.write( + Array::at(Self::PIO_PORT), regs::NV_PFALCON_FALCON_DMEMD::zeroed().with_data(u32::from_le_bytes(w)), ); } @@ -473,14 +482,12 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { &self, fw: &F, ) -> Result { - self.bar.update(regs::NV_PFALCON_FBIF_CTL::of::<E>(), |v| { + self.pfalcon.update(regs::NV_PFALCON_FBIF_CTL, |v| { v.with_allow_phys_no_ctx(true) }); - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_DMACTL::zeroed(), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_DMACTL::zeroed()); if let Some(imem_ns) = fw.imem_ns_load_params() { self.pio_wr_imem_slice(imem_ns)?; @@ -492,10 +499,8 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { self.hal.program_brom(self, &fw.brom_params()); - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_BOOTVEC::zeroed().with_value(fw.boot_addr()), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_BOOTVEC::zeroed().with_value(fw.boot_addr())); Ok(()) } @@ -506,7 +511,7 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { /// `sec` is set if the loaded firmware is expected to run in secure mode. fn dma_wr( &self, - dma_obj: &Coherent<[u8]>, + dma_obj: &Coherent<'_, [u8]>, target_mem: FalconMem, load_offsets: FalconDmaLoadTarget, ) -> Result { @@ -547,16 +552,13 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { // Set up the base source DMA address. - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_DMATRFBASE::zeroed().with_base( + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_DMATRFBASE::zeroed().with_base( // CAST: `as u32` is used on purpose since we do want to strip the upper bits, // which will be written to `NV_PFALCON_FALCON_DMATRFBASE1`. (dma_address >> 8) as u32, - ), - ); - self.bar.write( - WithBase::of::<E>(), + )); + self.pfalcon.write_reg( regs::NV_PFALCON_FALCON_DMATRFBASE1::zeroed().try_with_base(dma_address >> 40)?, ); @@ -566,23 +568,21 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { for pos in (0..num_transfers).map(|i| i * DMA_LEN) { // Perform a transfer of size `DMA_LEN`. - self.bar.write( - WithBase::of::<E>(), + self.pfalcon.write_reg( regs::NV_PFALCON_FALCON_DMATRFMOFFS::zeroed() .try_with_offs(load_offsets.dst_start + pos)?, ); - self.bar.write( - WithBase::of::<E>(), + self.pfalcon.write_reg( regs::NV_PFALCON_FALCON_DMATRFFBOFFS::zeroed().with_offs(src_start + pos), ); - self.bar.write(WithBase::of::<E>(), cmd); + self.pfalcon.write_reg(cmd); // Wait for the transfer to complete. // TIMEOUT: arbitrarily large value, no DMA transfer to the falcon's small memories // should ever take that long. read_poll_timeout( - || Ok(self.bar.read(regs::NV_PFALCON_FALCON_DMATRFCMD::of::<E>())), + || Ok(self.pfalcon.read(regs::NV_PFALCON_FALCON_DMATRFCMD)), |r| r.idle(), Delta::ZERO, Delta::from_secs(2), @@ -614,8 +614,8 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { }; self.dma_reset(); - self.bar - .update(regs::NV_PFALCON_FBIF_TRANSCFG::of::<E>().at(0), |v| { + self.pfalcon + .update(regs::NV_PFALCON_FBIF_TRANSCFG::at(0), |v| { v.with_target(FalconFbifTarget::CoherentSysmem) .with_mem_type(FalconFbifMemType::Physical) }); @@ -626,10 +626,8 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { self.hal.program_brom(self, &fw.brom_params()); // Set `BootVec` to start of non-secure code. - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_BOOTVEC::zeroed().with_value(fw.boot_addr()), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_BOOTVEC::zeroed().with_value(fw.boot_addr())); Ok(()) } @@ -638,7 +636,7 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { pub(crate) fn wait_till_halted(&self) -> Result<()> { // TIMEOUT: arbitrarily large value, firmwares should complete in less than 2 seconds. read_poll_timeout( - || Ok(self.bar.read(regs::NV_PFALCON_FALCON_CPUCTL::of::<E>())), + || Ok(self.pfalcon.read(regs::NV_PFALCON_FALCON_CPUCTL)), |r| r.halted(), Delta::ZERO, Delta::from_secs(2), @@ -649,19 +647,13 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { /// Start the falcon CPU. pub(crate) fn start(&self) -> Result<()> { - match self - .bar - .read(regs::NV_PFALCON_FALCON_CPUCTL::of::<E>()) - .alias_en() - { - true => self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_CPUCTL_ALIAS::zeroed().with_startcpu(true), - ), - false => self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_CPUCTL::zeroed().with_startcpu(true), - ), + match self.pfalcon.read(regs::NV_PFALCON_FALCON_CPUCTL).alias_en() { + true => self + .pfalcon + .write_reg(regs::NV_PFALCON_FALCON_CPUCTL_ALIAS::zeroed().with_startcpu(true)), + false => self + .pfalcon + .write_reg(regs::NV_PFALCON_FALCON_CPUCTL::zeroed().with_startcpu(true)), } Ok(()) @@ -670,32 +662,24 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { /// Writes values to the mailbox registers if provided. pub(crate) fn write_mailboxes(&self, mbox0: Option<u32>, mbox1: Option<u32>) { if let Some(mbox0) = mbox0 { - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_MAILBOX0::zeroed().with_value(mbox0), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_MAILBOX0::zeroed().with_value(mbox0)); } if let Some(mbox1) = mbox1 { - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_MAILBOX1::zeroed().with_value(mbox1), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_MAILBOX1::zeroed().with_value(mbox1)); } } /// Reads the value from `mbox0` register. pub(crate) fn read_mailbox0(&self) -> u32 { - self.bar - .read(regs::NV_PFALCON_FALCON_MAILBOX0::of::<E>()) - .value() + self.pfalcon.read(regs::NV_PFALCON_FALCON_MAILBOX0).value() } /// Reads the value from `mbox1` register. pub(crate) fn read_mailbox1(&self) -> u32 { - self.bar - .read(regs::NV_PFALCON_FALCON_MAILBOX1::of::<E>()) - .value() + self.pfalcon.read(regs::NV_PFALCON_FALCON_MAILBOX1).value() } /// Reads values from both mailbox registers. @@ -760,9 +744,7 @@ impl<'a, E: FalconEngine + 'static> Falcon<'a, E> { /// Write the application version to the OS register. pub(crate) fn write_os_version(&self, app_version: u32) { - self.bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON_FALCON_OS::zeroed().with_value(app_version), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_OS::zeroed().with_value(app_version)); } } diff --git a/drivers/gpu/nova-core/falcon/fsp.rs b/drivers/gpu/nova-core/falcon/fsp.rs index 0437180b8829..85f9c8c5d60e 100644 --- a/drivers/gpu/nova-core/falcon/fsp.rs +++ b/drivers/gpu/nova-core/falcon/fsp.rs @@ -8,13 +8,12 @@ use kernel::{ io::{ + io_project, poll::read_poll_timeout, - register::{ - Array, - RegisterBase, - WithBase, // - }, - Io, // + register, + register::Array, + Io, + Mmio, // }, prelude::*, sizes::SZ_1K, @@ -22,11 +21,13 @@ use kernel::{ }; use crate::{ + driver::{ + Bar0, + NovaRegisters, // + }, falcon::{ Falcon, - FalconEngine, - PFalcon2Base, - PFalconBase, // + FalconEngine, // }, num, regs, // @@ -41,15 +42,24 @@ const FSP_EMEM_CHANNEL_0_SIZE: usize = SZ_1K; /// Type specifying the `Fsp` falcon engine. Cannot be instantiated. pub(crate) struct Fsp(()); -impl RegisterBase<PFalconBase> for Fsp { - const BASE: usize = 0x8f2000; -} +register! { + base: NovaRegisters; -impl RegisterBase<PFalcon2Base> for Fsp { - const BASE: usize = 0x8f3000; + PFALCON: super::PFalconRegisters @ 0x8f2000; + PFALCON2: super::PFalcon2Registers @ 0x8f3000; } -impl FalconEngine for Fsp {} +impl FalconEngine for Fsp { + #[inline] + fn pfalcon(io: Bar0<'_>) -> Mmio<'_, super::PFalconRegisters> { + io_project!(io, build: PFALCON) + } + + #[inline] + fn pfalcon2(io: Bar0<'_>) -> Mmio<'_, super::PFalcon2Registers> { + io_project!(io, build: PFALCON2) + } +} impl<'a> Falcon<'a, Fsp> { /// Writes `data` to FSP external memory at offset `0`. @@ -62,19 +72,15 @@ impl<'a> Falcon<'a, Fsp> { } // Begin a write burst at offset `0`, auto-incrementing on each write. - self.bar.write( - WithBase::of::<Fsp>(), - regs::NV_PFALCON_FALCON_EMEMC::zeroed().with_aincw(true), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_EMEMC::zeroed().with_aincw(true)); for chunk in data.chunks_exact(4) { let value = u32::from_le_bytes([chunk[0], chunk[1], chunk[2], chunk[3]]); // Write the next 32-bit `value`; hardware advances the offset. - self.bar.write( - WithBase::of::<Fsp>(), - regs::NV_PFALCON_FALCON_EMEMD::zeroed().with_data(value), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_EMEMD::zeroed().with_data(value)); } Ok(()) @@ -90,17 +96,12 @@ impl<'a> Falcon<'a, Fsp> { } // Begin a read burst at offset `0`, auto-incrementing on each read. - self.bar.write( - WithBase::of::<Fsp>(), - regs::NV_PFALCON_FALCON_EMEMC::zeroed().with_aincr(true), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_EMEMC::zeroed().with_aincr(true)); for chunk in data.chunks_exact_mut(4) { // Read the next 32-bit word; hardware advances the offset. - let value = self - .bar - .read(regs::NV_PFALCON_FALCON_EMEMD::of::<Fsp>()) - .data(); + let value = self.pfalcon.read(regs::NV_PFALCON_FALCON_EMEMD).data(); chunk.copy_from_slice(&value.to_le_bytes()); } diff --git a/drivers/gpu/nova-core/falcon/gsp.rs b/drivers/gpu/nova-core/falcon/gsp.rs index ae32f401aeb0..4c96ae325fda 100644 --- a/drivers/gpu/nova-core/falcon/gsp.rs +++ b/drivers/gpu/nova-core/falcon/gsp.rs @@ -2,23 +2,24 @@ use kernel::{ io::{ + io_project, poll::read_poll_timeout, - register::{ - RegisterBase, - WithBase, // - }, + register, Io, + Mmio, // }, prelude::*, time::Delta, // }; use crate::{ + driver::{ + Bar0, + NovaRegisters, // + }, falcon::{ Falcon, - FalconEngine, - PFalcon2Base, - PFalconBase, // + FalconEngine, // }, regs, }; @@ -26,24 +27,31 @@ use crate::{ /// Type specifying the `Gsp` falcon engine. Cannot be instantiated. pub(crate) struct Gsp(()); -impl RegisterBase<PFalconBase> for Gsp { - const BASE: usize = 0x00110000; -} +register! { + base: NovaRegisters; -impl RegisterBase<PFalcon2Base> for Gsp { - const BASE: usize = 0x00111000; + PFALCON: super::PFalconRegisters @ 0x00110000; + PFALCON2: super::PFalcon2Registers @ 0x00111000; } -impl FalconEngine for Gsp {} +impl FalconEngine for Gsp { + #[inline] + fn pfalcon(io: Bar0<'_>) -> Mmio<'_, super::PFalconRegisters> { + io_project!(io, build: PFALCON) + } + + #[inline] + fn pfalcon2(io: Bar0<'_>) -> Mmio<'_, super::PFalcon2Registers> { + io_project!(io, build: PFALCON2) + } +} impl<'a> Falcon<'a, Gsp> { /// Clears the SWGEN0 bit in the Falcon's IRQ status clear register to /// allow GSP to signal CPU for processing new messages in message queue. pub(crate) fn clear_swgen0_intr(&self) { - self.bar.write( - WithBase::of::<Gsp>(), - regs::NV_PFALCON_FALCON_IRQSCLR::zeroed().with_swgen0(true), - ); + self.pfalcon + .write_reg(regs::NV_PFALCON_FALCON_IRQSCLR::zeroed().with_swgen0(true)); } /// Checks if GSP reload/resume has completed during the boot process. @@ -59,8 +67,8 @@ impl<'a> Falcon<'a, Gsp> { /// Returns whether the RISC-V branch privilege lockdown bit is set. pub(crate) fn riscv_branch_privilege_lockdown(&self) -> bool { - self.bar - .read(regs::NV_PFALCON_FALCON_HWCFG2::of::<Gsp>()) + self.pfalcon + .read(regs::NV_PFALCON_FALCON_HWCFG2) .riscv_br_priv_lockdown() } @@ -71,10 +79,7 @@ impl<'a> Falcon<'a, Gsp> { const LOCKED_PATTERN: u32 = 0xbadf_4100; const LOCKED_MASK: u32 = 0xffff_ff00; - let hwcfg2 = self - .bar - .read(regs::NV_PFALCON_FALCON_HWCFG2::of::<Gsp>()) - .into_raw(); + let hwcfg2 = self.pfalcon.read(regs::NV_PFALCON_FALCON_HWCFG2).into_raw(); hwcfg2 != 0 && (hwcfg2 & LOCKED_MASK) != LOCKED_PATTERN } diff --git a/drivers/gpu/nova-core/falcon/hal/ga102.rs b/drivers/gpu/nova-core/falcon/hal/ga102.rs index 7600ee07ca2e..f9a8444cf840 100644 --- a/drivers/gpu/nova-core/falcon/hal/ga102.rs +++ b/drivers/gpu/nova-core/falcon/hal/ga102.rs @@ -6,11 +6,9 @@ use kernel::{ device, io::{ poll::read_poll_timeout, - register::{ - Array, - WithBase, // - }, - Io, // + register::Array, + Io, + Mmio, // }, prelude::*, time::Delta, // @@ -24,6 +22,7 @@ use crate::{ FalconBromParams, FalconEngine, FalconModSelAlgo, + PFalcon2Registers, PeregrineCoreSelect, // }, regs, @@ -31,17 +30,16 @@ use crate::{ use super::FalconHal; -fn select_core_ga102<E: FalconEngine>(bar: Bar0<'_>) -> Result { - let bcr_ctrl = bar.read(regs::NV_PRISCV_RISCV_BCR_CTRL::of::<E>()); +fn select_core_ga102(pfalcon2: Mmio<'_, PFalcon2Registers>) -> Result { + let bcr_ctrl = pfalcon2.read(regs::NV_PRISCV_RISCV_BCR_CTRL); if bcr_ctrl.core_select() != PeregrineCoreSelect::Falcon { - bar.write( - WithBase::of::<E>(), + pfalcon2.write_reg( regs::NV_PRISCV_RISCV_BCR_CTRL::zeroed().with_core_select(PeregrineCoreSelect::Falcon), ); // TIMEOUT: falcon core should take less than 10ms to report being enabled. read_poll_timeout( - || Ok(bar.read(regs::NV_PRISCV_RISCV_BCR_CTRL::of::<E>())), + || Ok(pfalcon2.read(regs::NV_PRISCV_RISCV_BCR_CTRL)), |r| r.valid(), Delta::ZERO, Delta::from_millis(10), @@ -86,24 +84,20 @@ fn signature_reg_fuse_version_ga102( Ok(u16::BITS - reg_fuse_version.leading_zeros()) } -fn program_brom_ga102<E: FalconEngine>(bar: Bar0<'_>, params: &FalconBromParams) { - bar.write( - WithBase::of::<E>().at(0), +fn program_brom_ga102(pfalcon2: Mmio<'_, PFalcon2Registers>, params: &FalconBromParams) { + pfalcon2.write( + Array::at(0), regs::NV_PFALCON2_FALCON_BROM_PARAADDR::zeroed().with_value(params.pkc_data_offset), ); - bar.write( - WithBase::of::<E>(), + pfalcon2.write_reg( regs::NV_PFALCON2_FALCON_BROM_ENGIDMASK::zeroed() .with_value(u32::from(params.engine_id_mask)), ); - bar.write( - WithBase::of::<E>(), + pfalcon2.write_reg( regs::NV_PFALCON2_FALCON_BROM_CURR_UCODE_ID::zeroed().with_ucode_id(params.ucode_id), ); - bar.write( - WithBase::of::<E>(), - regs::NV_PFALCON2_FALCON_MOD_SEL::zeroed().with_algo(FalconModSelAlgo::Rsa3k), - ); + pfalcon2 + .write_reg(regs::NV_PFALCON2_FALCON_MOD_SEL::zeroed().with_algo(FalconModSelAlgo::Rsa3k)); } pub(super) struct Ga102<E: FalconEngine>(PhantomData<E>); @@ -116,7 +110,7 @@ impl<E: FalconEngine> Ga102<E> { impl<E: FalconEngine> FalconHal<E> for Ga102<E> { fn select_core(&self, falcon: &Falcon<'_, E>) -> Result { - select_core_ga102::<E>(falcon.bar) + select_core_ga102(falcon.pfalcon2) } fn signature_reg_fuse_version( @@ -129,27 +123,24 @@ impl<E: FalconEngine> FalconHal<E> for Ga102<E> { } fn program_brom(&self, falcon: &Falcon<'_, E>, params: &FalconBromParams) { - program_brom_ga102::<E>(falcon.bar, params); + program_brom_ga102(falcon.pfalcon2, params); } fn is_riscv_active(&self, falcon: &Falcon<'_, E>) -> bool { falcon - .bar - .read(regs::NV_PRISCV_RISCV_CPUCTL::of::<E>()) + .pfalcon2 + .read(regs::NV_PRISCV_RISCV_CPUCTL) .active_stat() } fn is_riscv_halted(&self, falcon: &Falcon<'_, E>) -> Result<bool> { - Ok(falcon - .bar - .read(regs::NV_PRISCV_RISCV_CPUCTL::of::<E>()) - .halted()) + Ok(falcon.pfalcon2.read(regs::NV_PRISCV_RISCV_CPUCTL).halted()) } fn reset_wait_mem_scrubbing(&self, falcon: &Falcon<'_, E>) -> Result { // TIMEOUT: memory scrubbing should complete in less than 20ms. read_poll_timeout( - || Ok(falcon.bar.read(regs::NV_PFALCON_FALCON_HWCFG2::of::<E>())), + || Ok(falcon.pfalcon.read(regs::NV_PFALCON_FALCON_HWCFG2)), |r| r.mem_scrubbing_done(), Delta::ZERO, Delta::from_millis(20), @@ -158,20 +149,18 @@ impl<E: FalconEngine> FalconHal<E> for Ga102<E> { } fn reset_eng(&self, falcon: &Falcon<'_, E>) -> Result { - let bar = falcon.bar; - - let _ = bar.read(regs::NV_PFALCON_FALCON_HWCFG2::of::<E>()); + let _ = falcon.pfalcon.read(regs::NV_PFALCON_FALCON_HWCFG2); // According to OpenRM's `kflcnPreResetWait_GA102` documentation, HW sometimes does not set // RESET_READY so a non-failing timeout is used. let _ = read_poll_timeout( - || Ok(bar.read(regs::NV_PFALCON_FALCON_HWCFG2::of::<E>())), + || Ok(falcon.pfalcon.read(regs::NV_PFALCON_FALCON_HWCFG2)), |r| r.reset_ready(), Delta::ZERO, Delta::from_micros(150), ); - regs::NV_PFALCON_FALCON_ENGINE::reset_engine::<E>(bar); + regs::NV_PFALCON_FALCON_ENGINE::reset_engine(falcon.pfalcon); self.reset_wait_mem_scrubbing(falcon)?; Ok(()) diff --git a/drivers/gpu/nova-core/falcon/hal/tu102.rs b/drivers/gpu/nova-core/falcon/hal/tu102.rs index 5291598fedf7..7fc6e83c2566 100644 --- a/drivers/gpu/nova-core/falcon/hal/tu102.rs +++ b/drivers/gpu/nova-core/falcon/hal/tu102.rs @@ -5,7 +5,6 @@ use core::marker::PhantomData; use kernel::{ io::{ poll::read_poll_timeout, - register::WithBase, Io, // }, prelude::*, @@ -50,8 +49,8 @@ impl<E: FalconEngine> FalconHal<E> for Tu102<E> { fn is_riscv_active(&self, falcon: &Falcon<'_, E>) -> bool { falcon - .bar - .read(regs::NV_PRISCV_RISCV_CORE_SWITCH_RISCV_STATUS::of::<E>()) + .pfalcon2 + .read(regs::NV_PRISCV_RISCV_CORE_SWITCH_RISCV_STATUS) .active_stat() } @@ -62,7 +61,7 @@ impl<E: FalconEngine> FalconHal<E> for Tu102<E> { fn reset_wait_mem_scrubbing(&self, falcon: &Falcon<'_, E>) -> Result { // TIMEOUT: memory scrubbing should complete in less than 10ms. read_poll_timeout( - || Ok(falcon.bar.read(regs::NV_PFALCON_FALCON_DMACTL::of::<E>())), + || Ok(falcon.pfalcon.read(regs::NV_PFALCON_FALCON_DMACTL)), |r| r.mem_scrubbing_done(), Delta::ZERO, Delta::from_millis(10), @@ -71,7 +70,7 @@ impl<E: FalconEngine> FalconHal<E> for Tu102<E> { } fn reset_eng(&self, falcon: &Falcon<'_, E>) -> Result { - regs::NV_PFALCON_FALCON_ENGINE::reset_engine::<E>(falcon.bar); + regs::NV_PFALCON_FALCON_ENGINE::reset_engine(falcon.pfalcon); self.reset_wait_mem_scrubbing(falcon)?; Ok(()) diff --git a/drivers/gpu/nova-core/falcon/sec2.rs b/drivers/gpu/nova-core/falcon/sec2.rs index 91ec7d49c1f5..6648a397d38a 100644 --- a/drivers/gpu/nova-core/falcon/sec2.rs +++ b/drivers/gpu/nova-core/falcon/sec2.rs @@ -1,22 +1,37 @@ // SPDX-License-Identifier: GPL-2.0 -use kernel::io::register::RegisterBase; +use kernel::io::{ + io_project, + register, + Mmio, // +}; -use crate::falcon::{ - FalconEngine, - PFalcon2Base, - PFalconBase, // +use crate::{ + driver::{ + Bar0, + NovaRegisters, // + }, + falcon::FalconEngine, // }; /// Type specifying the `Sec2` falcon engine. Cannot be instantiated. pub(crate) struct Sec2(()); -impl RegisterBase<PFalconBase> for Sec2 { - const BASE: usize = 0x00840000; -} +register! { + base: NovaRegisters; -impl RegisterBase<PFalcon2Base> for Sec2 { - const BASE: usize = 0x00841000; + PFALCON: super::PFalconRegisters @ 0x00840000; + PFALCON2: super::PFalcon2Registers @ 0x00841000; } -impl FalconEngine for Sec2 {} +impl FalconEngine for Sec2 { + #[inline] + fn pfalcon(io: Bar0<'_>) -> Mmio<'_, super::PFalconRegisters> { + io_project!(io, build: PFALCON) + } + + #[inline] + fn pfalcon2(io: Bar0<'_>) -> Mmio<'_, super::PFalcon2Registers> { + io_project!(io, build: PFALCON2) + } +} diff --git a/drivers/gpu/nova-core/fb.rs b/drivers/gpu/nova-core/fb.rs index 1576399389b1..b3a6ab8b57a6 100644 --- a/drivers/gpu/nova-core/fb.rs +++ b/drivers/gpu/nova-core/fb.rs @@ -49,7 +49,7 @@ pub(crate) struct SysmemFlush<'sys> { device: &'sys device::Device, bar: Bar0<'sys>, /// Keep the page alive as long as we need it. - page: CoherentHandle, + page: CoherentHandle<'sys>, } impl<'sys> SysmemFlush<'sys> { @@ -177,7 +177,7 @@ impl FbRanges { pub(crate) fn new( chipset: Chipset, bar: Bar0<'_>, - gsp_fw: &GspFirmware, + gsp_fw: &GspFirmware<'_>, vgpu_state: VgpuState, ) -> Result<Self> { let hal = hal::fb_hal(chipset); diff --git a/drivers/gpu/nova-core/fb/hal/gb100.rs b/drivers/gpu/nova-core/fb/hal/gb100.rs index d9e4d62ae632..9fa094939600 100644 --- a/drivers/gpu/nova-core/fb/hal/gb100.rs +++ b/drivers/gpu/nova-core/fb/hal/gb100.rs @@ -5,11 +5,10 @@ use kernel::{ io::{ - register::{ - RegisterBase, - WithBase, // - }, - Io, // + io_project, + register, + Io, + Mmio, // }, num::Bounded, prelude::*, @@ -21,7 +20,10 @@ use kernel::{ }; use crate::{ - driver::Bar0, + driver::{ + Bar0, + NovaRegisters, // + }, fb::{ hal::FbHal, regs, // @@ -31,17 +33,26 @@ use crate::{ struct Gb100; -impl RegisterBase<regs::Hshub0Base> for Gb100 { - const BASE: usize = 0x0087_0000; +register! { + base: NovaRegisters; + + HSHUB0: regs::Hshub0Registers @ 0x0087_0000; +} + +#[inline] +fn hshub0(bar: Bar0<'_>) -> Mmio<'_, regs::Hshub0Registers> { + io_project!(bar, build: HSHUB0) } -fn read_sysmem_flush_page_gb100(bar: Bar0<'_>) -> u64 { +fn read_sysmem_flush_page_gb100(hshub0: Mmio<'_, regs::Hshub0Registers>) -> u64 { let lo = u64::from( - bar.read(regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO::of::<Gb100>()) + hshub0 + .read(regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO) .adr(), ); let hi = u64::from( - bar.read(regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI::of::<Gb100>()) + hshub0 + .read(regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI) .adr(), ); @@ -52,7 +63,7 @@ fn read_sysmem_flush_page_gb100(bar: Bar0<'_>) -> u64 { /// /// Both the primary and EG (egress) register pairs must be programmed to the same address, /// as required by hardware. -fn write_sysmem_flush_page_gb100(bar: Bar0<'_>, addr: Bounded<u64, 52>) { +fn write_sysmem_flush_page_gb100(hshub0: Mmio<'_, regs::Hshub0Registers>, addr: Bounded<u64, 52>) { // CAST: lower 32 bits. Hardware ignores bits 7:0. let addr_lo = *addr as u32; let addr_hi = addr.shr::<32, 20>().cast::<u32>(); @@ -60,24 +71,12 @@ fn write_sysmem_flush_page_gb100(bar: Bar0<'_>, addr: Bounded<u64, 52>) { // Write HI first. The hardware will trigger the flush on the LO write. // Primary HSHUB pair. - bar.write( - regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI::of::<Gb100>(), - regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI::zeroed().with_adr(addr_hi), - ); - bar.write( - regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO::of::<Gb100>(), - regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO::zeroed().with_adr(addr_lo), - ); + hshub0.write_reg(regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI::zeroed().with_adr(addr_hi)); + hshub0.write_reg(regs::NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO::zeroed().with_adr(addr_lo)); // EG (egress) pair -- must match the primary pair. - bar.write( - regs::NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_HI::of::<Gb100>(), - regs::NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_HI::zeroed().with_adr(addr_hi), - ); - bar.write( - regs::NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_LO::of::<Gb100>(), - regs::NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_LO::zeroed().with_adr(addr_lo), - ); + hshub0.write_reg(regs::NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_HI::zeroed().with_adr(addr_hi)); + hshub0.write_reg(regs::NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_LO::zeroed().with_adr(addr_lo)); } // This PMU reservation size is r570-specific. @@ -88,13 +87,13 @@ pub(super) const fn pmu_reserved_size_gb100() -> u32 { impl FbHal for Gb100 { fn read_sysmem_flush_page(&self, bar: Bar0<'_>) -> u64 { - read_sysmem_flush_page_gb100(bar) + read_sysmem_flush_page_gb100(hshub0(bar)) } fn write_sysmem_flush_page(&self, bar: Bar0<'_>, addr: u64) -> Result { let addr = Bounded::<u64, 52>::try_new(addr).ok_or(EINVAL)?; - write_sysmem_flush_page_gb100(bar, addr); + write_sysmem_flush_page_gb100(hshub0(bar), addr); Ok(()) } diff --git a/drivers/gpu/nova-core/fb/regs.rs b/drivers/gpu/nova-core/fb/regs.rs index 95adbe124a30..131787996a24 100644 --- a/drivers/gpu/nova-core/fb/regs.rs +++ b/drivers/gpu/nova-core/fb/regs.rs @@ -2,12 +2,20 @@ use kernel::{ io::register, - sizes::SizeConstants, // + prelude::*, + sizes::{ + SizeConstants, + SZ_4K, // + }, // }; +use crate::driver::NovaRegisters; + // PDISP register! { + base: NovaRegisters; + pub(super) NV_PDISP_VGA_WORKSPACE_BASE(u32) @ 0x00625f04 { /// VGA workspace base address divided by 0x10000. 31:8 addr; @@ -30,6 +38,8 @@ impl NV_PDISP_VGA_WORKSPACE_BASE { // PFB register! { + base: NovaRegisters; + /// Low bits of the physical system memory address used by the GPU to perform sysmembar /// operations (see [`crate::fb::SysmemFlush`]). pub(super) NV_PFB_NISO_FLUSH_SYSMEM_ADDR(u32) @ 0x00100c10 { @@ -59,34 +69,42 @@ register! { } } -/// Base of the GB10x HSHUB0 register window (`NV_HSHUB0_PRIV_BASE` in Open RM). +const HSHUB0_REGION_SIZE: usize = SZ_4K; + +/// The GB10x HSHUB0 register window (Base defined as `NV_HSHUB0_PRIV_BASE` in Open RM). /// /// The base is provided by the GB10x framebuffer HAL. -pub(super) struct Hshub0Base(()); +#[repr(align(4))] +#[derive(FromBytes, IntoBytes)] +pub(super) struct Hshub0Registers([u8; HSHUB0_REGION_SIZE]); register! { + base: Hshub0Registers; + // GB10x sysmem flush registers, relative to the HSHUB0 base. GB10x routes sysmembar // through a primary and an EG (egress) pair that must both be programmed to the same // address. Hardware ignores bits 7:0 of each LO register. The boot path uses a fixed // HSHUB0 base, so the multiple runtime-discovered HSHUB bases are not needed here. - pub(super) NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO(u32) @ Hshub0Base + 0x00000e50 { + pub(super) NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_LO(u32) @ 0x00000e50 { 31:0 adr => u32; } - pub(super) NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI(u32) @ Hshub0Base + 0x00000e54 { + pub(super) NV_PFB_HSHUB_PCIE_FLUSH_SYSMEM_ADDR_HI(u32) @ 0x00000e54 { 19:0 adr; } - pub(super) NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_LO(u32) @ Hshub0Base + 0x000006c0 { + pub(super) NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_LO(u32) @ 0x000006c0 { 31:0 adr => u32; } - pub(super) NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_HI(u32) @ Hshub0Base + 0x000006c4 { + pub(super) NV_PFB_HSHUB_EG_PCIE_FLUSH_SYSMEM_ADDR_HI(u32) @ 0x000006c4 { 19:0 adr; } } register! { + base: NovaRegisters; + // GB20x FBHUB0 sysmem flush registers. Unlike the older // NV_PFB_NISO_FLUSH_SYSMEM_ADDR registers, which encode the address with an // 8-bit right-shift, these take the raw address split into lower and upper @@ -101,6 +119,8 @@ register! { } register! { + base: NovaRegisters; + /// Low bits of the physical system memory address used by the GPU to perform /// sysmembar operations on Hopper. /// diff --git a/drivers/gpu/nova-core/firmware.rs b/drivers/gpu/nova-core/firmware.rs index b49613a90bf0..c16fee6e2b2a 100644 --- a/drivers/gpu/nova-core/firmware.rs +++ b/drivers/gpu/nova-core/firmware.rs @@ -23,9 +23,9 @@ use crate::{ }; pub(crate) mod booter; -pub(crate) mod fsp; pub(crate) mod fwsec; pub(crate) mod gsp; +pub(crate) mod gsp_fmc; pub(crate) mod riscv; pub(crate) mod tlv; diff --git a/drivers/gpu/nova-core/firmware/booter.rs b/drivers/gpu/nova-core/firmware/booter.rs index dc071edba331..aa4458bb3312 100644 --- a/drivers/gpu/nova-core/firmware/booter.rs +++ b/drivers/gpu/nova-core/firmware/booter.rs @@ -186,7 +186,7 @@ impl BooterFirmware { &self, dev: &device::Device<device::Bound>, sec2_falcon: &Falcon<'_, Sec2>, - wpr_meta: &Coherent<T>, + wpr_meta: &Coherent<'_, T>, ) -> Result { sec2_falcon.reset()?; sec2_falcon.load(self)?; diff --git a/drivers/gpu/nova-core/firmware/fwsec/bootloader.rs b/drivers/gpu/nova-core/firmware/fwsec/bootloader.rs index ec4d92317a93..a87878fe2aec 100644 --- a/drivers/gpu/nova-core/firmware/fwsec/bootloader.rs +++ b/drivers/gpu/nova-core/firmware/fwsec/bootloader.rs @@ -12,7 +12,10 @@ use kernel::{ Device, // }, dma::Coherent, - io::{register::WithBase, Io}, + io::{ + register::Array, + Io, // + }, prelude::*, ptr::{ Alignable, @@ -23,7 +26,6 @@ use kernel::{ }; use crate::{ - driver::Bar0, falcon::{ self, gsp::Gsp, @@ -98,9 +100,9 @@ unsafe impl AsBytes for BootloaderDmemDescV2 {} /// Wrapper for [`FwsecFirmware`] that includes the bootloader performing the actual load /// operation. -pub(crate) struct FwsecFirmwareWithBl { +pub(crate) struct FwsecFirmwareWithBl<'a> { /// DMA object the bootloader will copy the firmware from. - _firmware_dma: Coherent<[u8]>, + _firmware_dma: Coherent<'a, [u8]>, /// Code of the bootloader to be loaded into non-secure IMEM. ucode: KVec<u8>, /// Descriptor to be loaded into DMEM for the bootloader to read. @@ -113,12 +115,12 @@ pub(crate) struct FwsecFirmwareWithBl { start_tag: u16, } -impl FwsecFirmwareWithBl { +impl<'a> FwsecFirmwareWithBl<'a> { /// Loads the bootloader firmware for `dev` and `chipset`, and wrap `firmware` so it can be /// loaded using it. pub(crate) fn new( firmware: FwsecFirmware, - dev: &Device<device::Bound>, + dev: &'a Device<device::Bound>, chipset: Chipset, ) -> Result<Self> { let fw = request_tlv(dev, chipset, "gen_bootloader")?; @@ -235,12 +237,7 @@ impl FwsecFirmwareWithBl { /// /// The bootloader will load the FWSEC firmware and then execute it. This function returns /// after FWSEC has reached completion. - pub(crate) fn run( - &self, - dev: &Device<device::Bound>, - falcon: &Falcon<'_, Gsp>, - bar: Bar0<'_>, - ) -> Result<()> { + pub(crate) fn run(&self, dev: &Device<device::Bound>, falcon: &Falcon<'_, Gsp>) -> Result<()> { // Reset falcon, load the firmware, and run it. falcon .reset() @@ -250,9 +247,8 @@ impl FwsecFirmwareWithBl { .inspect_err(|e| dev_err!(dev, "Failed to load FWSEC firmware: {:?}\n", e))?; // Configure DMA index for the bootloader to fetch the FWSEC firmware from system memory. - bar.update( - regs::NV_PFALCON_FBIF_TRANSCFG::of::<Gsp>() - .try_at(usize::from_safe_cast(self.dmem_desc.ctx_dma)) + falcon.pfalcon.update( + regs::NV_PFALCON_FBIF_TRANSCFG::try_at(usize::from_safe_cast(self.dmem_desc.ctx_dma)) .ok_or(EINVAL)?, |v| { v.with_target(FalconFbifTarget::CoherentSysmem) @@ -272,7 +268,7 @@ impl FwsecFirmwareWithBl { } } -impl FalconFirmware for FwsecFirmwareWithBl { +impl FalconFirmware for FwsecFirmwareWithBl<'_> { type Target = Gsp; fn brom_params(&self) -> FalconBromParams { @@ -286,7 +282,7 @@ impl FalconFirmware for FwsecFirmwareWithBl { } } -impl FalconPioLoadable for FwsecFirmwareWithBl { +impl FalconPioLoadable for FwsecFirmwareWithBl<'_> { fn imem_sec_load_params(&self) -> Option<FalconPioImemLoadTarget<'_>> { None } diff --git a/drivers/gpu/nova-core/firmware/gsp.rs b/drivers/gpu/nova-core/firmware/gsp.rs index e8f9491e84cc..22d1f9329c9f 100644 --- a/drivers/gpu/nova-core/firmware/gsp.rs +++ b/drivers/gpu/nova-core/firmware/gsp.rs @@ -44,7 +44,7 @@ use crate::{ /// Each page is 4KB, each entry is 8 bytes (64-bit DMA address). /// Also known as "Radix3" firmware. #[pin_data] -pub(crate) struct GspFirmware { +pub(crate) struct GspFirmware<'a> { /// The GSP firmware inside a [`VVec`], device-mapped via a SG table. #[pin] fw: SGTable<Owned<VVec<u8>>>, @@ -55,19 +55,19 @@ pub(crate) struct GspFirmware { #[pin] level1: SGTable<Owned<VVec<u8>>>, /// Level 0 page table (single 4KB page) with one entry: DMA address of first level 1 page. - level0: Coherent<[u64]>, + level0: Coherent<'a, [u64]>, /// Size in bytes of the firmware contained in [`Self::fw`]. pub(crate) size: usize, /// Device-mapped GSP signatures matching the GPU's [`Chipset`]. - pub(crate) signatures: Coherent<[u8]>, + pub(crate) signatures: Coherent<'a, [u8]>, /// GSP bootloader, verifies the GSP firmware before loading and running it. - pub(crate) bootloader: RiscvFirmware, + pub(crate) bootloader: RiscvFirmware<'a>, } -impl GspFirmware { +impl<'a> GspFirmware<'a> { /// Loads the GSP firmware binaries, map them into `dev`'s address-space, and creates the page /// tables expected by the GSP bootloader to load it. - pub(crate) fn new<'a>( + pub(crate) fn new( dev: &'a device::Device<device::Bound>, chipset: Chipset, ) -> impl PinInit<Self, Error> + 'a { @@ -120,7 +120,7 @@ impl GspFirmware { // Create level 0 page table data and fill its first entry with the level 1 // table. - let mut level0 = CoherentBox::<[u64]>::zeroed_slice( + let mut level0 = CoherentBox::<'_, [u64]>::zeroed_slice( dev, GSP_PAGE_SIZE / size_of::<u64>(), GFP_KERNEL diff --git a/drivers/gpu/nova-core/firmware/fsp.rs b/drivers/gpu/nova-core/firmware/gsp_fmc.rs index 5462e318410a..94fcc86c7dff 100644 --- a/drivers/gpu/nova-core/firmware/fsp.rs +++ b/drivers/gpu/nova-core/firmware/gsp_fmc.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: GPL-2.0 // SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -//! FSP is a hardware unit that runs FMC firmware. +//! GSP-FMC (First Mutable Code) is loaded by FSP into GSP to serve as the loader and verifier of +//! GSP-RM. use kernel::{ device, @@ -17,16 +18,16 @@ use crate::{ gpu::Chipset, // }; -/// Size of the FSP SHA-384 hash, in bytes. -const FSP_HASH_SIZE: usize = 48; -/// Maximum size of the FSP public key (RSA-3072), in bytes. +/// Size of the GSP-FMC SHA-384 hash, in bytes. +const FMC_HASH_SIZE: usize = 48; +/// Maximum size of the GSP-FMC public key (RSA-3072), in bytes. /// -/// The FMC `PKEY` tag may be shorter, so the remaining bytes are zero-padded. -const FSP_PKEY_SIZE: usize = 384; -/// Maximum size of the FSP signature (RSA-3072), in bytes. +/// The `PKEY` tag may be shorter, so the remaining bytes are zero-padded. +const FMC_PKEY_SIZE: usize = 384; +/// Maximum size of the GSP-FMC signature (RSA-3072), in bytes. /// -/// The FMC `SIGN` tag may be shorter, so the remaining bytes are zero-padded. -const FSP_SIG_SIZE: usize = 384; +/// The `SIGN` tag may be shorter, so the remaining bytes are zero-padded. +const FMC_SIG_SIZE: usize = 384; /// Structure to hold FMC signatures. /// @@ -34,23 +35,27 @@ const FSP_SIG_SIZE: usize = 384; #[derive(Debug, Clone, Copy, Zeroable)] #[repr(C)] pub(crate) struct FmcSignatures { - pub(crate) hash384: [u8; FSP_HASH_SIZE], - pub(crate) public_key: [u8; FSP_PKEY_SIZE], - pub(crate) signature: [u8; FSP_SIG_SIZE], + pub(crate) hash384: [u8; FMC_HASH_SIZE], + pub(crate) public_key: [u8; FMC_PKEY_SIZE], + pub(crate) signature: [u8; FMC_SIG_SIZE], } -pub(crate) struct FspFirmware { +pub(crate) struct GspFmcFirmware<'a> { /// FMC firmware image data - pub(crate) fmc_image: Coherent<[u8]>, + pub(crate) fmc_image: Coherent<'a, [u8]>, /// FMC firmware signatures. pub(crate) fmc_sigs: KBox<FmcSignatures>, } -impl FspFirmware { - pub(crate) fn new(dev: &device::Device<device::Bound>, chipset: Chipset) -> Result<Self> { +impl<'a> GspFmcFirmware<'a> { + pub(crate) fn new(dev: &'a device::Device<device::Bound>, chipset: Chipset) -> Result<Self> { let fw = request_tlv(dev, chipset, "fmc")?; let tlv = Tlv::new(fw.data())?; - dev_dbg!(dev, "loaded fsp firmware v{}\n", tlv.get_string(b"VERS")?); + dev_dbg!( + dev, + "loaded GSP-FMC firmware v{}\n", + tlv.get_string(b"VERS")? + ); let fmc_image_data = tlv.get_bytes(b"BLOB")?; let fmc_image = Coherent::from_slice(dev, fmc_image_data, GFP_KERNEL)?; @@ -70,34 +75,34 @@ impl FspFirmware { let pkey_section = tlv.get_bytes(b"PKEY")?; let sig_section = tlv.get_bytes(b"SIGN")?; - // The hash section is a SHA-384 output: it must be exactly FSP_HASH_SIZE bytes. - if hash_section.len() != FSP_HASH_SIZE { + // The hash section is a SHA-384 output: it must be exactly `FMC_HASH_SIZE` bytes. + if hash_section.len() != FMC_HASH_SIZE { dev_err!( dev, "FMC hash section size {} != expected {}\n", hash_section.len(), - FSP_HASH_SIZE + FMC_HASH_SIZE ); return Err(EINVAL); } // The key and signature sections are zero-padded to a fixed maximum, so they may be // shorter, but must not exceed the destination buffers. - if pkey_section.len() > FSP_PKEY_SIZE { + if pkey_section.len() > FMC_PKEY_SIZE { dev_err!( dev, "FMC public key section size {} > maximum {}\n", pkey_section.len(), - FSP_PKEY_SIZE + FMC_PKEY_SIZE ); return Err(EINVAL); } - if sig_section.len() > FSP_SIG_SIZE { + if sig_section.len() > FMC_SIG_SIZE { dev_err!( dev, "FMC signature section size {} > maximum {}\n", sig_section.len(), - FSP_SIG_SIZE + FMC_SIG_SIZE ); return Err(EINVAL); } @@ -106,11 +111,11 @@ impl FspFirmware { // stack, then fill each section from the firmware. let signatures = KBox::init( pin_init::init_zeroed::<FmcSignatures>().chain(|sigs| { - // PANIC: src and dst lengths are both FSP_HASH_SIZE (verified above). + // PANIC: src and dst lengths are both `FMC_HASH_SIZE` (verified above). sigs.hash384.copy_from_slice(hash_section); - // PANIC: dst is sliced to src.len(); src.len() <= FSP_PKEY_SIZE (verified above). + // PANIC: dst is sliced to src.len(); src.len() <= `FMC_PKEY_SIZE` (verified above). sigs.public_key[..pkey_section.len()].copy_from_slice(pkey_section); - // PANIC: dst is sliced to src.len(); src.len() <= FSP_SIG_SIZE (verified above). + // PANIC: dst is sliced to src.len(); src.len() <= `FMC_SIG_SIZE` (verified above). sigs.signature[..sig_section.len()].copy_from_slice(sig_section); Ok(()) }), diff --git a/drivers/gpu/nova-core/firmware/riscv.rs b/drivers/gpu/nova-core/firmware/riscv.rs index 1403f05a7305..f05cfb1c65da 100644 --- a/drivers/gpu/nova-core/firmware/riscv.rs +++ b/drivers/gpu/nova-core/firmware/riscv.rs @@ -13,7 +13,7 @@ use kernel::{ use crate::firmware::tlv::Tlv; /// A parsed firmware for a RISC-V core, ready to be loaded and run. -pub(crate) struct RiscvFirmware { +pub(crate) struct RiscvFirmware<'a> { /// Offset at which the code starts in the firmware image. pub(crate) code_offset: u32, /// Offset at which the data starts in the firmware image. @@ -23,12 +23,12 @@ pub(crate) struct RiscvFirmware { /// Application version. pub(crate) app_version: u32, /// Device-mapped firmware image. - pub(crate) ucode: Coherent<[u8]>, + pub(crate) ucode: Coherent<'a, [u8]>, } -impl RiscvFirmware { +impl<'a> RiscvFirmware<'a> { /// Parses the RISC-V firmware image contained in `fw`. - pub(crate) fn new(dev: &device::Device<device::Bound>, fw: &Firmware) -> Result<Self> { + pub(crate) fn new(dev: &'a device::Device<device::Bound>, fw: &Firmware) -> Result<Self> { let tlv = Tlv::new(fw.data())?; dev_dbg!( dev, diff --git a/drivers/gpu/nova-core/fsp.rs b/drivers/gpu/nova-core/fsp.rs index ab685fb4168f..b738dcabcdef 100644 --- a/drivers/gpu/nova-core/fsp.rs +++ b/drivers/gpu/nova-core/fsp.rs @@ -3,9 +3,12 @@ //! FSP (Foundation Security Processor) interface for Hopper/Blackwell GPUs. //! -//! Hopper/Blackwell use a simplified firmware boot sequence: FMC, then FSP, then GSP. +//! Hopper/Blackwell use a simplified firmware boot sequence: FSP secure-boots independently before +//! the driver starts. The driver then sends FSP a Chain-of-Trust request containing the GSP-FMC +//! image. FSP authenticates the image and launches GSP-FMC on the GSP RISC-V core; GSP-FMC +//! subsequently authenticates and boots GSP-RM. +//! //! Unlike Turing/Ampere/Ada, there is no SEC2 (Security Engine 2) usage. -//! FSP handles secure boot directly using FMC firmware and Chain of Trust. use kernel::{ device, @@ -32,9 +35,9 @@ use crate::{ Falcon, // }, fb::FbSizes, - firmware::fsp::{ + firmware::gsp_fmc::{ FmcSignatures, - FspFirmware, // + GspFmcFirmware, // }, gpu::Chipset, gsp::{ @@ -267,7 +270,7 @@ impl FspCotMessage { /// Returns an in-place initializer for [`FspCotMessage`]. fn new<'a>( fb_info: &FbSizes, - fsp_fw: &'a FspFirmware, + fmc_fw: &'a GspFmcFirmware<'_>, args: &'a FmcBootArgs<'_>, ) -> Result<impl Init<Self> + 'a> { let hal = hal::fsp_hal(args.chipset).ok_or(ENOTSUPP)?; @@ -296,13 +299,13 @@ impl FspCotMessage { .chain(move |msg| { msg.cot.version = version; msg.cot.size = size; - msg.cot.gsp_fmc_sysmem_offset = fsp_fw.fmc_image.dma_address(); + msg.cot.gsp_fmc_sysmem_offset = fmc_fw.fmc_image.dma_address(); msg.cot.frts_vidmem_offset = frts_vidmem_offset; msg.cot.frts_vidmem_size = frts_size; // frts_sysmem_* are left at zero because this path places FRTS in vidmem. The sysmem // fields point to an FRTS buffer in sysmem instead, for systems without VRAM. msg.cot.gsp_boot_args_sysmem_offset = args.fmc_boot_params.dma_address(); - msg.cot.sigs = *fsp_fw.fmc_sigs; + msg.cot.sigs = *fmc_fw.fmc_sigs; Ok(()) })) @@ -345,28 +348,28 @@ impl MessageToFsp for FspPrcMessage { /// Bundled arguments for FMC boot via FSP Chain of Trust. pub(crate) struct FmcBootArgs<'a> { chipset: Chipset, - fmc_boot_params: Coherent<GspFmcBootParams>, + fmc_boot_params: Coherent<'a, GspFmcBootParams>, resume: bool, // Additional dependencies required to be kept alive for FMC boot. - _wpr_meta: Coherent<GspFwWprMeta>, - _libos: &'a Coherent<[LibosMemoryRegionInitArgument]>, + _wpr_meta: Coherent<'a, GspFwWprMeta>, + _libos: &'a Coherent<'a, [LibosMemoryRegionInitArgument]>, } impl<'a> FmcBootArgs<'a> { /// Builds FMC boot arguments, allocating the DMA-coherent boot parameter /// structure that FSP will read. pub(crate) fn new( - dev: &device::Device<device::Bound>, + dev: &'a device::Device<device::Bound>, chipset: Chipset, - wpr_meta: Coherent<GspFwWprMeta>, - libos: &'a Coherent<[LibosMemoryRegionInitArgument]>, + wpr_meta: Coherent<'a, GspFwWprMeta>, + libos: &'a Coherent<'a, [LibosMemoryRegionInitArgument]>, resume: bool, ) -> Result<Self> { let init = GspFmcBootParams::new(wpr_meta.dma_address(), libos.dma_address()); Ok(Self { chipset, - fmc_boot_params: Coherent::<GspFmcBootParams>::init(dev, GFP_KERNEL, init)?, + fmc_boot_params: Coherent::init(dev, GFP_KERNEL, init)?, resume, _wpr_meta: wpr_meta, _libos: libos, @@ -374,19 +377,19 @@ impl<'a> FmcBootArgs<'a> { } /// Returns the FMC boot parameters allocation. - pub(crate) fn boot_params(&self) -> &Coherent<GspFmcBootParams> { + pub(crate) fn boot_params(&self) -> &Coherent<'_, GspFmcBootParams> { &self.fmc_boot_params } } /// FSP interface for Hopper/Blackwell GPUs. /// -/// An `Fsp` is produced by [`Fsp::wait_secure_boot`], which only returns once FSP secure boot -/// has completed. It owns the FSP falcon and the FMC firmware, which are used for the subsequent +/// An `Fsp` is produced by [`Fsp::wait_secure_boot`], which only returns once FSP secure boot has +/// completed. It owns the FSP falcon and the GSP-FMC firmware, which are used for the subsequent /// Chain of Trust boot. pub(crate) struct Fsp<'a> { falcon: Falcon<'a, FspEngine>, - fsp_fw: FspFirmware, + fmc_fw: GspFmcFirmware<'a>, } impl<'a> Fsp<'a> { @@ -422,7 +425,7 @@ impl<'a> Fsp<'a> { const FSP_SECURE_BOOT_TIMEOUT_MS: i64 = 5000; let falcon = Falcon::<FspEngine>::new(dev, chipset, bar)?; - let fsp_fw = FspFirmware::new(dev, chipset)?; + let fmc_fw = GspFmcFirmware::new(dev, chipset)?; read_poll_timeout( || Ok(hal.fsp_boot_status(bar)), @@ -434,7 +437,7 @@ impl<'a> Fsp<'a> { dev_err!(dev, "FSP secure boot completion error: {:?}\n", e); })?; - Ok(Fsp { falcon, fsp_fw }) + Ok(Fsp { falcon, fmc_fw }) } /// Sends a message to FSP and waits for the response. @@ -540,7 +543,7 @@ impl<'a> Fsp<'a> { ) -> Result { dev_dbg!(dev, "Starting FSP boot sequence for {}\n", args.chipset); - let msg = KBox::init(FspCotMessage::new(fb_info, &self.fsp_fw, args)?, GFP_KERNEL)?; + let msg = KBox::init(FspCotMessage::new(fb_info, &self.fmc_fw, args)?, GFP_KERNEL)?; let _response_buf = self.send_sync_fsp(dev, &*msg)?; diff --git a/drivers/gpu/nova-core/gpu.rs b/drivers/gpu/nova-core/gpu.rs index fd1414004dd0..d763bc8d3827 100644 --- a/drivers/gpu/nova-core/gpu.rs +++ b/drivers/gpu/nova-core/gpu.rs @@ -6,16 +6,25 @@ use kernel::{ device, dma::Device, fmt, + gpu::buddy::GpuBuddyParams, io::Io, num::Bounded, pci, prelude::*, - sizes::SizeConstants, // + ptr::Alignment, + sizes::{ + SizeConstants, + SZ_4K, // + }, + sync::Arc, }; use crate::{ bounded_enum, - driver::Bar0, + driver::{ + Bar0, + Bar1, // + }, falcon::{ gsp::Gsp as GspFalcon, sec2::Sec2 as Sec2Falcon, @@ -29,11 +38,17 @@ use crate::{ Gsp, GspBootContext, // }, - regs, + mm::{ + bar_user::BarUser, + pagetable::MmuVersion, + GpuMm, + VramAddress, // + }, vgpu::VgpuManager, // }; mod hal; +mod regs; macro_rules! define_chipset { ({ $($variant:ident = $value:expr),* $(,)* }) => @@ -139,6 +154,11 @@ impl Chipset { pub(crate) fn pci_config_mirror_range(self) -> Range<u32> { hal::gpu_hal(self).pci_config_mirror_range() } + + /// Returns the MMU version for this chipset. + pub(crate) fn mmu_version(self) -> MmuVersion { + MmuVersion::from(self.arch()) + } } // TODO @@ -272,9 +292,9 @@ struct GspResources<'gpu> { vgpu: VgpuManager, /// GSP runtime data. #[pin] - gsp: Gsp, + gsp: Gsp<'gpu>, /// GSP unload firmware bundle, if any. - unload_bundle: Option<gsp::UnloadBundle>, + unload_bundle: Option<gsp::UnloadBundle<'gpu>>, } /// Structure holding the resources required to operate the GPU. @@ -283,6 +303,13 @@ pub(crate) struct Gpu<'gpu> { spec: Spec, /// Static GPU information as provided by the GSP. gsp_static_info: GetGspStaticInfoReply, + /// GPU memory manager owning memory management resources. + /// + /// Must be kept declared *before* `gsp_resources`, so that its components are dropped while + /// the GSP is still operational. + mm: GpuMm<'gpu>, + /// BAR1 user interface for CPU access to GPU virtual memory. + bar_user: Arc<BarUser<'gpu>>, /// GSP and its resources. #[pin] gsp_resources: GspResources<'gpu>, @@ -326,6 +353,7 @@ impl<'gpu> Gpu<'gpu> { pub(crate) fn new<'a>( pdev: &'gpu pci::Device<device::Core<'a>>, bar: Bar0<'gpu>, + bar1: &'gpu Bar1<'gpu>, ) -> impl PinInit<Self, Error> + use<'gpu, 'a> { let dev = pdev.as_ref(); @@ -410,7 +438,64 @@ impl<'gpu> Gpu<'gpu> { } info - } + }, + + // Create GPU memory manager owning memory management resources. + mm: { + let usable_vram = gsp_static_info.usable_fb_regions.first().ok_or(ENODEV)?; + let buddy_params = GpuBuddyParams { + base_offset: usable_vram.start, + size: usable_vram.end - usable_vram.start, + chunk_size: Alignment::new::<SZ_4K>(), + }; + + GpuMm::new( + bar, + gsp_resources.spec.chipset, + buddy_params, + VramAddress::from_raw(gsp_static_info.total_fb_end), + )? + }, + + // Create BAR1 user interface for CPU access to GPU virtual memory. + bar_user: { + let pdb_addr = VramAddress::from_raw(gsp_static_info.bar1_pde_base); + let bar1_idx = crate::driver::bar1_resource_index(pdev)?; + let bar1_size = pdev.resource_len(bar1_idx)?; + Arc::pin_init( + BarUser::new( + pdb_addr, + gsp_resources.spec.chipset, + bar1_size, + bar1, + )?, + GFP_KERNEL, + )? + }, }) } + + /// Runs self-tests on the constructed [`Gpu`], logging failures without failing probe. + #[cfg(CONFIG_NOVA_CORE_SELFTESTS)] + pub(crate) fn run_selftests(self: Pin<&mut Self>, pdev: &pci::Device<device::Bound>) { + let this = self.project(); + let dev = pdev.as_ref(); + let regions = &this.gsp_static_info.usable_fb_regions; + + if let Err(err) = crate::mm::selftest::run( + dev, + this.mm, + regions, + this.bar_user, + this.gsp_static_info.bar1_pde_base, + this.spec.chipset, + ) { + dev_err!(dev, "self-tests failed: {:?}\n", err); + } + } +} + +/// Reads the boot0 register and returns its raw value. +pub(crate) fn boot_0_raw(bar: Bar0<'_>) -> u32 { + bar.read(regs::NV_PMC_BOOT_0).into_raw() } diff --git a/drivers/gpu/nova-core/gpu/regs.rs b/drivers/gpu/nova-core/gpu/regs.rs new file mode 100644 index 000000000000..54e740d847cd --- /dev/null +++ b/drivers/gpu/nova-core/gpu/regs.rs @@ -0,0 +1,86 @@ +// SPDX-License-Identifier: GPL-2.0 + +use kernel::{ + io::register, + prelude::*, // +}; + +use super::{ + Architecture, + Chipset, // +}; + +use crate::driver::NovaRegisters; + +// PMC + +register! { + base: NovaRegisters; + + /// Basic revision information about the GPU. + pub(super) NV_PMC_BOOT_0(u32) @ 0x00000000 { + /// Lower bits of the architecture. + 28:24 architecture_0; + /// Implementation version of the architecture. + 23:20 implementation; + /// MSB of the architecture. + 8:8 architecture_1; + /// Major revision of the chip. + 7:4 major_revision; + /// Minor revision of the chip. + 3:0 minor_revision; + } + + /// Extended architecture information. + pub(super) NV_PMC_BOOT_42(u32) @ 0x00000a00 { + /// Architecture value. + 29:24 architecture ?=> Architecture; + /// Implementation version of the architecture. + 23:20 implementation; + /// Major revision of the chip. + 19:16 major_revision; + /// Minor revision of the chip. + 15:12 minor_revision; + } +} + +impl NV_PMC_BOOT_0 { + pub(super) fn is_older_than_fermi(self) -> bool { + // From https://github.com/NVIDIA/open-gpu-doc/tree/master/manuals : + const NV_PMC_BOOT_0_ARCHITECTURE_GF100: u32 = 0xc; + + // Older chips left arch1 zeroed out. That, combined with an arch0 value that is less than + // GF100, means "older than Fermi". + self.architecture_1() == 0 && self.architecture_0() < NV_PMC_BOOT_0_ARCHITECTURE_GF100 + } +} + +impl NV_PMC_BOOT_42 { + /// Combines `architecture` and `implementation` to obtain a code unique to the chipset. + pub(super) fn chipset(self) -> Result<Chipset> { + self.architecture() + .map(|arch| { + ((arch as u32) << Self::IMPLEMENTATION_RANGE.len()) + | u32::from(self.implementation()) + }) + .and_then(Chipset::try_from) + } + + /// Returns the raw architecture value from the register. + fn architecture_raw(self) -> u8 { + ((self.into_raw() >> Self::ARCHITECTURE_RANGE.start()) + & ((1 << Self::ARCHITECTURE_RANGE.len()) - 1)) as u8 + } +} + +impl kernel::fmt::Display for NV_PMC_BOOT_42 { + fn fmt(&self, f: &mut kernel::fmt::Formatter<'_>) -> kernel::fmt::Result { + write!( + f, + "boot42 = 0x{:08x} (architecture 0x{:x}, implementation 0x{:x})", + self.inner, + self.architecture_raw(), + self.implementation() + ) + } +} diff --git a/drivers/gpu/nova-core/gsp.rs b/drivers/gpu/nova-core/gsp.rs index 13f361406a6c..25ea43f1cbe9 100644 --- a/drivers/gpu/nova-core/gsp.rs +++ b/drivers/gpu/nova-core/gsp.rs @@ -115,11 +115,11 @@ impl<const NUM_PAGES: usize> PteArray<NUM_PAGES> { /// then pp points to index into the buffer where the next logging entry will /// be written. Therefore, the logging data is valid if: /// 1 <= pp < sizeof(buffer)/sizeof(u64) -struct LogBuffer(Coherent<[u8; LOG_BUFFER_SIZE]>); +struct LogBuffer<'a>(Coherent<'a, [u8; LOG_BUFFER_SIZE]>); -impl LogBuffer { +impl<'a> LogBuffer<'a> { /// Creates a new `LogBuffer` mapped on `dev`. - fn new(dev: &device::Device<device::Bound>) -> Result<Self> { + fn new(dev: &'a device::Device<device::Bound>) -> Result<Self> { let obj = Self(Coherent::zeroed(dev, GFP_KERNEL)?); let start_addr = obj.0.dma_address(); @@ -135,33 +135,33 @@ impl LogBuffer { } } -struct LogBuffers { +struct LogBuffers<'a> { /// Init log buffer. - loginit: LogBuffer, + loginit: LogBuffer<'a>, /// Interrupts log buffer. - logintr: LogBuffer, + logintr: LogBuffer<'a>, /// RM log buffer. - logrm: LogBuffer, + logrm: LogBuffer<'a>, } /// GSP runtime data. #[pin_data] -pub(crate) struct Gsp { +pub(crate) struct Gsp<'gsp> { /// Libos arguments. - pub(crate) libos: Coherent<[LibosMemoryRegionInitArgument]>, + pub(crate) libos: Coherent<'gsp, [LibosMemoryRegionInitArgument]>, /// Log buffers, optionally exposed via debugfs. #[pin] - logs: debugfs::Scope<LogBuffers>, + logs: debugfs::Scope<LogBuffers<'gsp>>, /// Command queue. #[pin] - pub(crate) cmdq: Cmdq, + pub(crate) cmdq: Cmdq<'gsp>, /// RM arguments. - rmargs: Coherent<GspArgumentsPadded>, + rmargs: Coherent<'gsp, GspArgumentsPadded>, } -impl Gsp { +impl<'gsp> Gsp<'gsp> { // Creates an in-place initializer for a `Gsp` manager for `pdev`. - pub(crate) fn new(pdev: &pci::Device<device::Bound>) -> impl PinInit<Self, Error> + '_ { + pub(crate) fn new(pdev: &'gsp pci::Device<device::Bound>) -> impl PinInit<Self, Error> + 'gsp { pin_init::pin_init_scope(move || { let dev = pdev.as_ref(); @@ -223,4 +223,4 @@ impl Gsp { } /// Opaque bundle required to unload the GSP. Created by [`Gsp::boot`], consumed by [`Gsp::unload`]. -pub(crate) struct UnloadBundle(KBox<dyn hal::UnloadBundle>); +pub(crate) struct UnloadBundle<'a>(KBox<dyn hal::UnloadBundle + 'a>); diff --git a/drivers/gpu/nova-core/gsp/boot.rs b/drivers/gpu/nova-core/gsp/boot.rs index e03700ee7bea..60bed3dc2f5a 100644 --- a/drivers/gpu/nova-core/gsp/boot.rs +++ b/drivers/gpu/nova-core/gsp/boot.rs @@ -22,7 +22,7 @@ use crate::{ }, }; -impl super::Gsp { +impl<'gsp> super::Gsp<'gsp> { /// Attempt to boot the GSP. /// /// This is a GPU-dependent and complex procedure that involves loading firmware files from @@ -33,8 +33,8 @@ impl super::Gsp { /// [`Self::unload`]) returned. pub(crate) fn boot( self: Pin<&mut Self>, - mut ctx: super::GspBootContext<'_, '_>, - ) -> Result<Option<super::UnloadBundle>> { + mut ctx: super::GspBootContext<'_, 'gsp>, + ) -> Result<Option<super::UnloadBundle<'gsp>>> { let pdev = ctx.pdev; let bar = ctx.bar; let chipset = ctx.chipset; @@ -44,6 +44,11 @@ impl super::Gsp { let gsp_fw = KBox::pin_init(GspFirmware::new(dev, chipset), GFP_KERNEL)?; + self.cmdq + .send_command_no_wait(bar, commands::SetSystemInfo::new(pdev, chipset))?; + self.cmdq + .send_command_no_wait(bar, commands::SetRegistry::new(ctx.vgpu.state())?)?; + // Perform the chipset-specific boot sequence, and retrieve the unload bundle. let unload_bundle = hal.boot(&self, &mut ctx, &gsp_fw)?.or_else(|| { dev_warn!(dev, "The GSP won't be able to unload properly on unbind.\n"); @@ -73,11 +78,6 @@ impl super::Gsp { dev_dbg!(pdev, "RISC-V active? {}\n", gsp_falcon.is_riscv_active(),); - self.cmdq - .send_command_no_wait(bar, commands::SetSystemInfo::new(pdev, chipset))?; - self.cmdq - .send_command_no_wait(bar, commands::SetRegistry::new(ctx.vgpu.state())?)?; - hal.post_boot(&self, ctx, &gsp_fw)?; // Wait until GSP is fully initialized. @@ -88,7 +88,7 @@ impl super::Gsp { /// Shut down the GSP and wait until it is offline. fn shutdown_gsp( - cmdq: &Cmdq, + cmdq: &Cmdq<'_>, bar: Bar0<'_>, gsp_falcon: &Falcon<'_, Gsp>, mode: commands::PowerStateLevel, @@ -113,7 +113,7 @@ impl super::Gsp { pub(crate) fn unload( &self, mut ctx: super::GspBootContext<'_, '_>, - unload_bundle: Option<super::UnloadBundle>, + unload_bundle: Option<super::UnloadBundle<'_>>, ) -> Result { let dev = ctx.dev(); diff --git a/drivers/gpu/nova-core/gsp/cmdq.rs b/drivers/gpu/nova-core/gsp/cmdq.rs index 6da728201281..9f99e6bbb4fa 100644 --- a/drivers/gpu/nova-core/gsp/cmdq.rs +++ b/drivers/gpu/nova-core/gsp/cmdq.rs @@ -2,13 +2,7 @@ mod continuation; -use core::{ - mem, - sync::atomic::{ - fence, - Ordering, // - }, -}; +use core::mem; use kernel::{ device, @@ -26,7 +20,12 @@ use kernel::{ prelude::*, ptr, sync::{ - aref::ARef, + barrier::{ + dma_mb, + Full, + Read, + Write, // + }, Mutex, // }, time::Delta, @@ -230,19 +229,19 @@ unsafe impl FromBytes for GspMem {} /// pointer and the GSP read pointer. This region is returned by [`Self::driver_write_area`]. /// * The driver owns (i.e. can read from) the part of the GSP message queue between the CPU read /// pointer and the GSP write pointer. This region is returned by [`Self::driver_read_area`]. -struct DmaGspMem(Coherent<GspMem>); +struct DmaGspMem<'a>(Coherent<'a, GspMem>); -impl DmaGspMem { +impl<'a> DmaGspMem<'a> { /// Allocate a new instance and map it for `dev`. - fn new(dev: &device::Device<device::Bound>) -> Result<Self> { + fn new(dev: &'a device::Device<device::Bound>) -> Result<Self> { const MSGQ_SIZE: u32 = num::usize_into_u32::<{ size_of::<Msgq>() }>(); const RX_HDR_OFF: u32 = num::usize_into_u32::<{ mem::offset_of!(Msgq, rx) }>(); - let mut gsp_mem = CoherentBox::<GspMem>::zeroed(dev, GFP_KERNEL)?; + let mut gsp_mem = CoherentBox::<'_, GspMem>::zeroed(dev, GFP_KERNEL)?; gsp_mem.cpuq.tx = MsgqTxHeader::new(MSGQ_SIZE, RX_HDR_OFF, MSGQ_NUM_PAGES); gsp_mem.cpuq.rx = MsgqRxHeader::new(); - let gsp_mem: Coherent<_> = gsp_mem.into(); + let gsp_mem: Coherent<'_, _> = gsp_mem.into(); PteArray::init(io_project!(gsp_mem, .ptes), gsp_mem.dma_address())?; Ok(Self(gsp_mem)) @@ -404,7 +403,12 @@ impl DmaGspMem { // // - The returned value is within `0..MSGQ_NUM_PAGES`. fn gsp_write_ptr(&self) -> u32 { - MsgqTxHeader::write_ptr(io_project!(self.0, .gspq.tx)) % MSGQ_NUM_PAGES + let ptr = MsgqTxHeader::write_ptr(io_project!(self.0, .gspq.tx)) % MSGQ_NUM_PAGES; + + // ORDERING: LOAD->LOAD ordering needed to order `gsp_write_ptr` read before data read. + dma_mb(Read); + + ptr } // Returns the index of the memory page the GSP will read the next command from. @@ -413,7 +417,12 @@ impl DmaGspMem { // // - The returned value is within `0..MSGQ_NUM_PAGES`. fn gsp_read_ptr(&self) -> u32 { - MsgqRxHeader::read_ptr(io_project!(self.0, .gspq.rx)) % MSGQ_NUM_PAGES + let ptr = MsgqRxHeader::read_ptr(io_project!(self.0, .gspq.rx)) % MSGQ_NUM_PAGES; + + // ORDERING: LOAD->STORE ordering needed to order `gsp_read_ptr` read before data write. + dma_mb(Full); + + ptr } // Returns the index of the memory page the CPU can read the next message from. @@ -427,12 +436,11 @@ impl DmaGspMem { // Informs the GSP that it can send `elem_count` new pages into the message queue. fn advance_cpu_read_ptr(&mut self, elem_count: u32) { + // ORDERING: LOAD->STORE ordering needed to order `cpu_read_ptr` write after data read. + dma_mb(Full); + let rx = io_project!(self.0, .cpuq.rx); let rptr = MsgqRxHeader::read_ptr(rx).wrapping_add(elem_count) % MSGQ_NUM_PAGES; - - // Ensure read pointer is properly ordered. - fence(Ordering::SeqCst); - MsgqRxHeader::set_read_ptr(rx, rptr) } @@ -447,12 +455,12 @@ impl DmaGspMem { // Informs the GSP that it can process `elem_count` new pages from the command queue. fn advance_cpu_write_ptr(&mut self, elem_count: u32) { + // ORDERING: STORE->STORE ordering needed to order `cpu_write_ptr` write after data write. + dma_mb(Write); + let tx = io_project!(self.0, .cpuq.tx); let wptr = MsgqTxHeader::write_ptr(tx).wrapping_add(elem_count) % MSGQ_NUM_PAGES; MsgqTxHeader::set_write_ptr(tx, wptr); - - // Ensure all command data is visible before triggering the GSP read. - fence(Ordering::SeqCst); } } @@ -469,7 +477,7 @@ struct GspCommand<'a> { /// A message ready to be processed from the message queue. /// -/// This is the type returned by [`Cmdq::wait_for_msg`]. +/// This is the type returned by [`CmdqInner::wait_for_msg`]. struct GspMessage<'a> { // Reference to the header of the message. header: &'a GspMsgElement, @@ -483,15 +491,15 @@ struct GspMessage<'a> { /// Provides the ability to send commands and receive messages from the GSP using a shared memory /// area. #[pin_data] -pub(crate) struct Cmdq { +pub(crate) struct Cmdq<'cmdq> { /// Inner mutex-protected state. #[pin] - inner: Mutex<CmdqInner>, + inner: Mutex<CmdqInner<'cmdq>>, /// DMA address of the command queue's shared memory region. pub(super) dma_addr: DmaAddress, } -impl Cmdq { +impl<'cmdq> Cmdq<'cmdq> { /// Offset of the data after the PTEs. const POST_PTE_OFFSET: usize = core::mem::offset_of!(GspMem, cpuq); @@ -512,14 +520,16 @@ impl Cmdq { pub(super) const RECEIVE_TIMEOUT: Delta = Delta::from_secs(5); /// Creates a new command queue for `dev`. - pub(crate) fn new(dev: &device::Device<device::Bound>) -> impl PinInit<Self, Error> + '_ { + pub(crate) fn new( + dev: &'cmdq device::Device<device::Bound>, + ) -> impl PinInit<Self, Error> + 'cmdq { pin_init_scope(move || { let gsp_mem = DmaGspMem::new(dev)?; Ok(try_pin_init!(Self { dma_addr: gsp_mem.0.dma_address(), inner <- new_mutex!(CmdqInner { - dev: dev.into(), + dev, gsp_mem, seq: 0, }), @@ -610,16 +620,16 @@ impl Cmdq { } /// Inner mutex protected state of [`Cmdq`]. -struct CmdqInner { +struct CmdqInner<'a> { /// Device this command queue belongs to. - dev: ARef<device::Device>, + dev: &'a device::Device, /// Current command sequence number. seq: u32, /// Memory area shared with the GSP for communicating commands and messages. - gsp_mem: DmaGspMem, + gsp_mem: DmaGspMem<'a>, } -impl CmdqInner { +impl CmdqInner<'_> { /// Timeout for waiting for space on the command queue. const ALLOCATE_TIMEOUT: Delta = Delta::from_secs(1); diff --git a/drivers/gpu/nova-core/gsp/commands.rs b/drivers/gpu/nova-core/gsp/commands.rs index ffc25fd8c47b..e087c9e8c35c 100644 --- a/drivers/gpu/nova-core/gsp/commands.rs +++ b/drivers/gpu/nova-core/gsp/commands.rs @@ -187,7 +187,7 @@ impl MessageFromGsp for GspInitDone { } /// Waits for GSP initialization to complete. -pub(crate) fn wait_gsp_init_done(cmdq: &Cmdq) -> Result { +pub(crate) fn wait_gsp_init_done(cmdq: &Cmdq<'_>) -> Result { loop { match cmdq.receive_msg::<GspInitDone>(Cmdq::RECEIVE_TIMEOUT) { Ok(_) => break Ok(()), @@ -214,8 +214,12 @@ impl CommandToGsp for GetGspStaticInfo { /// The reply from the GSP to the [`GetGspStaticInfo`] command. pub(crate) struct GetGspStaticInfoReply { gpu_name: [u8; 64], + /// BAR1 Page Directory Entry base address. + pub(crate) bar1_pde_base: u64, /// Usable FB (VRAM) regions for driver memory allocation. pub(crate) usable_fb_regions: KVec<Range<u64>>, + /// Exclusive end of the FB physical address space. + pub(crate) total_fb_end: u64, } impl MessageFromGsp for GetGspStaticInfoReply { @@ -231,10 +235,13 @@ impl MessageFromGsp for GetGspStaticInfoReply { for region in msg.usable_fb_regions() { usable_fb_regions.push(region, GFP_KERNEL)?; } + let total_fb_end = msg.total_fb_end().ok_or(EINVAL)?; Ok(GetGspStaticInfoReply { gpu_name: msg.gpu_name_str(), + bar1_pde_base: msg.bar1_pde_base(), usable_fb_regions, + total_fb_end, }) } } diff --git a/drivers/gpu/nova-core/gsp/fw.rs b/drivers/gpu/nova-core/gsp/fw.rs index 05f54fee6186..8778c4bf79c0 100644 --- a/drivers/gpu/nova-core/gsp/fw.rs +++ b/drivers/gpu/nova-core/gsp/fw.rs @@ -179,7 +179,7 @@ impl GspFwWprMeta { /// Returns an initializer for a `GspFwWprMeta` suitable for booting `gsp_firmware` using the /// framebuffer ranges `ranges`. pub(crate) fn from_ranges<'a>( - gsp_firmware: &'a GspFirmware, + gsp_firmware: &'a GspFirmware<'_>, ranges: &'a FbRanges, ) -> impl Init<Self> + 'a { let init_inner = init!(bindings::GspFwWprMeta { @@ -231,7 +231,7 @@ impl GspFwWprMeta { /// /// The region offsets are left at zero: the ACR ucode computes them when it sets up WPR2. pub(crate) fn from_sizes<'a>( - gsp_firmware: &'a GspFirmware, + gsp_firmware: &'a GspFirmware<'_>, sizes: &'a FbSizes, ) -> impl Init<Self> + 'a { /// VGA workspace size to reserve at the end of the framebuffer, in bytes. @@ -665,7 +665,7 @@ unsafe impl FromBytes for LibosMemoryRegionInitArgument {} impl LibosMemoryRegionInitArgument { pub(crate) fn new<'a, A: AsBytes + FromBytes + KnownSize + ?Sized>( name: &'static str, - obj: &'a Coherent<A>, + obj: &'a Coherent<'_, A>, ) -> impl Init<Self> + 'a { /// Generates the `ID8` identifier required for some GSP objects. fn id8(name: &str) -> u64 { @@ -897,7 +897,7 @@ pub(crate) struct GspArgumentsCached { impl GspArgumentsCached { /// Creates the arguments for starting the GSP up using `cmdq` as its command queue. - pub(crate) fn new(cmdq: &Cmdq) -> impl Init<Self> + '_ { + pub(crate) fn new<'a, 'b>(cmdq: &'a Cmdq<'b>) -> impl Init<Self> + use<'a, 'b> { let init_inner = init!(bindings::GSP_ARGUMENTS_CACHED { messageQueueInitArguments <- MessageQueueInitArguments::new(cmdq), bDmemStack: 1, @@ -924,7 +924,7 @@ pub(crate) struct GspArgumentsPadded { } impl GspArgumentsPadded { - pub(crate) fn new(cmdq: &Cmdq) -> impl Init<Self> + '_ { + pub(crate) fn new<'a, 'b>(cmdq: &'a Cmdq<'b>) -> impl Init<Self> + use<'a, 'b> { init!(GspArgumentsPadded { inner <- GspArgumentsCached::new(cmdq), ..Zeroable::init_zeroed() @@ -944,7 +944,7 @@ type MessageQueueInitArguments = bindings::MESSAGE_QUEUE_INIT_ARGUMENTS; impl MessageQueueInitArguments { /// Creates a new init arguments structure for `cmdq`. - fn new(cmdq: &Cmdq) -> impl Init<Self> + '_ { + fn new<'a, 'b>(cmdq: &'a Cmdq<'b>) -> impl Init<Self> + use<'a, 'b> { init!(MessageQueueInitArguments { sharedMemPhysAddr: cmdq.dma_addr, pageTableEntryCount: num::usize_into_u32::<{ Cmdq::NUM_PTES }>(), diff --git a/drivers/gpu/nova-core/gsp/fw/commands.rs b/drivers/gpu/nova-core/gsp/fw/commands.rs index 6dc31d1bf5ae..32856ff74183 100644 --- a/drivers/gpu/nova-core/gsp/fw/commands.rs +++ b/drivers/gpu/nova-core/gsp/fw/commands.rs @@ -131,6 +131,14 @@ impl GspStaticConfigInfo { self.0.gpuNameString } + /// Returns the BAR1 Page Directory Entry base address. + /// + /// This is the root page table address for BAR1 virtual memory, + /// set up by GSP-RM firmware. + pub(crate) fn bar1_pde_base(&self) -> u64 { + self.0.bar1PdeBase + } + /// Returns an iterator over valid FB regions from GSP firmware data. fn fb_regions( &self, @@ -165,6 +173,11 @@ impl GspStaticConfigInfo { } }) } + + /// Computes the exclusive end of the FB physical address space. + pub(crate) fn total_fb_end(&self) -> Option<u64> { + self.fb_regions().map(|reg| reg.limit).max()?.checked_add(1) + } } // SAFETY: Padding is explicit and will not contain uninitialized data. diff --git a/drivers/gpu/nova-core/gsp/hal.rs b/drivers/gpu/nova-core/gsp/hal.rs index 5850fa0fe0e9..d8329f6fcc65 100644 --- a/drivers/gpu/nova-core/gsp/hal.rs +++ b/drivers/gpu/nova-core/gsp/hal.rs @@ -35,12 +35,12 @@ pub(super) trait GspHal: Send { /// /// Upon success, returns the [`crate::gsp::UnloadBundle`] to use with [`Gsp::unload`], if one /// could be created. - fn boot( + fn boot<'gpu>( &self, - gsp: &Gsp, - ctx: &mut GspBootContext<'_, '_>, - gsp_fw: &GspFirmware, - ) -> Result<Option<crate::gsp::UnloadBundle>>; + gsp: &Gsp<'gpu>, + ctx: &mut GspBootContext<'_, 'gpu>, + gsp_fw: &GspFirmware<'gpu>, + ) -> Result<Option<super::UnloadBundle<'gpu>>>; /// Performs HAL-specific post-GSP boot tasks. /// @@ -48,9 +48,9 @@ pub(super) trait GspHal: Send { /// after the initialization commands have been pushed onto its queue. fn post_boot( &self, - _gsp: &Gsp, + _gsp: &Gsp<'_>, _ctx: &mut GspBootContext<'_, '_>, - _gsp_fw: &GspFirmware, + _gsp_fw: &GspFirmware<'_>, ) -> Result { Ok(()) } diff --git a/drivers/gpu/nova-core/gsp/hal/gh100.rs b/drivers/gpu/nova-core/gsp/hal/gh100.rs index e283429a95dd..91201b51030e 100644 --- a/drivers/gpu/nova-core/gsp/hal/gh100.rs +++ b/drivers/gpu/nova-core/gsp/hal/gh100.rs @@ -58,7 +58,7 @@ impl GspMbox { fn lockdown_released_or_error( &self, gsp_falcon: &Falcon<'_, GspEngine>, - fmc_boot_params: &Coherent<GspFmcBootParams>, + fmc_boot_params: &Coherent<'_, GspFmcBootParams>, ) -> bool { // GSP-FMC normally clears the boot parameters address from the mailboxes early during // boot. If the address is still there, keep polling rather than treating it as an error. @@ -75,7 +75,7 @@ impl GspMbox { fn wait_for_gsp_lockdown_release( dev: &device::Device<device::Bound>, gsp_falcon: &Falcon<'_, GspEngine>, - fmc_boot_params: &Coherent<GspFmcBootParams>, + fmc_boot_params: &Coherent<'_, GspFmcBootParams>, ) -> Result { dev_dbg!(dev, "Waiting for GSP lockdown release\n"); @@ -141,12 +141,12 @@ impl GspHal for Gh100 { /// /// This path uses FSP to establish a chain of trust and boot GSP-FMC. FSP handles /// the GSP boot internally - no manual GSP reset/boot is needed. - fn boot( + fn boot<'gpu>( &self, - gsp: &Gsp, - ctx: &mut GspBootContext<'_, '_>, - gsp_fw: &GspFirmware, - ) -> Result<Option<crate::gsp::UnloadBundle>> { + gsp: &Gsp<'gpu>, + ctx: &mut GspBootContext<'_, 'gpu>, + gsp_fw: &GspFirmware<'gpu>, + ) -> Result<Option<crate::gsp::UnloadBundle<'gpu>>> { let dev = ctx.dev(); let chipset = ctx.chipset; let gsp_falcon = ctx.gsp_falcon; @@ -159,7 +159,7 @@ impl GspHal for Gh100 { let args = FmcBootArgs::new(dev, chipset, wpr_meta, &gsp.libos, false)?; let unload_bundle = crate::gsp::UnloadBundle( - KBox::new(FspUnloadBundle, GFP_KERNEL)? as KBox<dyn UnloadBundle> + KBox::new(FspUnloadBundle, GFP_KERNEL)? as KBox<dyn UnloadBundle + 'gpu> ); // Wait for the GSP RISC-V core to halt in case of error. We create this guard after `args` diff --git a/drivers/gpu/nova-core/gsp/hal/tu102.rs b/drivers/gpu/nova-core/gsp/hal/tu102.rs index a5c0ca355493..e90db1a23032 100644 --- a/drivers/gpu/nova-core/gsp/hal/tu102.rs +++ b/drivers/gpu/nova-core/gsp/hal/tu102.rs @@ -52,34 +52,33 @@ use crate::{ // // Since there are two variants of the prepared firmware (with and without a bootloader), this type // abstracts the difference. -enum FwsecUnloadFirmware { +enum FwsecUnloadFirmware<'a> { WithoutBl(FwsecFirmware), - WithBl(FwsecFirmwareWithBl), + WithBl(FwsecFirmwareWithBl<'a>), } -impl FwsecUnloadFirmware { +impl FwsecUnloadFirmware<'_> { /// Runs the FWSEC SB firmware. fn run( &self, dev: &device::Device<device::Bound>, - bar: Bar0<'_>, gsp_falcon: &Falcon<'_, GspEngine>, ) -> Result { match self { Self::WithoutBl(fw) => fw.run(dev, gsp_falcon), - Self::WithBl(fw) => fw.run(dev, gsp_falcon, bar), + Self::WithBl(fw) => fw.run(dev, gsp_falcon), } } } // Contains the firmware required to fully reset GSP on chipsets where the GSP is started using // FWSEC/Booter. -struct Sec2UnloadBundle { - fwsec_sb: FwsecUnloadFirmware, +struct Sec2UnloadBundle<'a> { + fwsec_sb: FwsecUnloadFirmware<'a>, booter_unloader: BooterFirmware, } -impl UnloadBundle for Sec2UnloadBundle { +impl UnloadBundle for Sec2UnloadBundle<'_> { fn run(&self, ctx: &mut GspBootContext<'_, '_>) -> Result { let dev = ctx.dev(); let bar = ctx.bar; @@ -88,7 +87,7 @@ impl UnloadBundle for Sec2UnloadBundle { // Log errors but keep going if it fails. let fwsec_sb_res = self .fwsec_sb - .run(dev, bar, ctx.gsp_falcon) + .run(dev, ctx.gsp_falcon) .inspect_err(|e| dev_err!(dev, "FWSEC-SB failed to run: {:?}\n", e)); // Remove WPR2 region if set. @@ -168,7 +167,7 @@ impl Tu102 { if self.needs_fwsec_bootloader { let fwsec_frts_bl = FwsecFirmwareWithBl::new(fwsec_frts, dev, chipset)?; // Load and run the bootloader, which will load FWSEC-FRTS and run it. - fwsec_frts_bl.run(dev, falcon, bar)?; + fwsec_frts_bl.run(dev, falcon)?; } else { // Load and run FWSEC-FRTS directly. fwsec_frts.run(dev, falcon)?; @@ -213,14 +212,14 @@ impl Tu102 { } /// Load and prepare the resources required to properly reset the GSP after it has been stopped. - fn build_unload_bundle( + fn build_unload_bundle<'gpu>( &self, - dev: &device::Device<device::Bound>, + dev: &'gpu device::Device<device::Bound>, chipset: Chipset, bios: &Vbios, gsp_falcon: &Falcon<'_, GspEngine>, sec2_falcon: &Falcon<'_, Sec2>, - ) -> Result<crate::gsp::UnloadBundle> { + ) -> Result<crate::gsp::UnloadBundle<'gpu>> { // Load the FWSEC SB firmware, as well as its bootloader if required. let fwsec_sb = FwsecFirmware::new(dev, gsp_falcon, bios, FwsecCommand::Sb)?; let fwsec_sb = if self.needs_fwsec_bootloader { @@ -241,18 +240,18 @@ impl Tu102 { }, GFP_KERNEL, ) - .map(|b| crate::gsp::UnloadBundle(b)) + .map(|b| crate::gsp::UnloadBundle(b as KBox<dyn UnloadBundle + 'gpu>)) .map_err(Into::into) } } impl GspHal for Tu102 { - fn boot( + fn boot<'gpu>( &self, - gsp: &Gsp, - ctx: &mut GspBootContext<'_, '_>, - gsp_fw: &GspFirmware, - ) -> Result<Option<crate::gsp::UnloadBundle>> { + gsp: &Gsp<'gpu>, + ctx: &mut GspBootContext<'_, 'gpu>, + gsp_fw: &GspFirmware<'gpu>, + ) -> Result<Option<crate::gsp::UnloadBundle<'gpu>>> { let dev = ctx.dev(); let bar = ctx.bar; let chipset = ctx.chipset; @@ -317,9 +316,9 @@ impl GspHal for Tu102 { fn post_boot( &self, - gsp: &Gsp, + gsp: &Gsp<'_>, ctx: &mut GspBootContext<'_, '_>, - gsp_fw: &GspFirmware, + gsp_fw: &GspFirmware<'_>, ) -> Result { GspSequencer::run(&gsp.cmdq, ctx, &gsp.libos, gsp_fw.bootloader.app_version)?; diff --git a/drivers/gpu/nova-core/gsp/regs.rs b/drivers/gpu/nova-core/gsp/regs.rs index 9a48aa87e7fb..3c410d65e8e4 100644 --- a/drivers/gpu/nova-core/gsp/regs.rs +++ b/drivers/gpu/nova-core/gsp/regs.rs @@ -2,11 +2,16 @@ use kernel::io::register; -use crate::regs::NV_PBUS_SW_SCRATCH; +use crate::{ + driver::NovaRegisters, + regs::NV_PBUS_SW_SCRATCH, // +}; // PGSP register! { + base: NovaRegisters; + pub(super) NV_PGSP_QUEUE_HEAD(u32) @ 0x00110c00 { 31:0 address; } @@ -15,6 +20,8 @@ register! { // PBUS register! { + base: NovaRegisters; + /// Scratch register 0xe used as FRTS firmware error code. pub(super) NV_PBUS_SW_SCRATCH_0E_FRTS_ERR(u32) => NV_PBUS_SW_SCRATCH[0xe] { 31:16 frts_err_code; diff --git a/drivers/gpu/nova-core/gsp/sequencer.rs b/drivers/gpu/nova-core/gsp/sequencer.rs index bcad1421953a..dae34c11eb05 100644 --- a/drivers/gpu/nova-core/gsp/sequencer.rs +++ b/drivers/gpu/nova-core/gsp/sequencer.rs @@ -138,7 +138,7 @@ pub(crate) struct GspSequencer<'a> { /// GSP falcon for core operations. gsp_falcon: &'a Falcon<'a, Gsp>, /// LibOS memory region init arguments. - libos: &'a Coherent<[LibosMemoryRegionInitArgument]>, + libos: &'a Coherent<'a, [LibosMemoryRegionInitArgument]>, /// Bootloader application version. bootloader_app_version: u32, /// Device for logging. @@ -338,9 +338,9 @@ impl<'a> Iterator for GspSeqIter<'a> { impl<'a> GspSequencer<'a> { pub(crate) fn run( - cmdq: &Cmdq, + cmdq: &Cmdq<'_>, ctx: &'a GspBootContext<'_, '_>, - libos: &'a Coherent<[LibosMemoryRegionInitArgument]>, + libos: &'a Coherent<'a, [LibosMemoryRegionInitArgument]>, bootloader_app_version: u32, ) -> Result { let seq_info = loop { diff --git a/drivers/gpu/nova-core/mm.rs b/drivers/gpu/nova-core/mm.rs new file mode 100644 index 000000000000..a5bc4042577b --- /dev/null +++ b/drivers/gpu/nova-core/mm.rs @@ -0,0 +1,336 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Memory management subsystems. + +#![cfg_attr(not(CONFIG_NOVA_CORE_SELFTESTS), expect(dead_code))] + +/// Implements `From` conversions between a frame-number type and `Bounded<u64, N>`. +/// +/// Each MMU version module should invoke this for the specific bit widths used by that version's +/// PTE/PDE bitfield definitions. +macro_rules! impl_frame_number_bounded { + ($type:ty, $bits:literal) => { + impl From<Bounded<u64, $bits>> for $type { + fn from(val: Bounded<u64, $bits>) -> Self { + Self::new(val.get()) + } + } + + impl From<$type> for Bounded<u64, $bits> { + fn from(v: $type) -> Self { + Bounded::from_expr(v.raw() & ::kernel::bits::genmask_u64(0..=($bits - 1))) + } + } + }; +} + +/// Implements `From` conversions between [`Pfn`] and `Bounded<u64, N>` for bitfield interop. +macro_rules! impl_pfn_bounded { + ($bits:literal) => { + impl_frame_number_bounded!(Pfn, $bits); + }; +} + +use core::{ + fmt::LowerHex, + ops, // +}; + +use kernel::{ + bitfield, + fmt, + gpu::buddy::{ + GpuBuddy, + GpuBuddyParams, // + }, + num::Bounded, + prelude::*, + ptr::{ + Alignable, + Alignment, // + }, + sizes::SZ_4K, // +}; + +use crate::{ + driver::Bar0, + gpu::Chipset, // +}; + +pub(crate) use tlb::Tlb; + +pub(crate) mod bar_user; +mod hal; +pub(super) mod pagetable; +mod pramin; +mod regs; +pub(super) mod tlb; +pub(super) mod vmm; + +/// GPU Memory Manager - owns all core MM components. +/// +/// Provides centralized ownership of memory management resources: +/// - [`GpuBuddy`] allocator for VRAM page table allocation. +/// - [`pramin::Pramin`] for direct VRAM access. +/// - [`Tlb`] manager for translation buffer flush operations. +pub(crate) struct GpuMm<'gpu> { + buddy: GpuBuddy, + pramin: pramin::Pramin<'gpu>, + tlb: Pin<KBox<Tlb<'gpu>>>, +} + +impl<'gpu> GpuMm<'gpu> { + /// Creates the GPU memory manager. + pub(crate) fn new( + bar: Bar0<'gpu>, + chipset: Chipset, + buddy_params: GpuBuddyParams, + total_fb_end: VramAddress, + ) -> Result<Self> { + // PRAMIN covers all physical VRAM (including GSP-reserved areas + // above the usable region, e.g. the BAR1 page directory). + let vram_region = VramAddress::ZERO..total_fb_end; + + Ok(Self { + buddy: GpuBuddy::new(buddy_params)?, + pramin: pramin::Pramin::new(bar, chipset, vram_region)?, + tlb: KBox::pin_init(Tlb::new(bar), GFP_KERNEL)?, + }) + } + + /// Access the [`GpuBuddy`] allocator. + pub(crate) fn buddy(&self) -> &GpuBuddy { + &self.buddy + } + + /// Access the [`pramin::Pramin`]. + fn pramin_mut(&mut self) -> &mut pramin::Pramin<'gpu> { + &mut self.pramin + } + + /// Access the [`Tlb`] manager. + pub(crate) fn tlb(&self) -> &Tlb<'gpu> { + self.tlb.as_ref().get_ref() + } +} + +/// Page size in bytes (4 KiB). +pub(crate) const PAGE_SIZE: usize = SZ_4K; + +/// Physical VRAM address in GPU video memory. +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +#[repr(transparent)] +pub(crate) struct VramAddress(u64); + +impl VramAddress { + /// The zero address. + pub(crate) const ZERO: Self = Self::from_raw(0); + + /// Creates an address from a raw value. + pub(crate) const fn from_raw(addr: u64) -> Self { + Self(addr) + } + + /// Returns the address as a raw value. + pub(crate) const fn into_raw(self) -> u64 { + self.0 + } + + /// Adds `rhs` to this address, returning [`None`] on overflow. + pub(crate) const fn checked_add(self, rhs: u64) -> Option<Self> { + match self.into_raw().checked_add(rhs) { + Some(addr) => Some(Self::from_raw(addr)), + None => None, + } + } +} + +impl Alignable for VramAddress { + fn align_down(self, alignment: Alignment) -> Self { + Self::from_raw(self.into_raw().align_down(alignment)) + } + + fn align_up(self, alignment: Alignment) -> Option<Self> { + self.into_raw().align_up(alignment).map(Self::from_raw) + } +} + +impl LowerHex for VramAddress { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + LowerHex::fmt(&self.into_raw(), f) + } +} + +impl fmt::Debug for VramAddress { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_fmt(fmt!("{:#x}", self)) + } +} + +impl ops::Add<u64> for VramAddress { + type Output = Self; + + fn add(self, rhs: u64) -> Self::Output { + Self::from_raw(self.into_raw() + rhs) + } +} + +impl ops::Sub for VramAddress { + type Output = u64; + + fn sub(self, rhs: Self) -> Self::Output { + self.into_raw() - rhs.into_raw() + } +} + +impl From<Pfn> for VramAddress { + fn from(pfn: Pfn) -> Self { + Self::from_raw(pfn.raw() << 12) + } +} + +bitfield! { + /// Virtual address in GPU address space. + pub(crate) struct VirtualAddress(u64) { + /// Offset within 4KB page. + 11:0 offset; + /// Virtual frame number. + 63:12 frame_number => Vfn; + } +} + +impl VirtualAddress { + /// Create a new virtual address from a raw value. + #[expect(dead_code)] + pub(crate) const fn new(addr: u64) -> Self { + Self::from_raw(addr) + } +} + +impl From<Vfn> for VirtualAddress { + fn from(vfn: Vfn) -> Self { + Self::zeroed().with_frame_number(vfn) + } +} + +/// Physical Frame Number. +/// +/// Represents a physical page in VRAM. +#[repr(transparent)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub(crate) struct Pfn(u64); + +impl Pfn { + /// Create a new PFN from a frame number. + pub(crate) const fn new(frame_number: u64) -> Self { + Self(frame_number) + } + + /// Get the raw frame number. + pub(crate) const fn raw(self) -> u64 { + self.0 + } +} + +impl From<VramAddress> for Pfn { + fn from(addr: VramAddress) -> Self { + Self::new(addr.into_raw() >> 12) + } +} + +impl From<u64> for Pfn { + fn from(val: u64) -> Self { + Self(val) + } +} + +impl From<Pfn> for u64 { + fn from(pfn: Pfn) -> Self { + pfn.0 + } +} + +impl_pfn_bounded!(52); + +/// Virtual Frame Number. +/// +/// Represents a virtual page in GPU address space. +#[repr(transparent)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub(crate) struct Vfn(u64); + +impl Vfn { + /// Create a new VFN from a frame number. + pub(crate) const fn new(frame_number: u64) -> Self { + Self(frame_number) + } + + /// Get the raw frame number. + pub(crate) const fn raw(self) -> u64 { + self.0 + } +} + +impl From<VirtualAddress> for Vfn { + fn from(addr: VirtualAddress) -> Self { + addr.frame_number() + } +} + +impl From<u64> for Vfn { + fn from(val: u64) -> Self { + Self(val) + } +} + +impl From<Vfn> for u64 { + fn from(vfn: Vfn) -> Self { + vfn.0 + } +} + +impl_frame_number_bounded!(Vfn, 52); + +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +pub(crate) mod selftest { + use core::ops::Range; + + use kernel::{ + device, + sizes::SizeConstants, + sync::Arc, // + }; + + use super::*; + + /// Run MM subsystem self-tests during probe. + pub(crate) fn run( + dev: &device::Device<device::Bound>, + mm: &mut GpuMm<'_>, + usable_fb_regions: &[Range<u64>], + bar_user: &Arc<bar_user::BarUser<'_>>, + bar1_pdb: u64, + chipset: Chipset, + ) -> Result { + // VRAM span the self-tests are free to overwrite, from the chosen test base. + const SELFTEST_SPAN: u64 = u64::SZ_64M; + + let base = usable_fb_regions.iter().find_map(|region| { + // Tests rely on this being 8 byte aligned for checking misalignment handling. + let base = region.start.align_up(Alignment::new::<8>())?; + (base.checked_add(SELFTEST_SPAN)? <= region.end).then_some(base) + }); + let Some(base) = base else { + dev_warn!( + dev, + "PRAMIN: skipping self-tests, no usable VRAM region of {:#x} bytes\n", + SELFTEST_SPAN + ); + return Ok(()); + }; + + pramin::selftest::run(dev, mm.pramin_mut(), VramAddress::from_raw(base))?; + bar_user::run_self_test(dev, mm, bar_user, bar1_pdb, chipset) + } +} diff --git a/drivers/gpu/nova-core/mm/bar_user.rs b/drivers/gpu/nova-core/mm/bar_user.rs new file mode 100644 index 000000000000..8f4a27c1fd14 --- /dev/null +++ b/drivers/gpu/nova-core/mm/bar_user.rs @@ -0,0 +1,424 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! BAR1 user interface for CPU access to GPU virtual memory. Used for USERD +//! for GPU work submission, and applications to access GPU buffers via mmap(). + +use kernel::{ + io::Io, + new_mutex, + prelude::*, + sync::{ + Arc, + Mutex, // + }, +}; + +use crate::{ + driver::Bar1, + gpu::Chipset, + mm::{ + vmm::{ + MappedRange, + Vmm, // + }, + GpuMm, + Pfn, + Vfn, + VirtualAddress, + VramAddress, + PAGE_SIZE, // + }, + num::IntoSafeCast, +}; + +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +use kernel::device; + +/// BAR1 user interface for virtual memory mappings. +/// +/// Owns the [`Vmm`] for the BAR1 address space. +#[pin_data] +pub(crate) struct BarUser<'gpu> { + #[pin] + vmm: Mutex<Vmm>, + bar1: &'gpu Bar1<'gpu>, +} + +impl<'gpu> BarUser<'gpu> { + /// Create a pin-initializer for [`BarUser`]. + pub(crate) fn new( + pdb_addr: VramAddress, + chipset: Chipset, + va_size: u64, + bar1: &'gpu Bar1<'gpu>, + ) -> Result<impl PinInit<Self> + 'gpu> { + let vmm = Vmm::new(pdb_addr, chipset.mmu_version(), va_size)?; + Ok(pin_init!(Self { + vmm <- new_mutex!(vmm, "bar_user_vmm"), + bar1, + })) + } + + /// Map physical pages to a contiguous BAR1 virtual range. + pub(crate) fn map( + self: &Arc<Self>, + mm: &mut GpuMm<'_>, + pfns: &[Pfn], + writable: bool, + ) -> Result<BarUserAccess<'gpu>> { + if pfns.is_empty() { + return Err(EINVAL); + } + let mut vmm = self.vmm.lock(); + let mapped = vmm.map_pages(mm, pfns, None, writable)?; + + Ok(BarUserAccess { + bar_user: self.clone(), + mapped: Some(mapped), + }) + } +} + +/// Access object for a mapped BAR1 region. +pub(crate) struct BarUserAccess<'gpu> { + bar_user: Arc<BarUser<'gpu>>, + /// [`BarUserAccess::release`] [`Option::take`]s this; `Some` at + /// drop time means `release()` was never called. + mapped: Option<MappedRange>, +} + +#[expect(dead_code)] +impl BarUserAccess<'_> { + /// Tear down the BAR1 mapping. + pub(crate) fn release(mut self, mm: &mut GpuMm<'_>) -> Result { + let mapped = self.mapped.take().ok_or(EINVAL)?; + let mut vmm = self.bar_user.vmm.lock(); + vmm.unmap_pages(mm, mapped)?; + Ok(()) + } + + /// Returns the active mapping. + fn mapped(&self) -> &MappedRange { + // `mapped` is only `None` after `take()` in `release`; hence unwrap() + // cannot panic here. + self.mapped.as_ref().unwrap() + } + + /// Get the base virtual address of this mapping. + pub(crate) fn base(&self) -> VirtualAddress { + VirtualAddress::from(self.mapped().vfn_start) + } + + /// Get the total size of the mapped region in bytes. + pub(crate) fn size(&self) -> usize { + self.mapped().num_pages * PAGE_SIZE + } + + /// Get the starting virtual frame number. + pub(crate) fn vfn_start(&self) -> Vfn { + self.mapped().vfn_start + } + + /// Get the number of pages in this mapping. + pub(crate) fn num_pages(&self) -> usize { + self.mapped().num_pages + } + + /// Translate an offset within this mapping to a BAR1 aperture offset. + fn bar_offset(&self, offset: usize) -> Result<usize> { + if offset >= self.size() { + return Err(EINVAL); + } + + let base_vfn: usize = self.mapped().vfn_start.raw().into_safe_cast(); + let base = base_vfn.checked_mul(PAGE_SIZE).ok_or(EOVERFLOW)?; + base.checked_add(offset).ok_or(EOVERFLOW) + } + + // Fallible accessors with runtime bounds checking. + + /// Read a 32-bit value at the given offset. + pub(crate) fn try_read32(&self, offset: usize) -> Result<u32> { + let off = self.bar_offset(offset)?; + self.bar_user.bar1.try_read32(off) + } + + /// Write a 32-bit value at the given offset. + pub(crate) fn try_write32(&self, value: u32, offset: usize) -> Result { + let off = self.bar_offset(offset)?; + self.bar_user.bar1.try_write32(value, off) + } + + /// Read a 64-bit value at the given offset. + pub(crate) fn try_read64(&self, offset: usize) -> Result<u64> { + let off = self.bar_offset(offset)?; + self.bar_user.bar1.try_read64(off) + } + + /// Write a 64-bit value at the given offset. + pub(crate) fn try_write64(&self, value: u64, offset: usize) -> Result { + let off = self.bar_offset(offset)?; + self.bar_user.bar1.try_write64(value, off) + } +} + +impl Drop for BarUserAccess<'_> { + fn drop(&mut self) { + if self.mapped.is_some() { + kernel::pr_warn!( + "BarUserAccess dropped without calling release(). BarUser address space will leak.\n" + ); + } + // The inner `MappedRange`'s own `MustUnmapGuard` will also fire, + // identifying the leaked VA range. + } +} + +/// Run MM subsystem self-tests during probe. +/// +/// Tests page table infrastructure and `BAR1` MMIO access using the `BAR1` +/// address space. Uses the `GpuMm`'s buddy allocator to allocate page tables +/// and test pages as needed. +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +pub(crate) fn run_self_test( + dev: &device::Device<device::Bound>, + mm: &mut GpuMm<'_>, + bar_user: &Arc<BarUser<'_>>, + bar1_pdb: u64, + chipset: Chipset, +) -> Result { + use kernel::{ + gpu::buddy::{ + GpuBuddyAllocFlags, + GpuBuddyAllocMode, // + }, + ptr::Alignment, + sizes::{ + SZ_16K, + SZ_32K, + SZ_4K, + SZ_64K, // + }, + }; + + // Test patterns. + const PATTERN_PRAMIN: u32 = 0xDEAD_BEEF; + const PATTERN_BAR1: u32 = 0xCAFE_BABE; + + let bar1 = bar_user.bar1; + dev_info!(dev, "MM: Starting self-test...\n"); + + let pdb_addr = VramAddress::from_raw(bar1_pdb); + + // Check if initial page tables are in VRAM. + if crate::mm::pagetable::check_pdb_valid(mm.pramin_mut(), pdb_addr, chipset).is_err() { + dev_info!(dev, "MM: Self-test SKIPPED - no valid VRAM page tables\n"); + return Ok(()); + } + + // Set up a test page from the buddy allocator. + let test_page_blocks = KBox::pin_init( + mm.buddy().alloc_blocks( + GpuBuddyAllocMode::Simple, + SZ_4K.into_safe_cast(), + Alignment::new::<SZ_4K>(), + GpuBuddyAllocFlags::default(), + ), + GFP_KERNEL, + )?; + let test_vram_offset = test_page_blocks.iter().next().ok_or(ENOMEM)?.offset(); + let test_vram = VramAddress::from_raw(test_vram_offset); + let test_pfn = Pfn::from(test_vram); + + // Create a VMM of size 64K to track virtual memory mappings. + let mut vmm = Vmm::new(pdb_addr, chipset.mmu_version(), SZ_64K.into_safe_cast())?; + + // Create a test mapping. + let mapped = vmm.map_pages(mm, &[test_pfn], None, true)?; + let test_vfn = mapped.vfn_start; + + // Pre-compute test addresses for the PRAMIN to BAR1 read test. + let vfn_offset: usize = test_vfn.raw().into_safe_cast(); + let bar1_base_offset = vfn_offset.checked_mul(PAGE_SIZE).ok_or(EOVERFLOW)?; + let bar1_read_offset: usize = bar1_base_offset + 0x100; + let vram_read_addr = test_vram + 0x100; + + // Test 1: Write via PRAMIN, read via BAR1. + mm.pramin_mut() + .window_at::<u32>(vram_read_addr)? + .view() + .write_val(PATTERN_PRAMIN); + + // Read back via BAR1 aperture. + let bar1_value = bar1.try_read32(bar1_read_offset)?; + + let test1_passed = if bar1_value == PATTERN_PRAMIN { + true + } else { + dev_err!( + dev, + "MM: Test 1 FAILED - Expected {:#010x}, got {:#010x}\n", + PATTERN_PRAMIN, + bar1_value + ); + false + }; + + // Cleanup - invalidate PTE. + vmm.unmap_pages(mm, mapped)?; + + // Test 2: Two-phase prepare/execute API. + let prepared = vmm.prepare_map(mm, 1, None)?; + let mapped2 = vmm.execute_map(mm, prepared, &[test_pfn], true)?; + let readback = vmm.read_mapping(mm, mapped2.vfn_start)?; + let test2_passed = if readback == Some(test_pfn) { + true + } else { + dev_err!(dev, "MM: Test 2 FAILED - Two-phase map readback mismatch\n"); + false + }; + vmm.unmap_pages(mm, mapped2)?; + + // Test 3: Range-constrained allocation with a hole — exercises block.size()-driven + // BAR1 mapping. A 4K hole is punched at base+16K, then a single 32K allocation + // is requested within [base, base+36K). The buddy allocator must split around the + // hole, returning multiple blocks (expected: {16K, 4K, 8K, 4K} = 32K total). + // Each block is mapped into BAR1 and verified via PRAMIN read-back. + // + // Address layout (base = 0x10000): + // [ 16K ] [HOLE 4K] [4K] [ 8K ] [4K] + // 0x10000 0x14000 0x15000 0x16000 0x18000 0x19000 + let range_base: u64 = SZ_64K.into_safe_cast(); + let sz_4k: u64 = SZ_4K.into_safe_cast(); + let sz_16k: u64 = SZ_16K.into_safe_cast(); + let sz_32k_4k: u64 = (SZ_32K + SZ_4K).into_safe_cast(); + + // Punch a 4K hole at base+16K so the subsequent 32K allocation must split. + let _hole = KBox::pin_init( + mm.buddy().alloc_blocks( + GpuBuddyAllocMode::Range(range_base + sz_16k..range_base + sz_16k + sz_4k), + SZ_4K.into_safe_cast(), + Alignment::new::<SZ_4K>(), + GpuBuddyAllocFlags::default(), + ), + GFP_KERNEL, + )?; + + // Allocate 32K within [base, base+36K). The hole forces the allocator to return + // split blocks whose sizes are determined by buddy alignment. + let blocks = KBox::pin_init( + mm.buddy().alloc_blocks( + GpuBuddyAllocMode::Range(range_base..range_base + sz_32k_4k), + SZ_32K.into_safe_cast(), + Alignment::new::<SZ_4K>(), + GpuBuddyAllocFlags::default(), + ), + GFP_KERNEL, + )?; + + let mut test3_passed = true; + let mut total_size = 0usize; + + for block in blocks.iter() { + total_size += IntoSafeCast::<usize>::into_safe_cast(block.size()); + + // Map all pages of this block. + let page_size: u64 = PAGE_SIZE.into_safe_cast(); + let num_pages: usize = (block.size() / page_size).into_safe_cast(); + + let mut pfns = KVec::new(); + for j in 0..num_pages { + let j_u64: u64 = j.into_safe_cast(); + pfns.push( + Pfn::from(VramAddress::from_raw( + block.offset() + j_u64.checked_mul(page_size).ok_or(EOVERFLOW)?, + )), + GFP_KERNEL, + )?; + } + + let mapped = vmm.map_pages(mm, &pfns, None, true)?; + let bar1_base_vfn: usize = mapped.vfn_start.raw().into_safe_cast(); + let bar1_base = bar1_base_vfn.checked_mul(PAGE_SIZE).ok_or(EOVERFLOW)?; + + for j in 0..num_pages { + let page_bar1_off = bar1_base + j * PAGE_SIZE; + let j_u64: u64 = j.into_safe_cast(); + let page_phys = block.offset() + + j_u64 + .checked_mul(PAGE_SIZE.into_safe_cast()) + .ok_or(EOVERFLOW)?; + + bar1.try_write32(PATTERN_BAR1, page_bar1_off)?; + + let pramin_val = mm + .pramin_mut() + .window_at::<u32>(VramAddress::from_raw(page_phys))? + .view() + .read_val(); + + if pramin_val != PATTERN_BAR1 { + dev_err!( + dev, + "MM: Test 3 FAILED block offset {:#x} page {} (val={:#x})\n", + block.offset(), + j, + pramin_val + ); + test3_passed = false; + } + } + + vmm.unmap_pages(mm, mapped)?; + } + + // Verify aggregate: all returned block sizes must sum to allocation size. + if total_size != SZ_32K { + dev_err!( + dev, + "MM: Test 3 FAILED - total size {} != expected {}\n", + total_size, + SZ_32K + ); + test3_passed = false; + } + + // Release Tests 1-3's Vmm before Test 4 constructs a fresh BarUser on + // the same PDB. + drop(vmm); + + // Test 4: Exercise `BarUser::map()` end-to-end. + let bar_user = Arc::pin_init( + BarUser::new(pdb_addr, chipset, SZ_64K.into_safe_cast(), bar1)?, + GFP_KERNEL, + )?; + let access = bar_user.map(mm, &[test_pfn], true)?; + + // Write pattern via PRAMIN, read via BarUserAccess. + mm.pramin_mut() + .window_at::<u32>(test_vram)? + .view() + .write_val(PATTERN_BAR1); + + let readback = access.try_read32(0)?; + let test4_passed = if readback == PATTERN_BAR1 { + true + } else { + dev_err!( + dev, + "MM: Test 4 FAILED - Expected {:#010x}, got {:#010x}\n", + PATTERN_BAR1, + readback + ); + false + }; + access.release(mm)?; + + if test1_passed && test2_passed && test3_passed && test4_passed { + dev_info!(dev, "MM: All self-tests PASSED\n"); + Ok(()) + } else { + dev_err!(dev, "MM: Self-tests FAILED\n"); + Err(EIO) + } +} diff --git a/drivers/gpu/nova-core/mm/hal.rs b/drivers/gpu/nova-core/mm/hal.rs new file mode 100644 index 000000000000..e7fd1e38bd38 --- /dev/null +++ b/drivers/gpu/nova-core/mm/hal.rs @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Memory management HAL. + +use kernel::{ + num::Bounded, + prelude::*, // +}; + +use crate::{ + driver::Bar0, + gpu::{ + Architecture, + Chipset, // + }, + mm::VramAddress, // +}; + +mod gb100; +mod gh100; +mod tu102; + +/// Trait implemented by per-architecture MM HALs. +/// +/// `Sync` is required so that the `&'static dyn MmHal` references can be stored in `Send` +/// structures. +pub(super) trait MmHal: Sync { + /// Positions the PRAMIN window at `base`. + /// + /// This fails if `base` is not aligned to the 64 KiB window alignment or is too large for + /// the receiving register. + fn write_pramin_window_base(&self, bar: Bar0<'_>, base: VramAddress) -> Result; +} + +/// Returns the HAL corresponding to `chipset`. +pub(super) fn mm_hal(chipset: Chipset) -> &'static dyn MmHal { + match chipset.arch() { + Architecture::Turing | Architecture::Ampere | Architecture::Ada => tu102::TU102_HAL, + Architecture::Hopper => gh100::GH100_HAL, + Architecture::BlackwellGB10x | Architecture::BlackwellGB20x => gb100::GB100_HAL, + } +} + +/// Converts `base` into the value of the window-base register field. +/// +/// Fails with [`EINVAL`] if `base` is not aligned to the window alignment required by the register +/// field's shift, or if the shifted value does not fit within `RES` bits. +fn window_base<const RES: u32>(base: VramAddress) -> Result<Bounded<u64, RES>> { + const WINDOW_BASE_SHIFT: u32 = 16; + + Bounded::<u64, 64>::from(base.into_raw()) + .shr_exact::<WINDOW_BASE_SHIFT, { 64 - WINDOW_BASE_SHIFT }>() + .and_then(Bounded::try_shrink) + .ok_or(EINVAL) +} diff --git a/drivers/gpu/nova-core/mm/hal/gb100.rs b/drivers/gpu/nova-core/mm/hal/gb100.rs new file mode 100644 index 000000000000..3781e143dea7 --- /dev/null +++ b/drivers/gpu/nova-core/mm/hal/gb100.rs @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Blackwell GB10x/GB20x memory management HAL. + +use kernel::{ + io::Io, + prelude::*, // +}; + +use crate::{ + driver::Bar0, + mm::{ + hal::{ + window_base, + MmHal, // + }, + regs, + VramAddress, // + }, +}; + +struct Gb100; + +impl MmHal for Gb100 { + fn write_pramin_window_base(&self, bar: Bar0<'_>, base: VramAddress) -> Result { + bar.write_reg( + regs::gb100::NV_XAL_EP_BAR0_WINDOW::zeroed().with_base(window_base(base)?.cast()), + ); + Ok(()) + } +} + +const GB100: Gb100 = Gb100; +pub(super) const GB100_HAL: &dyn MmHal = &GB100; diff --git a/drivers/gpu/nova-core/mm/hal/gh100.rs b/drivers/gpu/nova-core/mm/hal/gh100.rs new file mode 100644 index 000000000000..8af384db2921 --- /dev/null +++ b/drivers/gpu/nova-core/mm/hal/gh100.rs @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Hopper memory management HAL. + +use kernel::{ + io::Io, + prelude::*, // +}; + +use crate::{ + driver::Bar0, + mm::{ + hal::{ + window_base, + MmHal, // + }, + regs, + VramAddress, // + }, +}; + +struct Gh100; + +impl MmHal for Gh100 { + fn write_pramin_window_base(&self, bar: Bar0<'_>, base: VramAddress) -> Result { + bar.write_reg( + regs::gh100::NV_XAL_EP_BAR0_WINDOW::zeroed().with_base(window_base(base)?.cast()), + ); + Ok(()) + } +} + +const GH100: Gh100 = Gh100; +pub(super) const GH100_HAL: &dyn MmHal = &GH100; diff --git a/drivers/gpu/nova-core/mm/hal/tu102.rs b/drivers/gpu/nova-core/mm/hal/tu102.rs new file mode 100644 index 000000000000..e4fe7561223c --- /dev/null +++ b/drivers/gpu/nova-core/mm/hal/tu102.rs @@ -0,0 +1,37 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Turing, Ampere and Ada memory management HAL. + +use kernel::{ + io::Io, + prelude::*, // +}; + +use crate::{ + driver::Bar0, + mm::{ + hal::{ + window_base, + MmHal, // + }, + regs, + VramAddress, // + }, +}; + +struct Tu102; + +impl MmHal for Tu102 { + fn write_pramin_window_base(&self, bar: Bar0<'_>, base: VramAddress) -> Result { + bar.write_reg( + regs::NV_PBUS_BAR0_WINDOW::zeroed() + .with_target(regs::Bar0WindowTarget::VidMem) + .with_base(window_base(base)?.cast()), + ); + Ok(()) + } +} + +const TU102: Tu102 = Tu102; +pub(super) const TU102_HAL: &dyn MmHal = &TU102; diff --git a/drivers/gpu/nova-core/mm/pagetable.rs b/drivers/gpu/nova-core/mm/pagetable.rs new file mode 100644 index 000000000000..63a7e1855caa --- /dev/null +++ b/drivers/gpu/nova-core/mm/pagetable.rs @@ -0,0 +1,424 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! Common page table types shared between MMU v2 and v3. +//! +//! This module provides foundational types used by both MMU versions: +//! - Page table level hierarchy +//! - Memory aperture types for PDEs and PTEs + +#![expect(dead_code)] + +pub(super) mod map; +pub(super) mod ver2; +pub(super) mod ver3; +pub(super) mod walk; + +use kernel::{ + io::Io, + num::Bounded, + prelude::*, // +}; + +use crate::{ + gpu::Architecture, + mm::{ + pramin, + Pfn, + VirtualAddress, + VramAddress, // + }, +}; + +/// Extracts the page table index at a given level from a virtual address. +pub(super) trait VaLevelIndex { + /// Return the page table index at `level` for this virtual address. + fn level_index(&self, level: u64) -> u64; +} + +/// MMU version enumeration. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum MmuVersion { + /// MMU v2 for Turing/Ampere/Ada. + V2, + /// MMU v3 for Hopper and later. + V3, +} + +impl From<Architecture> for MmuVersion { + fn from(arch: Architecture) -> Self { + match arch { + Architecture::Turing | Architecture::Ampere | Architecture::Ada => Self::V2, + Architecture::Hopper | Architecture::BlackwellGB10x | Architecture::BlackwellGB20x => { + Self::V3 + } + } + } +} + +/// Page Table Level hierarchy for MMU v2/v3. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum PageTableLevel { + /// Level 0 - Page Directory Base (root). + Pdb, + /// Level 1 - Intermediate page directory. + L1, + /// Level 2 - Intermediate page directory. + L2, + /// Level 3 - Intermediate page directory or dual PDE (version-dependent). + L3, + /// Level 4 - PTE level for v2, intermediate page directory for v3. + L4, + /// Level 5 - PTE level used for MMU v3 only. + L5, +} + +impl PageTableLevel { + /// Number of entries per page table (512 for 4KB pages). + pub(super) const ENTRIES_PER_TABLE: usize = 512; + + /// Get the next level in the hierarchy. + pub(super) const fn next(&self) -> Option<PageTableLevel> { + match self { + Self::Pdb => Some(Self::L1), + Self::L1 => Some(Self::L2), + Self::L2 => Some(Self::L3), + Self::L3 => Some(Self::L4), + Self::L4 => Some(Self::L5), + Self::L5 => None, + } + } + + /// Convert level to index. + pub(super) const fn as_index(&self) -> u64 { + match self { + Self::Pdb => 0, + Self::L1 => 1, + Self::L2 => 2, + Self::L3 => 3, + Self::L4 => 4, + Self::L5 => 5, + } + } +} + +// Trait abstractions for page table operations. + +/// Operations on Page Table Entries (`PTE`s). +pub(super) trait PteOps: Copy + core::fmt::Debug + Into<u64> { + /// Create a `PTE` from a raw `u64` value. + fn from_raw(val: u64) -> Self; + + /// Create an invalid `PTE`. + fn invalid() -> Self; + + /// Create a valid `PTE` for the given memory aperture. + fn new(aperture: AperturePte, pfn: Pfn, writable: bool) -> Self; + + /// Check if this `PTE` is valid. + fn is_valid(&self) -> bool; + + /// Get the physical frame number. + fn frame_number(&self) -> Pfn; + + /// Read a `PTE` from VRAM. + fn read(pramin: &mut pramin::Pramin<'_>, addr: VramAddress) -> Result<Self> { + let val = pramin.window_at::<u64>(addr)?.view().read_val(); + Ok(Self::from_raw(val)) + } + + /// Write this `PTE` to VRAM. + fn write(&self, pramin: &mut pramin::Pramin<'_>, addr: VramAddress) -> Result { + pramin + .window_at::<u64>(addr)? + .view() + .write_val((*self).into()); + Ok(()) + } +} + +/// Operations on Page Directory Entries (`PDE`s). +pub(super) trait PdeOps: Copy + core::fmt::Debug + Into<u64> { + /// Create a `PDE` from a raw `u64` value. + fn from_raw(val: u64) -> Self; + + /// Create a valid `PDE` pointing to a page table in the given aperture. + fn new(aperture: AperturePde, table_pfn: Pfn) -> Self; + + /// Create an invalid `PDE`. + fn invalid() -> Self; + + /// Check if this `PDE` is valid. + fn is_valid(&self) -> bool; + + /// Get the memory aperture of this `PDE`. + fn aperture(&self) -> AperturePde; + + /// Get the VRAM address of the page table. + fn table_vram_address(&self) -> VramAddress; + + /// Read a `PDE` from VRAM. + fn read(pramin: &mut pramin::Pramin<'_>, addr: VramAddress) -> Result<Self> { + let val = pramin.window_at::<u64>(addr)?.view().read_val(); + Ok(Self::from_raw(val)) + } + + /// Write this `PDE` to VRAM. + fn write(&self, pramin: &mut pramin::Pramin<'_>, addr: VramAddress) -> Result { + pramin + .window_at::<u64>(addr)? + .view() + .write_val((*self).into()); + Ok(()) + } + + /// Check if this `PDE` is valid and points to video memory. + fn is_valid_vram(&self) -> bool { + self.is_valid() && self.aperture() == AperturePde::VideoMemory + } +} + +/// Operations on Dual Page Directory Entries (128-bit `DualPde`s). +pub(super) trait DualPdeOps: Copy + core::fmt::Debug { + /// Create a `DualPde` from raw 128-bit value (two `u64`s). + fn from_raw(big: u64, small: u64) -> Self; + + /// Create a `DualPde` with only the small page table pointer set. + fn new_small(table_pfn: Pfn) -> Self; + + /// Check if the small page table pointer is valid. + fn has_small(&self) -> bool; + + /// Get the small page table VRAM address. + fn small_vram_address(&self) -> VramAddress; + + /// Get the raw `u64` value of the big PDE. + fn big_raw_u64(&self) -> u64; + + /// Get the raw `u64` value of the small PDE. + fn small_raw_u64(&self) -> u64; + + /// Read a dual PDE (128-bit) from VRAM. + fn read(pramin: &mut pramin::Pramin<'_>, addr: VramAddress) -> Result<Self> { + let lo = pramin.window_at::<u64>(addr)?.view().read_val(); + let hi = pramin.window_at::<u64>(addr + 8)?.view().read_val(); + Ok(Self::from_raw(lo, hi)) + } + + /// Write this dual PDE (128-bit) to VRAM. + fn write(&self, pramin: &mut pramin::Pramin<'_>, addr: VramAddress) -> Result { + pramin + .window_at::<u64>(addr)? + .view() + .write_val(self.big_raw_u64()); + pramin + .window_at::<u64>(addr + 8)? + .view() + .write_val(self.small_raw_u64()); + Ok(()) + } +} + +/// MMU configuration trait -- encodes version-specific constants and types. +pub(super) trait MmuConfig: 'static { + /// Page Table Entry type. + type Pte: PteOps; + /// Page Directory Entry type. + type Pde: PdeOps; + /// Dual Page Directory Entry type (128-bit). + type DualPde: DualPdeOps; + + /// PDE levels (excluding PTE level) for page table walking. + const PDE_LEVELS: &'static [PageTableLevel]; + /// PTE level for this MMU version. + const PTE_LEVEL: PageTableLevel; + /// Dual PDE level (128-bit entries) for this MMU version. + const DUAL_PDE_LEVEL: PageTableLevel; + + /// Get the number of entries per page table page for a given level. + fn entries_per_page(level: PageTableLevel) -> usize; + + /// Extract the page table index at `level` from `va`. + fn level_index(va: VirtualAddress, level: u64) -> u64; + + /// Get the entry size in bytes for a given level. + fn entry_size(level: PageTableLevel) -> usize { + if level == Self::DUAL_PDE_LEVEL { + 16 // 128-bit dual PDE + } else { + 8 // 64-bit PDE/PTE + } + } + + /// Compute upper bound on page table pages needed for `num_virt_pages`. + /// + /// Walks from PTE level up through PDE levels, accumulating the tree. + fn pt_pages_upper_bound(num_virt_pages: usize) -> usize { + let mut total = 0; + + // PTE pages at the leaf level. + let pte_epp = Self::entries_per_page(Self::PTE_LEVEL); + let mut pages_at_level = num_virt_pages.div_ceil(pte_epp); + total += pages_at_level; + + // Walk PDE levels bottom-up (reverse of PDE_LEVELS). + for &level in Self::PDE_LEVELS.iter().rev() { + let epp = Self::entries_per_page(level); + + // How many pages at this level do we need to point to + // the previous pages_at_level? + pages_at_level = pages_at_level.div_ceil(epp); + total += pages_at_level; + } + + total + } +} + +/// Marker struct for MMU v2 (Turing/Ampere/Ada). +pub(super) struct MmuV2; + +impl MmuConfig for MmuV2 { + type Pte = ver2::Pte; + type Pde = ver2::Pde; + type DualPde = ver2::DualPde; + + const PDE_LEVELS: &'static [PageTableLevel] = ver2::PDE_LEVELS; + const PTE_LEVEL: PageTableLevel = ver2::PTE_LEVEL; + const DUAL_PDE_LEVEL: PageTableLevel = ver2::DUAL_PDE_LEVEL; + + fn entries_per_page(level: PageTableLevel) -> usize { + // TODO: Calculate these values from the bitfield dynamically + // instead of hardcoding them. + match level { + PageTableLevel::Pdb => 4, // PD3 root: bits [48:47] = 2 bits + PageTableLevel::L3 => 256, // PD0 dual: bits [28:21] = 8 bits + _ => 512, // PD2, PD1, PT: 9 bits each + } + } + + fn level_index(va: VirtualAddress, level: u64) -> u64 { + ver2::VirtualAddressV2::new(va).level_index(level) + } +} + +/// Marker struct for MMU v3 (Hopper and later). +pub(super) struct MmuV3; + +impl MmuConfig for MmuV3 { + type Pte = ver3::Pte; + type Pde = ver3::Pde; + type DualPde = ver3::DualPde; + + const PDE_LEVELS: &'static [PageTableLevel] = ver3::PDE_LEVELS; + const PTE_LEVEL: PageTableLevel = ver3::PTE_LEVEL; + const DUAL_PDE_LEVEL: PageTableLevel = ver3::DUAL_PDE_LEVEL; + + fn entries_per_page(level: PageTableLevel) -> usize { + match level { + PageTableLevel::Pdb => 2, // PDE4 root: bit [56] = 1 bit, 2 entries + PageTableLevel::L4 => 256, // PDE0 dual: bits [28:21] = 8 bits + _ => 512, // PDE3, PDE2, PDE1, PT: 9 bits each + } + } + + fn level_index(va: VirtualAddress, level: u64) -> u64 { + ver3::VirtualAddressV3::new(va).level_index(level) + } +} + +/// Memory aperture for Page Table Entries (`PTE`s). +/// +/// Determines which memory region the `PTE` points to. +#[repr(u8)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub(super) enum AperturePte { + /// Local video memory (VRAM). + #[default] + VideoMemory = 0, + /// Peer GPU's video memory. + PeerMemory = 1, + /// System memory with cache coherence. + SystemCoherent = 2, + /// System memory without cache coherence. + SystemNonCoherent = 3, +} + +// TODO[FPRI]: Replace with `#[derive(FromPrimitive)]` when available. +impl From<Bounded<u64, 2>> for AperturePte { + fn from(val: Bounded<u64, 2>) -> Self { + match *val { + 0 => Self::VideoMemory, + 1 => Self::PeerMemory, + 2 => Self::SystemCoherent, + 3 => Self::SystemNonCoherent, + _ => Self::VideoMemory, + } + } +} + +// TODO[FPRI]: Replace with `#[derive(ToPrimitive)]` when available. +impl From<AperturePte> for Bounded<u64, 2> { + fn from(val: AperturePte) -> Self { + Bounded::from_expr(val as u64 & 0x3) + } +} + +/// Memory aperture for Page Directory Entries (`PDE`s). +/// +/// Note: For `PDE`s, `Invalid` (0) means the entry is not valid. +#[repr(u8)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub(super) enum AperturePde { + /// Invalid/unused entry. + #[default] + Invalid = 0, + /// Page table is in video memory. + VideoMemory = 1, + /// Page table is in system memory with coherence. + SystemCoherent = 2, + /// Page table is in system memory without coherence. + SystemNonCoherent = 3, +} + +// TODO[FPRI]: Replace with `#[derive(FromPrimitive)]` when available. +impl From<Bounded<u64, 2>> for AperturePde { + fn from(val: Bounded<u64, 2>) -> Self { + match *val { + 1 => Self::VideoMemory, + 2 => Self::SystemCoherent, + 3 => Self::SystemNonCoherent, + _ => Self::Invalid, + } + } +} + +// TODO[FPRI]: Replace with `#[derive(ToPrimitive)]` when available. +impl From<AperturePde> for Bounded<u64, 2> { + fn from(val: AperturePde) -> Self { + Bounded::from_expr(val as u64 & 0x3) + } +} + +/// Check if the PDB has valid, VRAM-backed page tables. +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +fn check_pdb_inner<M: MmuConfig>(pramin: &mut pramin::Pramin<'_>, pdb_addr: VramAddress) -> Result { + let raw = pramin.window_at::<u64>(pdb_addr)?.view().read_val(); + + if !M::Pde::from_raw(raw).is_valid_vram() { + return Err(ENOENT); + } + Ok(()) +} + +/// Check if the PDB has valid, VRAM-backed page tables, dispatching by MMU version. +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +pub(super) fn check_pdb_valid( + pramin: &mut pramin::Pramin<'_>, + pdb_addr: VramAddress, + chipset: crate::gpu::Chipset, +) -> Result { + match MmuVersion::from(chipset.arch()) { + MmuVersion::V2 => check_pdb_inner::<MmuV2>(pramin, pdb_addr), + MmuVersion::V3 => check_pdb_inner::<MmuV3>(pramin, pdb_addr), + } +} diff --git a/drivers/gpu/nova-core/mm/pagetable/map.rs b/drivers/gpu/nova-core/mm/pagetable/map.rs new file mode 100644 index 000000000000..77431c509a89 --- /dev/null +++ b/drivers/gpu/nova-core/mm/pagetable/map.rs @@ -0,0 +1,345 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! Page table mapping operations for NVIDIA GPUs. + +use core::marker::PhantomData; + +use kernel::{ + gpu::buddy::{ + AllocatedBlocks, + GpuBuddyAllocFlags, + GpuBuddyAllocMode, // + }, + io::io_write, + prelude::*, + ptr::Alignment, + rbtree::{ + RBTree, + RBTreeNode, // + }, + sizes::SZ_4K, // +}; + +use super::{ + walk::{ + PtWalkInner, + WalkPdeResult, + WalkResult, // + }, + AperturePde, + AperturePte, + DualPdeOps, + MmuConfig, + MmuV2, + MmuV3, + MmuVersion, + PageTableLevel, + PdeOps, + PteOps, // +}; +use crate::{ + mm::{ + GpuMm, + Pfn, + Vfn, + VramAddress, + PAGE_SIZE, // + }, + num::{ + IntoSafeCast, // + }, +}; + +/// A pre-allocated and zeroed page table page. +/// +/// Created during the mapping prepare phase and consumed during the execute phase. +/// Stored in an [`RBTree`] keyed by the PDE slot address (`install_addr`). +pub(in crate::mm) struct PreparedPtPage { + /// The allocated and zeroed page table page. + pub(in crate::mm) alloc: Pin<KBox<AllocatedBlocks>>, + /// Page table level -- needed to determine if this PT page is for a dual PDE. + pub(in crate::mm) level: PageTableLevel, +} + +/// Page table mapper. +pub(in crate::mm) struct PtMapInner<M: MmuConfig> { + walker: PtWalkInner<M>, + pdb_addr: VramAddress, + _phantom: PhantomData<M>, +} + +impl<M: MmuConfig> PtMapInner<M> { + /// Create a new [`PtMapInner`]. + pub(super) fn new(pdb_addr: VramAddress) -> Self { + Self { + walker: PtWalkInner::<M>::new(pdb_addr), + pdb_addr, + _phantom: PhantomData, + } + } + + /// Allocate and zero a physical page table page. + fn alloc_and_zero_page(mm: &mut GpuMm<'_>, level: PageTableLevel) -> Result<PreparedPtPage> { + let blocks = KBox::pin_init( + mm.buddy().alloc_blocks( + GpuBuddyAllocMode::Simple, + SZ_4K.into_safe_cast(), + Alignment::new::<SZ_4K>(), + GpuBuddyAllocFlags::default(), + ), + GFP_KERNEL, + )?; + + let page_vram = VramAddress::from_raw(blocks.iter().next().ok_or(ENOMEM)?.offset()); + + // Zero via PRAMIN. + let window = mm + .pramin_mut() + .window_at::<[u64; PAGE_SIZE / 8]>(page_vram)?; + for i in 0..PAGE_SIZE / 8 { + io_write!(window.view(), [build: i], 0); + } + + Ok(PreparedPtPage { + alloc: blocks, + level, + }) + } + + /// Ensure all intermediate page table pages exist for a single VFN. + /// + /// The mutable PRAMIN borrow ends before each allocation. + fn ensure_single_pte_path( + &self, + mm: &mut GpuMm<'_>, + vfn: Vfn, + pt_pages: &mut RBTree<VramAddress, PreparedPtPage>, + ) -> Result { + let max_iter = 2 * M::PDE_LEVELS.len(); + + for _ in 0..max_iter { + let result = self + .walker + .walk_pde_levels(mm.pramin_mut(), vfn, |install_addr| { + pt_pages.get(&install_addr).and_then(|p| { + p.alloc + .iter() + .next() + .map(|b| VramAddress::from_raw(b.offset())) + }) + })?; + + match result { + WalkPdeResult::Complete { .. } => { + return Ok(()); + } + WalkPdeResult::Missing { + install_addr, + level, + } => { + let page = Self::alloc_and_zero_page(mm, level)?; + let node = RBTreeNode::new(install_addr, page, GFP_KERNEL)?; + let old = pt_pages.insert(node); + if old.is_some() { + kernel::pr_warn_once!( + "VMM: duplicate install_addr in pt_pages (internal consistency error)\n" + ); + return Err(EIO); + } + } + } + } + + kernel::pr_warn!( + "VMM: ensure_pte_path: loop exhausted after {} iters (VFN {:?})\n", + max_iter, + vfn + ); + Err(EIO) + } + + /// Prepare page table resources for mapping `num_pages` pages starting at `vfn_start`. + /// + /// Reserves capacity in `page_table_allocs`, then walks the hierarchy + /// per-VFN to prepare pages for all missing PDEs. + pub(super) fn prepare_map( + &self, + mm: &mut GpuMm<'_>, + vfn_start: Vfn, + num_pages: usize, + page_table_allocs: &mut KVec<Pin<KBox<AllocatedBlocks>>>, + pt_pages: &mut RBTree<VramAddress, PreparedPtPage>, + ) -> Result { + // Pre-reserve so install_mappings() can use push_within_capacity (no alloc + // in fence signalling critical path). + let pt_upper_bound = M::pt_pages_upper_bound(num_pages); + page_table_allocs.reserve(pt_upper_bound, GFP_KERNEL)?; + + // Walk the hierarchy per-VFN to prepare pages for all missing PDEs. + for i in 0..num_pages { + let i_u64: u64 = i.into_safe_cast(); + let vfn = Vfn::new(vfn_start.raw() + i_u64); + self.ensure_single_pte_path(mm, vfn, pt_pages)?; + } + Ok(()) + } + + /// Install prepared PDEs and write PTEs, then flush TLB. + /// + /// Drains `pt_pages` and moves allocations into `page_table_allocs`. + pub(super) fn install_mappings( + &self, + mm: &mut GpuMm<'_>, + pt_pages: &mut RBTree<VramAddress, PreparedPtPage>, + page_table_allocs: &mut KVec<Pin<KBox<AllocatedBlocks>>>, + vfn_start: Vfn, + pfns: &[Pfn], + writable: bool, + ) -> Result { + { + let pramin = mm.pramin_mut(); + + // Drain prepared PT pages, install all pending PDEs. + let mut cursor = pt_pages.cursor_front_mut(); + while let Some(c) = cursor { + let (next, node) = c.remove_current(); + let (install_addr, page) = node.to_key_value(); + let page_vram = + VramAddress::from_raw(page.alloc.iter().next().ok_or(ENOMEM)?.offset()); + + if page.level == M::DUAL_PDE_LEVEL { + let new_dpde = M::DualPde::new_small(Pfn::from(page_vram)); + new_dpde.write(pramin, install_addr)?; + } else { + let new_pde = M::Pde::new(AperturePde::VideoMemory, Pfn::from(page_vram)); + new_pde.write(pramin, install_addr)?; + } + + page_table_allocs + .push_within_capacity(page.alloc) + .map_err(|_| ENOMEM)?; + + cursor = next; + } + + // Write PTEs (all PDEs now installed in HW). + for (i, &pfn) in pfns.iter().enumerate() { + let i_u64: u64 = i.into_safe_cast(); + let vfn = Vfn::new(vfn_start.raw() + i_u64); + let result = self.walker.walk_to_pte_lookup_with_window(pramin, vfn)?; + + match result { + WalkResult::Unmapped { pte_addr } | WalkResult::Mapped { pte_addr, .. } => { + let pte = M::Pte::new(AperturePte::VideoMemory, pfn, writable); + pte.write(pramin, pte_addr)?; + } + WalkResult::PageTableMissing => { + kernel::pr_warn_once!("VMM: page table missing for VFN {vfn:?}\n"); + return Err(EIO); + } + } + } + } + + // Flush TLB. + mm.tlb().flush(self.pdb_addr) + } + + /// Invalidate PTEs for a range and flush TLB. + pub(super) fn invalidate_ptes( + &self, + mm: &mut GpuMm<'_>, + vfn_start: Vfn, + num_pages: usize, + ) -> Result { + let invalid_pte = M::Pte::invalid(); + + { + let pramin = mm.pramin_mut(); + for i in 0..num_pages { + let i_u64: u64 = i.into_safe_cast(); + let vfn = Vfn::new(vfn_start.raw() + i_u64); + let result = self.walker.walk_to_pte_lookup_with_window(pramin, vfn)?; + + match result { + WalkResult::Mapped { pte_addr, .. } | WalkResult::Unmapped { pte_addr } => { + invalid_pte.write(pramin, pte_addr)?; + } + WalkResult::PageTableMissing => { + continue; + } + } + } + } + + mm.tlb().flush(self.pdb_addr) + } +} + +macro_rules! pt_map_dispatch { + ($self:expr, $method:ident ( $($arg:expr),* $(,)? )) => { + match $self { + PtMap::V2(inner) => inner.$method($($arg),*), + PtMap::V3(inner) => inner.$method($($arg),*), + } + }; +} + +/// Page table mapper dispatch. +pub(in crate::mm) enum PtMap { + /// MMU v2 (Turing/Ampere/Ada). + V2(PtMapInner<MmuV2>), + /// MMU v3 (Hopper+). + V3(PtMapInner<MmuV3>), +} + +impl PtMap { + /// Create a new page table mapper for the given MMU version. + pub(in crate::mm) fn new(pdb_addr: VramAddress, version: MmuVersion) -> Self { + match version { + MmuVersion::V2 => Self::V2(PtMapInner::<MmuV2>::new(pdb_addr)), + MmuVersion::V3 => Self::V3(PtMapInner::<MmuV3>::new(pdb_addr)), + } + } + + /// Prepare page table resources for a mapping. + pub(in crate::mm) fn prepare_map( + &self, + mm: &mut GpuMm<'_>, + vfn_start: Vfn, + num_pages: usize, + page_table_allocs: &mut KVec<Pin<KBox<AllocatedBlocks>>>, + pt_pages: &mut RBTree<VramAddress, PreparedPtPage>, + ) -> Result { + pt_map_dispatch!( + self, + prepare_map(mm, vfn_start, num_pages, page_table_allocs, pt_pages) + ) + } + + /// Install prepared PDEs and write PTEs, then flush TLB. + pub(in crate::mm) fn install_mappings( + &self, + mm: &mut GpuMm<'_>, + pt_pages: &mut RBTree<VramAddress, PreparedPtPage>, + page_table_allocs: &mut KVec<Pin<KBox<AllocatedBlocks>>>, + vfn_start: Vfn, + pfns: &[Pfn], + writable: bool, + ) -> Result { + pt_map_dispatch!( + self, + install_mappings(mm, pt_pages, page_table_allocs, vfn_start, pfns, writable) + ) + } + + /// Invalidate PTEs for a range and flush TLB. + pub(in crate::mm) fn invalidate_ptes( + &self, + mm: &mut GpuMm<'_>, + vfn_start: Vfn, + num_pages: usize, + ) -> Result { + pt_map_dispatch!(self, invalidate_ptes(mm, vfn_start, num_pages)) + } +} diff --git a/drivers/gpu/nova-core/mm/pagetable/ver2.rs b/drivers/gpu/nova-core/mm/pagetable/ver2.rs new file mode 100644 index 000000000000..d7169a0fcff9 --- /dev/null +++ b/drivers/gpu/nova-core/mm/pagetable/ver2.rs @@ -0,0 +1,275 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! MMU v2 page table types for Turing, Ampere and Ada GPUs. +//! +//! This module defines MMU version 2 specific types (Turing, Ampere and Ada GPUs). +//! +//! Bit field layouts derived from the NVIDIA OpenRM documentation: +//! `open-gpu-kernel-modules/src/common/inc/swref/published/turing/tu102/dev_mmu.h` + +#![allow(dead_code)] + +use kernel::{ + bitfield, + num::Bounded, // +}; + +use pin_init::Zeroable; + +use super::{ + AperturePde, + AperturePte, + DualPdeOps, + PageTableLevel, + PdeOps, + PteOps, + VaLevelIndex, // +}; + +use crate::mm::{ + Pfn, + VirtualAddress, + VramAddress, // +}; + +// Bounded to version 2 Pfn bitfield conversions: +// 25 bits for video memory frame numbers (bits 32:8). +impl_pfn_bounded!(25); +// 46 bits for system memory frame numbers (bits 53:8). +impl_pfn_bounded!(46); + +bitfield! { + /// MMU v2 49-bit virtual address layout. + pub(super) struct VirtualAddressV2(u64) { + /// Page offset [11:0]. + 11:0 offset; + /// PT index [20:12]. + 20:12 pt_idx; + /// PDE0 index [28:21]. + 28:21 pde0_idx; + /// PDE1 index [37:29]. + 37:29 pde1_idx; + /// PDE2 index [46:38]. + 46:38 pde2_idx; + /// PDE3 index [48:47]. + 48:47 pde3_idx; + } +} + +impl VirtualAddressV2 { + /// Create a [`VirtualAddressV2`] from a [`VirtualAddress`]. + pub(super) fn new(va: VirtualAddress) -> Self { + Self::from_raw(va.into_raw()) + } +} + +impl VaLevelIndex for VirtualAddressV2 { + fn level_index(&self, level: u64) -> u64 { + match level { + 0 => *self.pde3_idx(), + 1 => *self.pde2_idx(), + 2 => *self.pde1_idx(), + 3 => *self.pde0_idx(), + 4 => *self.pt_idx(), + _ => 0, + } + } +} + +/// `PDE` levels for MMU v2 (5-level hierarchy: `PDB` -> `L1` -> `L2` -> `L3` -> `L4`). +pub(super) const PDE_LEVELS: &[PageTableLevel] = &[ + PageTableLevel::Pdb, + PageTableLevel::L1, + PageTableLevel::L2, + PageTableLevel::L3, +]; + +/// `PTE` level for MMU v2. +pub(super) const PTE_LEVEL: PageTableLevel = PageTableLevel::L4; + +/// Dual `PDE` level for MMU v2 (128-bit entries). +pub(super) const DUAL_PDE_LEVEL: PageTableLevel = PageTableLevel::L3; + +// Page Table Entry (PTE) for MMU v2 - 64-bit entry at level 4. +bitfield! { + /// Page Table Entry for MMU v2. + pub(in crate::mm) struct Pte(u64) { + /// Entry is valid. + 0:0 valid; + /// Memory aperture type. + 2:1 aperture => AperturePte; + /// Volatile (bypass L2 cache). + 3:3 volatile; + /// Encryption enabled (Confidential Computing). + 4:4 encrypted; + /// Privileged access only. + 5:5 privilege; + /// Write protection. + 6:6 read_only; + /// Atomic operations disabled. + 7:7 atomic_disable; + /// Frame number for system memory. + 53:8 frame_number_sys => Pfn; + /// Frame number for video memory. + 32:8 frame_number_vid => Pfn; + /// Peer GPU ID for peer memory (0-7). + 35:33 peer_id; + /// Compression tag line bits. + 53:36 comptagline; + /// Surface kind/format. + 63:56 kind; + } +} + +impl PteOps for Pte { + fn from_raw(val: u64) -> Self { + Self::from_raw(val) + } + + fn invalid() -> Self { + Self::zeroed() + } + + fn new(aperture: AperturePte, pfn: Pfn, writable: bool) -> Self { + let base = Self::zeroed() + .with_valid(true) + .with_aperture(aperture) + .with_read_only(!writable); + match aperture { + AperturePte::VideoMemory => base.with_frame_number_vid(pfn), + // Sysmem PTEs use VOL=1 to bypass L2 for cache coherency. + AperturePte::SystemCoherent => base.with_frame_number_sys(pfn).with_volatile(true), + AperturePte::PeerMemory | AperturePte::SystemNonCoherent => { + kernel::pr_warn!("MMU v2 PTE aperture {:?} not supported\n", aperture); + Self::invalid() + } + } + } + + fn is_valid(&self) -> bool { + self.valid().into_bool() + } + + fn frame_number(&self) -> Pfn { + match self.aperture() { + AperturePte::VideoMemory => self.frame_number_vid(), + _ => self.frame_number_sys(), + } + } +} + +// Page Directory Entry (PDE) for MMU v2 - 64-bit entry at levels 0-2. +bitfield! { + /// Page Directory Entry for MMU v2. + pub(in crate::mm) struct Pde(u64) { + /// Valid bit (inverted logic). + 0:0 valid_inverted; + /// Memory aperture type. + 2:1 aperture => AperturePde; + /// Volatile (bypass L2 cache). + 3:3 volatile; + /// Disable Address Translation Services. + 5:5 no_ats; + /// Table frame number for system memory. + 53:8 table_frame_sys => Pfn; + /// Table frame number for video memory. + 32:8 table_frame_vid => Pfn; + /// Peer GPU ID (0-7). + 35:33 peer_id; + } +} + +impl PdeOps for Pde { + fn from_raw(val: u64) -> Self { + Self::from_raw(val) + } + + fn new(aperture: AperturePde, table_pfn: Pfn) -> Self { + let base = Self::zeroed() + .with_valid_inverted(false) // 0 = valid + .with_aperture(aperture); + match aperture { + AperturePde::VideoMemory => base.with_table_frame_vid(table_pfn), + // Sysmem PTEs use VOL=1 to bypass L2 for cache coherency. + AperturePde::SystemCoherent => base.with_table_frame_sys(table_pfn).with_volatile(true), + AperturePde::Invalid | AperturePde::SystemNonCoherent => { + kernel::pr_warn!("MMU v2 PDE aperture {:?} not supported\n", aperture); + Self::invalid() + } + } + } + + fn invalid() -> Self { + Self::zeroed() + .with_valid_inverted(true) + .with_aperture(AperturePde::Invalid) + } + + fn is_valid(&self) -> bool { + !self.valid_inverted().into_bool() && self.aperture() != AperturePde::Invalid + } + + fn aperture(&self) -> AperturePde { + Pde::aperture(*self) + } + + fn table_vram_address(&self) -> VramAddress { + debug_assert!( + Pde::aperture(*self) == AperturePde::VideoMemory, + "table_vram_address called on non-VRAM PDE (aperture: {:?})", + Pde::aperture(*self) + ); + VramAddress::from(self.table_frame_vid()) + } +} + +/// Dual `PDE` at Level 3 - 128-bit entry of Large/Small Page Table pointers. +/// +/// The dual `PDE` supports both large (64KB) and small (4KB) page tables. +#[repr(C)] +#[derive(Debug, Clone, Copy)] +pub(in crate::mm) struct DualPde { + /// Large/Big Page Table pointer (lower 64 bits). + pub(super) big: Pde, + /// Small Page Table pointer (upper 64 bits). + pub(super) small: Pde, +} + +impl DualPde { + /// Check if the big page table pointer is valid. + fn has_big(&self) -> bool { + PdeOps::is_valid(&self.big) + } +} + +impl DualPdeOps for DualPde { + fn from_raw(big: u64, small: u64) -> Self { + Self { + big: PdeOps::from_raw(big), + small: PdeOps::from_raw(small), + } + } + + fn new_small(table_pfn: Pfn) -> Self { + Self { + big: PdeOps::from_raw(0), + small: PdeOps::new(AperturePde::VideoMemory, table_pfn), + } + } + + fn has_small(&self) -> bool { + PdeOps::is_valid(&self.small) + } + + fn small_vram_address(&self) -> VramAddress { + PdeOps::table_vram_address(&self.small) + } + + fn big_raw_u64(&self) -> u64 { + self.big.into_raw() + } + + fn small_raw_u64(&self) -> u64 { + self.small.into_raw() + } +} diff --git a/drivers/gpu/nova-core/mm/pagetable/ver3.rs b/drivers/gpu/nova-core/mm/pagetable/ver3.rs new file mode 100644 index 000000000000..47ed3339026b --- /dev/null +++ b/drivers/gpu/nova-core/mm/pagetable/ver3.rs @@ -0,0 +1,421 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! MMU v3 page table types for Hopper and later GPUs. +//! +//! This module defines MMU version 3 specific types (Hopper and later GPUs). +//! +//! Key differences from MMU v2: +//! - Unified 40-bit address field for all apertures (v2 had separate sys/vid fields). +//! - PCF (Page Classification Field) replaces separate privilege/RO/atomic/cache bits. +//! - KIND field is 4 bits (not 8). +//! - IS_PTE bit in PDE to support large pages directly. +//! - No COMPTAGLINE field (compression handled differently in v3). +//! - No separate ENCRYPTED bit. +//! +//! Bit field layouts derived from the NVIDIA OpenRM documentation: +//! `open-gpu-kernel-modules/src/common/inc/swref/published/hopper/gh100/dev_mmu.h` + +#![allow(dead_code)] + +use kernel::{ + bitfield, + num::Bounded, + prelude::*, // +}; + +use pin_init::Zeroable; + +use super::{ + AperturePde, + AperturePte, + DualPdeOps, + PageTableLevel, + PdeOps, + PteOps, + VaLevelIndex, // +}; + +use crate::mm::{ + Pfn, + VirtualAddress, + VramAddress, // +}; + +// Bounded to version 3 Pfn conversion. +impl_pfn_bounded!(40); + +bitfield! { + /// MMU v3 57-bit virtual address layout. + pub(super) struct VirtualAddressV3(u64) { + /// Page offset [11:0]. + 11:0 offset; + /// PT index [20:12]. + 20:12 pt_idx; + /// PDE0 index [28:21]. + 28:21 pde0_idx; + /// PDE1 index [37:29]. + 37:29 pde1_idx; + /// PDE2 index [46:38]. + 46:38 pde2_idx; + /// PDE3 index [55:47]. + 55:47 pde3_idx; + /// PDE4 index [56]. + 56:56 pde4_idx; + } +} + +impl VirtualAddressV3 { + /// Create a [`VirtualAddressV3`] from a [`VirtualAddress`]. + pub(super) fn new(va: VirtualAddress) -> Self { + Self::from_raw(va.into_raw()) + } +} + +impl VaLevelIndex for VirtualAddressV3 { + fn level_index(&self, level: u64) -> u64 { + match level { + 0 => *self.pde4_idx(), + 1 => *self.pde3_idx(), + 2 => *self.pde2_idx(), + 3 => *self.pde1_idx(), + 4 => *self.pde0_idx(), + 5 => *self.pt_idx(), + _ => 0, + } + } +} + +/// PDE levels for MMU v3 (6-level hierarchy). +pub(super) const PDE_LEVELS: &[PageTableLevel] = &[ + PageTableLevel::Pdb, + PageTableLevel::L1, + PageTableLevel::L2, + PageTableLevel::L3, + PageTableLevel::L4, +]; + +/// PTE level for MMU v3. +pub(super) const PTE_LEVEL: PageTableLevel = PageTableLevel::L5; + +/// Dual PDE level for MMU v3 (128-bit entries). +pub(super) const DUAL_PDE_LEVEL: PageTableLevel = PageTableLevel::L4; + +bitfield! { + /// Page Classification Field for PTEs (5 bits) in MMU v3. + pub(in crate::mm) struct PtePcf(u8) { + /// Bypass L2 cache (0=cached, 1=bypass). + 0:0 uncached; + /// Access counting disabled (0=enabled, 1=disabled). + 1:1 acd; + /// Read-only access (0=read-write, 1=read-only). + 2:2 read_only; + /// Atomics disabled (0=enabled, 1=disabled). + 3:3 no_atomic; + /// Privileged access only (0=regular, 1=privileged). + 4:4 privileged; + } +} + +impl PtePcf { + /// Create PCF for read-write mapping (cached, no atomics, regular mode). + fn rw() -> Self { + Self::zeroed().with_no_atomic(true) + } + + /// Create PCF for read-only mapping (cached, no atomics, regular mode). + fn ro() -> Self { + Self::zeroed().with_read_only(true).with_no_atomic(true) + } + + /// Get the raw `u8` value. + fn raw_u8(&self) -> u8 { + self.into_raw() + } +} + +impl From<Bounded<u64, 5>> for PtePcf { + fn from(val: Bounded<u64, 5>) -> Self { + Self::from_raw(u8::from(val)) + } +} + +impl From<PtePcf> for Bounded<u64, 5> { + fn from(pcf: PtePcf) -> Self { + Bounded::from_expr(u64::from(pcf.into_raw()) & 0x1F) + } +} + +bitfield! { + /// Page Classification Field for PDEs (3 bits) in MMU v3. + /// + /// Controls Address Translation Services (ATS) and caching. + pub(in crate::mm) struct PdePcf(u8) { + /// Bypass L2 cache (0=cached, 1=bypass). + 0:0 uncached; + /// ATS disabled (0=enabled, 1=disabled). + 1:1 no_ats; + } +} + +impl PdePcf { + /// Create PCF for cached mapping with ATS enabled (default). + fn cached() -> Self { + Self::zeroed() + } + + /// Get the raw `u8` value. + fn raw_u8(&self) -> u8 { + self.into_raw() + } +} + +impl From<Bounded<u64, 3>> for PdePcf { + fn from(val: Bounded<u64, 3>) -> Self { + Self::from_raw(u8::from(val)) + } +} + +impl From<PdePcf> for Bounded<u64, 3> { + fn from(pcf: PdePcf) -> Self { + Bounded::from_expr(u64::from(pcf.into_raw()) & 0x7) + } +} + +bitfield! { + /// Page Table Entry for MMU v3. + pub(in crate::mm) struct Pte(u64) { + /// Entry is valid. + 0:0 valid; + /// Memory aperture type. + 2:1 aperture => AperturePte; + /// Page Classification Field. + 7:3 pcf => PtePcf; + /// Surface kind (4 bits, 0x0=pitch, 0xF=invalid). + 11:8 kind; + /// Physical frame number (for all apertures). + 51:12 frame_number => Pfn; + /// Peer GPU ID for peer memory (0-7). + 63:61 peer_id; + } +} + +impl PteOps for Pte { + fn from_raw(val: u64) -> Self { + Self::from_raw(val) + } + + fn invalid() -> Self { + Self::zeroed() + } + + fn new(aperture: AperturePte, pfn: Pfn, writable: bool) -> Self { + let pcf = match (aperture, writable) { + (AperturePte::VideoMemory, true) => PtePcf::rw(), + (AperturePte::VideoMemory, false) => PtePcf::ro(), + // Sysmem PTEs use uncached+no_atomic PCF for cache coherency. + (AperturePte::SystemCoherent, true) => { + PtePcf::zeroed().with_uncached(true).with_no_atomic(true) + } + (AperturePte::SystemCoherent, false) => PtePcf::zeroed() + .with_uncached(true) + .with_no_atomic(true) + .with_read_only(true), + (AperturePte::PeerMemory | AperturePte::SystemNonCoherent, _) => { + kernel::pr_warn!("MMU v3 PTE aperture {:?} not supported\n", aperture); + return Self::invalid(); + } + }; + Self::zeroed() + .with_valid(true) + .with_aperture(aperture) + .with_pcf(pcf) + .with_frame_number(pfn) + } + + fn is_valid(&self) -> bool { + self.valid().into_bool() + } + + fn frame_number(&self) -> Pfn { + Pte::frame_number(*self) + } +} + +bitfield! { + /// Page Directory Entry for MMU v3 (Hopper+). + /// + /// ## Note + /// + /// v3 uses a unified 40-bit address field (v2 had separate sys/vid address fields). + pub(in crate::mm) struct Pde(u64) { + /// Entry is a PTE (0=PDE, 1=large page PTE). + 0:0 is_pte; + /// Memory aperture type. + 2:1 aperture => AperturePde; + /// Page Classification Field (3 bits for PDE). + 5:3 pcf => PdePcf; + /// Table frame number (40-bit unified address). + 51:12 table_frame => Pfn; + } +} + +impl PdeOps for Pde { + fn from_raw(val: u64) -> Self { + Self::from_raw(val) + } + + fn new(aperture: AperturePde, table_pfn: Pfn) -> Self { + match aperture { + AperturePde::VideoMemory => Self::zeroed() + .with_is_pte(false) + .with_aperture(aperture) + .with_table_frame(table_pfn), + AperturePde::Invalid | AperturePde::SystemCoherent | AperturePde::SystemNonCoherent => { + kernel::pr_warn!("MMU v3 PDE aperture {:?} not supported\n", aperture); + Self::invalid() + } + } + } + + fn invalid() -> Self { + Self::zeroed().with_aperture(AperturePde::Invalid) + } + + fn is_valid(&self) -> bool { + Pde::aperture(*self) != AperturePde::Invalid + } + + fn aperture(&self) -> AperturePde { + Pde::aperture(*self) + } + + fn table_vram_address(&self) -> VramAddress { + debug_assert!( + Pde::aperture(*self) == AperturePde::VideoMemory, + "table_vram_address called on non-VRAM PDE (aperture: {:?})", + Pde::aperture(*self) + ); + VramAddress::from(self.table_frame()) + } +} + +bitfield! { + /// Big Page Table pointer in Dual PDE (MMU v3). + /// + /// 64-bit lower word of the 128-bit Dual PDE. + pub(super) struct DualPdeBig(u64) { + /// Entry is a PTE (for large pages). + 0:0 is_pte; + /// Memory aperture type. + 2:1 aperture => AperturePde; + /// Page Classification Field. + 5:3 pcf => PdePcf; + /// Table frame (table address 256-byte aligned). + 51:8 table_frame; + } +} + +impl DualPdeBig { + /// Create an invalid big page table pointer. + fn invalid() -> Self { + Self::zeroed().with_aperture(AperturePde::Invalid) + } + + /// Create a valid big PDE pointing to a page table in the given aperture. + fn new(aperture: AperturePde, table_addr: VramAddress) -> Result<Self> { + // Big page table addresses must be 256-byte aligned (shift 8). + if table_addr.into_raw() & 0xFF != 0 { + return Err(EINVAL); + } + let table_frame = Bounded::from_expr(table_addr.into_raw() >> 8); + match aperture { + AperturePde::VideoMemory => Ok(Self::zeroed() + .with_is_pte(false) + .with_aperture(aperture) + .with_table_frame(table_frame)), + AperturePde::Invalid | AperturePde::SystemCoherent | AperturePde::SystemNonCoherent => { + kernel::pr_warn!("MMU v3 DualPdeBig aperture {:?} not supported\n", aperture); + Ok(Self::invalid()) + } + } + } + + /// Check if this big PDE is valid. + fn is_valid(&self) -> bool { + self.aperture() != AperturePde::Invalid + } + + /// Get the VRAM address of the big page table. + fn table_vram_address(&self) -> VramAddress { + debug_assert!( + self.aperture() == AperturePde::VideoMemory, + "table_vram_address called on non-VRAM DualPdeBig (aperture: {:?})", + self.aperture() + ); + VramAddress::from_raw(*self.table_frame() << 8) + } +} + +/// Dual PDE at Level 4 for MMU v3 - 128-bit entry. +/// +/// Contains both big (64KB) and small (4KB) page table pointers: +/// - Lower 64 bits: Big Page Table pointer. +/// - Upper 64 bits: Small Page Table pointer. +/// +/// ## Note +/// +/// The big and small page table pointers have different address layouts: +/// - Big address = field value << 8 (256-byte alignment). +/// - Small address = field value << 12 (4KB alignment). +/// +/// This is why `DualPdeBig` is a separate type from `Pde`. +#[repr(C)] +#[derive(Debug, Clone, Copy)] +pub(in crate::mm) struct DualPde { + /// Big Page Table pointer. + pub(super) big: DualPdeBig, + /// Small Page Table pointer. + pub(super) small: Pde, +} + +// SAFETY: Both `DualPdeBig` and `Pde` fields are `Zeroable` (bitfield types are Zeroable). +unsafe impl Zeroable for DualPde {} + +impl DualPde { + /// Check if the big page table pointer is valid. + fn has_big(&self) -> bool { + self.big.is_valid() + } +} + +impl DualPdeOps for DualPde { + fn from_raw(big: u64, small: u64) -> Self { + Self { + big: DualPdeBig::from_raw(big), + small: PdeOps::from_raw(small), + } + } + + fn new_small(table_pfn: Pfn) -> Self { + Self { + big: DualPdeBig::invalid(), + small: PdeOps::new(AperturePde::VideoMemory, table_pfn), + } + } + + fn has_small(&self) -> bool { + PdeOps::is_valid(&self.small) + } + + fn small_vram_address(&self) -> VramAddress { + PdeOps::table_vram_address(&self.small) + } + + fn big_raw_u64(&self) -> u64 { + self.big.into_raw() + } + + fn small_raw_u64(&self) -> u64 { + self.small.into_raw() + } +} diff --git a/drivers/gpu/nova-core/mm/pagetable/walk.rs b/drivers/gpu/nova-core/mm/pagetable/walk.rs new file mode 100644 index 000000000000..76c1729971f5 --- /dev/null +++ b/drivers/gpu/nova-core/mm/pagetable/walk.rs @@ -0,0 +1,244 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! Page table walker implementation for NVIDIA GPUs. +//! +//! This module provides page table walking functionality for MMU v2 and v3. +//! The walker traverses the page table hierarchy to resolve virtual addresses +//! to physical addresses or to find PTE locations. +//! +//! # Page Table Hierarchy +//! +//! ## MMU v2 (Turing/Ampere/Ada) - 5 levels +//! +//! ```text +//! +-------+ +-------+ +-------+ +---------+ +-------+ +//! | PDB |---->| L1 |---->| L2 |---->| L3 Dual |---->| L4 | +//! | (L0) | | | | | | PDE | | (PTE) | +//! +-------+ +-------+ +-------+ +---------+ +-------+ +//! 64-bit 64-bit 64-bit 128-bit 64-bit +//! PDE PDE PDE (big+small) PTE +//! ``` +//! +//! ## MMU v3 (Hopper+) - 6 levels +//! +//! ```text +//! +-------+ +-------+ +-------+ +-------+ +---------+ +-------+ +//! | PDB |---->| L1 |---->| L2 |---->| L3 |---->| L4 Dual |---->| L5 | +//! | (L0) | | | | | | | | PDE | | (PTE) | +//! +-------+ +-------+ +-------+ +-------+ +---------+ +-------+ +//! 64-bit 64-bit 64-bit 64-bit 128-bit 64-bit +//! PDE PDE PDE PDE (big+small) PTE +//! ``` +//! +//! # Result of a page table walk +//! +//! The walker returns a [`WalkResult`] indicating the outcome. + +use core::marker::PhantomData; + +use kernel::prelude::*; + +use super::{ + DualPdeOps, + MmuConfig, + MmuV2, + MmuV3, + MmuVersion, + PageTableLevel, + PdeOps, + PteOps, // +}; +use crate::{ + mm::{ + pramin, + GpuMm, + Pfn, + Vfn, + VirtualAddress, + VramAddress, // + }, + num::{ + IntoSafeCast, // + }, +}; + +/// Result of walking to a PTE. +#[derive(Debug, Clone, Copy)] +pub(in crate::mm) enum WalkResult { + /// Intermediate page tables are missing (only returned in lookup mode). + PageTableMissing, + /// PTE exists but is invalid (page not mapped). + Unmapped { pte_addr: VramAddress }, + /// PTE exists and is valid (page is mapped). + Mapped { pte_addr: VramAddress, pfn: Pfn }, +} + +/// Result of walking PDE levels only. +/// +/// Returned by [`PtWalkInner::walk_pde_levels()`] to indicate whether all PDE +/// levels resolved or a PDE is missing. +#[derive(Debug, Clone, Copy)] +pub(in crate::mm) enum WalkPdeResult { + /// All PDE levels resolved -- returns PTE page table address. + Complete { + /// VRAM address of the PTE-level page table. + pte_table: VramAddress, + }, + /// A PDE is missing and no prepared page was provided by the closure. + Missing { + /// PDE slot address in the parent page table (where to install). + install_addr: VramAddress, + /// The page table level that is missing. + level: PageTableLevel, + }, +} + +/// Page table walker. +pub(in crate::mm) struct PtWalkInner<M: MmuConfig> { + pdb_addr: VramAddress, + _phantom: PhantomData<M>, +} + +impl<M: MmuConfig> PtWalkInner<M> { + /// Calculate the VRAM address of an entry within a page table. + fn entry_addr(table: VramAddress, level: PageTableLevel, index: u64) -> VramAddress { + let entry_size: u64 = M::entry_size(level).into_safe_cast(); + table + index * entry_size + } + + /// Create a new page table walker. + pub(super) fn new(pdb_addr: VramAddress) -> Self { + Self { + pdb_addr, + _phantom: PhantomData, + } + } + + /// Walk PDE levels with closure-based resolution for missing PDEs. + /// + /// Traverses all PDE levels for the MMU version. At each level, reads the PDE. + /// If valid, extracts the child table address and continues. If missing, calls + /// `resolve_prepared(install_addr)` to resolve the missing PDE. + pub(super) fn walk_pde_levels( + &self, + pramin: &mut pramin::Pramin<'_>, + vfn: Vfn, + resolve_prepared: impl Fn(VramAddress) -> Option<VramAddress>, + ) -> Result<WalkPdeResult> { + let va = VirtualAddress::from(vfn); + let mut cur_table = self.pdb_addr; + + for &level in M::PDE_LEVELS { + let idx = M::level_index(va, level.as_index()); + let install_addr = Self::entry_addr(cur_table, level, idx); + + if level == M::DUAL_PDE_LEVEL { + // 128-bit dual PDE with big+small page table pointers. + let dpde = M::DualPde::read(pramin, install_addr)?; + if dpde.has_small() { + cur_table = dpde.small_vram_address(); + continue; + } + } else { + // Regular 64-bit PDE. Use `is_valid_vram()` because + // `table_vram_address()` only reads the VRAM frame-number + // bitfield; system-memory PDEs store the address in a + // different (wider) field and would be silently truncated. + let pde = M::Pde::read(pramin, install_addr)?; + if pde.is_valid_vram() { + cur_table = pde.table_vram_address(); + continue; + } + } + + // PDE missing in HW. Ask caller for resolution. + if let Some(prepared_addr) = resolve_prepared(install_addr) { + cur_table = prepared_addr; + continue; + } + + return Ok(WalkPdeResult::Missing { + install_addr, + level, + }); + } + + Ok(WalkPdeResult::Complete { + pte_table: cur_table, + }) + } + + /// Walk to PTE for lookup only (no allocation). + /// + /// Returns [`WalkResult::PageTableMissing`] if intermediate tables don't exist. + pub(super) fn walk_to_pte_lookup(&self, mm: &mut GpuMm<'_>, vfn: Vfn) -> Result<WalkResult> { + self.walk_to_pte_lookup_with_window(mm.pramin_mut(), vfn) + } + + /// Walk to PTE using a caller-provided PRAMIN manager (lookup only). + pub(super) fn walk_to_pte_lookup_with_window( + &self, + pramin: &mut pramin::Pramin<'_>, + vfn: Vfn, + ) -> Result<WalkResult> { + match self.walk_pde_levels(pramin, vfn, |_| None)? { + WalkPdeResult::Complete { pte_table } => { + Self::read_pte_at_level(pramin, vfn, pte_table) + } + WalkPdeResult::Missing { .. } => Ok(WalkResult::PageTableMissing), + } + } + + /// Read the PTE at the PTE level given the PTE table address. + fn read_pte_at_level( + pramin: &mut pramin::Pramin<'_>, + vfn: Vfn, + pte_table: VramAddress, + ) -> Result<WalkResult> { + let va = VirtualAddress::from(vfn); + let pte_level = M::PTE_LEVEL; + let pte_idx = M::level_index(va, pte_level.as_index()); + let pte_addr = Self::entry_addr(pte_table, pte_level, pte_idx); + let pte = M::Pte::read(pramin, pte_addr)?; + + if pte.is_valid() { + return Ok(WalkResult::Mapped { + pte_addr, + pfn: pte.frame_number(), + }); + } + Ok(WalkResult::Unmapped { pte_addr }) + } +} + +macro_rules! pt_walk_dispatch { + ($self:expr, $method:ident ( $($arg:expr),* $(,)? )) => { + match $self { + PtWalk::V2(inner) => inner.$method($($arg),*), + PtWalk::V3(inner) => inner.$method($($arg),*), + } + }; +} + +/// Page table walker dispatch. +pub(in crate::mm) enum PtWalk { + /// MMU v2 (Turing/Ampere/Ada). + V2(PtWalkInner<MmuV2>), + /// MMU v3 (Hopper+). + V3(PtWalkInner<MmuV3>), +} + +impl PtWalk { + /// Create a new page table walker for the given MMU version. + pub(in crate::mm) fn new(pdb_addr: VramAddress, version: MmuVersion) -> Self { + match version { + MmuVersion::V2 => Self::V2(PtWalkInner::<MmuV2>::new(pdb_addr)), + MmuVersion::V3 => Self::V3(PtWalkInner::<MmuV3>::new(pdb_addr)), + } + } + + /// Walk to PTE for lookup. + pub(in crate::mm) fn walk_to_pte(&self, mm: &mut GpuMm<'_>, vfn: Vfn) -> Result<WalkResult> { + pt_walk_dispatch!(self, walk_to_pte_lookup(mm, vfn)) + } +} diff --git a/drivers/gpu/nova-core/mm/pramin.rs b/drivers/gpu/nova-core/mm/pramin.rs new file mode 100644 index 000000000000..7f89c093d591 --- /dev/null +++ b/drivers/gpu/nova-core/mm/pramin.rs @@ -0,0 +1,312 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Utilities for accessing VRAM through the PRAMIN window. + +use core::ops::Range; + +use kernel::{ + io::{ + io_project, + register, + register::OffsetLoc, + Io, + Mmio, // + }, + prelude::*, + ptr::{ + Alignable, + Alignment, // + }, + sizes::{ + SZ_1M, + SZ_64K, // + }, +}; + +use crate::{ + driver::{ + Bar0, + NovaRegisters, // + }, + gpu::Chipset, + mm::{ + hal::{ + self, + MmHal, // + }, + VramAddress, // + }, + num::IntoSafeCast, // +}; + +/// Size of the PRAMIN window (1 MiB). +const WINDOW_SIZE: usize = SZ_1M; + +/// The PRAMIN window, which is a 1 MiB window into VRAM at a fixed BAR0 offset. +#[derive(FromBytes, IntoBytes)] +struct PraminWindow([u8; WINDOW_SIZE]); + +register! { + base: NovaRegisters; + + /// Location of the window inside BAR0. + PRAMIN: PraminWindow @ 0x700000; +} + +/// Owner of the PRAMIN window state. +/// +/// [`Pramin::window_at()`] repositions the window as needed and returns a typed MMIO view into +/// it, holding the manager borrowed for the lifetime of the view. +pub(super) struct Pramin<'gpu> { + bar: Bar0<'gpu>, + hal: &'static dyn MmHal, + /// MMIO view of the PRAMIN window in BAR0. + window: Mmio<'gpu, PraminWindow>, + /// VRAM range to keep the PRAMIN window inside. + vram_range: Range<VramAddress>, + /// Cached window position. + window_range: Range<VramAddress>, +} + +/// Typed view of VRAM through the PRAMIN window. +/// +/// Inserts an ordering point after previous writes through the window on drop. Views returned +/// by [`PraminAccess::view()`] cannot outlive this access, so the ordering point covers every +/// write made through them. +pub(super) struct PraminAccess<'a, T> +where + T: FromBytes + IntoBytes, +{ + view: Mmio<'a, T>, +} + +impl<T> PraminAccess<'_, T> +where + T: FromBytes + IntoBytes, +{ + /// Returns the MMIO view of the accessed location. + pub(super) fn view(&self) -> Mmio<'_, T> { + self.view + } +} + +impl<T> Drop for PraminAccess<'_, T> +where + T: FromBytes + IntoBytes, +{ + fn drop(&mut self) { + // Insert an ordering point after previous writes through this window. + self.view.cast::<u8>().read_val(); + } +} + +impl<'gpu> Pramin<'gpu> { + /// Alignment required by the PRAMIN window. + const BASE_ALIGN: Alignment = Alignment::new::<SZ_64K>(); + + /// Creates the window manager for the given VRAM region. + pub(super) fn new( + bar: Bar0<'gpu>, + chipset: Chipset, + vram_range: Range<VramAddress>, + ) -> Result<Self> { + let hal = hal::mm_hal(chipset); + let window = io_project!(bar, build: PRAMIN); + let base = vram_range.start.align_down(Self::BASE_ALIGN); + let window_range = Self::window_range(base)?; + hal.write_pramin_window_base(bar, base)?; + + Ok(Self { + bar, + hal, + window, + vram_range, + window_range, + }) + } + + /// Returns the VRAM range a window based at `base` exposes. + fn window_range(base: VramAddress) -> Result<Range<VramAddress>> { + let end = base + .checked_add(WINDOW_SIZE.into_safe_cast()) + .ok_or(EINVAL)?; + Ok(base..end) + } + + /// Check the window covers `len` bytes at `addr`, moving it if needed. + /// + /// Returns the window offset at which to perform the access. + fn window_offset(&mut self, addr: VramAddress, len: usize) -> Result<usize> { + let end = addr.checked_add(len.into_safe_cast()).ok_or(EINVAL)?; + + let inside = |r: &Range<VramAddress>| r.contains(&addr) && end <= r.end; + if !inside(&self.vram_range) { + return Err(EINVAL); + } + + // Reposition the window if the access falls outside it. + if !inside(&self.window_range) { + let base = addr.align_down(Self::BASE_ALIGN); + let window_range = Self::window_range(base)?; + if !inside(&window_range) { + return Err(EINVAL); + } + self.hal.write_pramin_window_base(self.bar, base)?; + self.window_range = window_range; + } + + Ok((addr - self.window_range.start).into_safe_cast()) + } + + /// Return a typed MMIO view of a `T` at `vram_addr`. + /// + /// Returns an error if `vram_addr` is not aligned to `T`'s alignment, or if + /// a `T` at `vram_addr` does not fit within the VRAM region. + pub(super) fn window_at<'a, T>( + &'a mut self, + vram_addr: VramAddress, + ) -> Result<PraminAccess<'a, T>> + where + T: FromBytes + IntoBytes, + { + let offset = self.window_offset(vram_addr, size_of::<T>())?; + let view = io_project!(self.window, try: OffsetLoc::new(offset)); + + Ok(PraminAccess { view }) + } +} + +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +pub(super) mod selftest { + use kernel::{ + device, + io::io_read, + sizes::SizeConstants, // + }; + + use super::*; + use crate::{ + selftest_assert, + selftest_assert_eq, // + }; + + /// Test read/write at byte granularity, at unaligned addresses. + fn test_byte_readwrite( + dev: &device::Device<device::Bound>, + pramin: &mut Pramin<'_>, + base: VramAddress, + ) -> Result { + for i in 0u8..4 { + let addr = base + 1 + u64::from(i); + pramin.window_at::<u8>(addr)?.view().write_val(0xA0 + i); + } + + for i in 0u8..4 { + let addr = base + 1 + u64::from(i); + selftest_assert_eq!( + dev, + pramin.window_at::<u8>(addr)?.view().read_val(), + 0xA0 + i + ); + } + Ok(()) + } + + /// Test writing a `u32` and reading back as individual `u8`s. + fn test_u32_as_bytes( + dev: &device::Device<device::Bound>, + pramin: &mut Pramin<'_>, + base: VramAddress, + ) -> Result { + let addr = base + 0x10; + let val: u32 = 0xDEADBEEF; + pramin.window_at::<u32>(addr)?.view().write_val(val); + + let window = pramin.window_at::<[u8; 4]>(addr)?; + for (i, &expected) in val.to_le_bytes().iter().enumerate() { + selftest_assert_eq!(dev, io_read!(window.view(), [build: i]), expected); + } + Ok(()) + } + + /// Test window repositioning across 1 MiB boundaries. + fn test_window_reposition( + dev: &device::Device<device::Bound>, + pramin: &mut Pramin<'_>, + base: VramAddress, + ) -> Result { + let addr_a = base; + let addr_b = base + u64::SZ_2M; // base + 2 MiB (different 1 MiB region). + let val_a: u32 = 0x11111111; + let val_b: u32 = 0x22222222; + + pramin.window_at::<u32>(addr_a)?.view().write_val(val_a); + pramin.window_at::<u32>(addr_b)?.view().write_val(val_b); + + selftest_assert_eq!( + dev, + pramin.window_at::<u32>(addr_a)?.view().read_val(), + val_a + ); + selftest_assert_eq!( + dev, + pramin.window_at::<u32>(addr_b)?.view().read_val(), + val_b + ); + Ok(()) + } + + /// Test that offsets outside the VRAM region are rejected. + fn test_invalid_offset( + dev: &device::Device<device::Bound>, + pramin: &mut Pramin<'_>, + vram_end: VramAddress, + ) -> Result { + selftest_assert!(dev, pramin.window_at::<u32>(vram_end).is_err()); + Ok(()) + } + + /// Test that misaligned accesses are rejected. + fn test_misaligned_access( + dev: &device::Device<device::Bound>, + pramin: &mut Pramin<'_>, + base: VramAddress, + ) -> Result { + // `u16` at odd offset (not 2-byte aligned). + selftest_assert!(dev, pramin.window_at::<u16>(base + 0x21).is_err()); + + // `u32` at 2-byte-aligned (not 4-byte-aligned) offset. + selftest_assert!(dev, pramin.window_at::<u32>(base + 2).is_err()); + + // `u64` at a 4-byte-aligned (not 8-byte-aligned) address. + selftest_assert!(dev, pramin.window_at::<u64>(base + 0x44).is_err()); + + // A `u16` view at an even address is allowed. + pramin.window_at::<u16>(base + 0x22)?; + Ok(()) + } + + /// Run PRAMIN self-tests during probe. + /// + /// `base` is the start of a driver-usable VRAM span that the tests are free to + /// overwrite. + pub(crate) fn run( + dev: &device::Device<device::Bound>, + pramin: &mut Pramin<'_>, + base: VramAddress, + ) -> Result { + dev_dbg!(dev, "PRAMIN: starting self-tests\n"); + + let vram_end = pramin.vram_range.end; + + test_byte_readwrite(dev, pramin, base)?; + test_u32_as_bytes(dev, pramin, base)?; + test_window_reposition(dev, pramin, base)?; + test_invalid_offset(dev, pramin, vram_end)?; + test_misaligned_access(dev, pramin, base)?; + + dev_info!(dev, "PRAMIN: self-tests passed\n"); + Ok(()) + } +} diff --git a/drivers/gpu/nova-core/mm/regs.rs b/drivers/gpu/nova-core/mm/regs.rs new file mode 100644 index 000000000000..82de6dfa4e8b --- /dev/null +++ b/drivers/gpu/nova-core/mm/regs.rs @@ -0,0 +1,70 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Registers used by the memory management subsystems: the BAR0 PRAMIN window. + +use kernel::io::register; + +use crate::{ + bounded_enum, + driver::NovaRegisters, // +}; + +// PRAMIN window + +bounded_enum! { + /// Target memory type for the BAR0 window register. + /// + /// Only VRAM is needed by the driver. Pre-Hopper window registers also define + /// system-memory targets that are unused here; Hopper+ uses a separate register + /// without a target field. + #[derive(Debug, Copy, Clone)] + pub(super) enum Bar0WindowTarget with TryFrom<Bounded<u32, 2>> { + /// Video memory (GPU framebuffer memory). + VidMem = 0, + } +} + +register! { + base: NovaRegisters; + + /// BAR0 window control for PRAMIN access. + pub(super) NV_PBUS_BAR0_WINDOW(u32) @ 0x00001700 { + /// Target memory aperture for the window. + 25:24 target ?=> Bar0WindowTarget; + /// PRAMIN window base bits 39:16. + 23:0 base; + } +} + +pub(super) mod gh100 { + use kernel::io::register; + + use crate::driver::NovaRegisters; + + register! { + base: NovaRegisters; + + /// Hopper register for PRAMIN window. + pub(crate) NV_XAL_EP_BAR0_WINDOW(u32) @ 0x0010fd40 { + /// PRAMIN window base bits 37:16. + 21:0 base; + } + } +} + +pub(super) mod gb100 { + use kernel::io::register; + + use crate::driver::NovaRegisters; + + register! { + base: NovaRegisters; + + /// Blackwell GB10x/GB20x register for PRAMIN window. + pub(crate) NV_XAL_EP_BAR0_WINDOW(u32) @ 0x0010fd40 { + /// PRAMIN window base bits 38:16. + 22:0 base; + } + } +} diff --git a/drivers/gpu/nova-core/mm/tlb.rs b/drivers/gpu/nova-core/mm/tlb.rs new file mode 100644 index 000000000000..cc862e8159a1 --- /dev/null +++ b/drivers/gpu/nova-core/mm/tlb.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! TLB (Translation Lookaside Buffer) flush support for GPU MMU. +//! +//! After modifying page table entries, the GPU's TLB must be flushed to +//! ensure the new mappings take effect. This module provides TLB flush +//! functionality for virtual memory managers. +//! +//! # Examples +//! +//! ```ignore +//! use crate::mm::tlb::Tlb; +//! +//! fn page_table_update(tlb: &Tlb, pdb_addr: VramAddress) -> Result<()> { +//! // ... modify page tables ... +//! +//! // Flush TLB to make changes visible (polls for completion). +//! tlb.flush(pdb_addr)?; +//! +//! Ok(()) +//! } +//! ``` + +use kernel::{ + io::poll::read_poll_timeout, + io::Io, + new_mutex, + prelude::*, + sync::Mutex, + time::Delta, // +}; + +use crate::{ + bounded_enum, + driver::Bar0, + mm::VramAddress, + regs, // +}; + +bounded_enum! { + /// TLB invalidation acknowledgment scope. + /// + /// Controls how far the hardware waits for the invalidation to propagate + /// before clearing the `trigger` bit of `NV_TLB_FLUSH_CTRL`. + #[derive(Debug, Copy, Clone, PartialEq, Eq)] + pub(crate) enum TlbAckMode with TryFrom<Bounded<u32, 2>> { + /// Fire-and-forget: no acknowledgment required. + None = 0, + /// Wait for acknowledgment from all consumers, including remote GPUs + /// reachable over NVLink. + /// + /// Globally is strictly required only during unmap or permission + /// tightening, because the backing memory may be reassigned after the + /// flush returns and a stale TLB entry could let the GPU access freed + /// memory. For new mapping or relaxing permissions, a stale entry would + /// merely cause a redundant fault and retry, so [`TlbAckMode::None`] + /// would suffice. + Globally = 1, + /// Wait for acknowledgment from consumers within the local NVLink + /// fabric node only; skip cross-node ack. + Intranode = 2, + } +} + +/// TLB manager for GPU translation buffer operations. +#[pin_data] +pub(crate) struct Tlb<'gpu> { + bar: Bar0<'gpu>, + /// TLB flush serialization lock: This lock is designed to be acquired during + /// the DMA fence signalling critical path. It should NEVER be held across any + /// reclaimable CPU memory allocations because the memory reclaim path can + /// call `dma_fence_wait()` (when implemented), which would deadlock if lock held. + #[pin] + lock: Mutex<()>, +} + +impl<'gpu> Tlb<'gpu> { + /// Create a new TLB manager. + pub(super) fn new(bar: Bar0<'gpu>) -> impl PinInit<Self> { + pin_init!(Self { + bar, + lock <- new_mutex!((), "tlb_flush"), + }) + } + + /// Flush the GPU TLB for a specific page directory base. + /// + /// This invalidates all TLB entries associated with the given PDB address. + /// Must be called after modifying page table entries to ensure the GPU sees + /// the updated mappings. + pub(super) fn flush(&self, pdb_addr: VramAddress) -> Result { + let _guard = self.lock.lock(); + + // Write PDB address. + self.bar.write_reg(regs::NV_TLB_FLUSH_PDB_LO::from_pdb_addr( + pdb_addr.into_raw(), + )); + self.bar.write_reg(regs::NV_TLB_FLUSH_PDB_HI::from_pdb_addr( + pdb_addr.into_raw(), + )); + + // Trigger flush. + self.bar.write_reg( + regs::NV_TLB_FLUSH_CTRL::zeroed() + .with_all_va(true) + .with_ack(TlbAckMode::None) + .with_trigger(true), + ); + + // Poll for completion. + read_poll_timeout( + || Ok(self.bar.read(regs::NV_TLB_FLUSH_CTRL)), + |ctrl: ®s::NV_TLB_FLUSH_CTRL| !ctrl.trigger(), + Delta::ZERO, + Delta::from_secs(2), + )?; + + Ok(()) + } +} diff --git a/drivers/gpu/nova-core/mm/vmm.rs b/drivers/gpu/nova-core/mm/vmm.rs new file mode 100644 index 000000000000..51b500a27233 --- /dev/null +++ b/drivers/gpu/nova-core/mm/vmm.rs @@ -0,0 +1,346 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! Virtual Memory Manager for NVIDIA GPU page table management. +//! +//! The [`Vmm`] provides high-level page mapping and unmapping operations for GPU +//! virtual address spaces (Channels, BAR1, BAR2). + +use kernel::{ + gpu::buddy::AllocatedBlocks, + maple_tree::MapleTreeAlloc, + prelude::*, + rbtree::RBTree, // +}; + +use core::{ + cell::Cell, + ops::Range, // +}; + +use crate::{ + mm::{ + pagetable::{ + map::{ + PtMap, // + }, + walk::{ + PtWalk, + WalkResult, // + }, + MmuVersion, // + }, + GpuMm, + Pfn, + Vfn, + VramAddress, + PAGE_SIZE, // + }, + num::{ + IntoSafeCast, // + }, +}; + +/// Multi-page prepared mapping -- VA range allocated, ready for execute. +/// +/// Produced by [`Vmm::prepare_map()`], consumed by [`Vmm::execute_map()`]. +/// The VA space allocation is tracked in the [`Vmm`]'s maple tree and freed +/// on error or via [`Vmm::unmap_pages()`]. +/// +/// Dropping without calling [`Vmm::execute_map()`] logs a warning and leaks +/// the VA range in the maple tree. +pub(crate) struct PreparedMapping { + vfn_start: Vfn, + num_pages: usize, + /// Logs a warning if dropped without executing. + _drop_guard: MustExecuteGuard, +} + +/// Result of a mapping operation -- tracks the active mapped range. +/// +/// Returned by [`Vmm::execute_map()`] and [`Vmm::map_pages()`]. +/// Callers must call [`Vmm::unmap_pages()`] before dropping to invalidate +/// PTEs and free the VA range. Dropping without unmapping logs a warning +/// and leaks the VA range in the maple tree. +pub(crate) struct MappedRange { + pub(super) vfn_start: Vfn, + pub(super) num_pages: usize, + /// Logs a warning if dropped without unmapping. + _drop_guard: MustUnmapGuard, +} + +/// Guard that logs a warning if a [`PreparedMapping`] is dropped without +/// being consumed by [`Vmm::execute_map()`]. +struct MustExecuteGuard { + armed: Cell<bool>, +} + +impl MustExecuteGuard { + const fn new() -> Self { + Self { + armed: Cell::new(true), + } + } + + fn disarm(&self) { + self.armed.set(false); + } +} + +impl Drop for MustExecuteGuard { + fn drop(&mut self) { + if self.armed.get() { + kernel::pr_warn!("PreparedMapping dropped without calling execute_map()\n"); + } + } +} + +/// Guard that logs a warning if a [`MappedRange`] is dropped without +/// calling [`Vmm::unmap_pages()`]. +struct MustUnmapGuard { + armed: Cell<bool>, +} + +impl MustUnmapGuard { + const fn new() -> Self { + Self { + armed: Cell::new(true), + } + } + + fn disarm(&self) { + self.armed.set(false); + } +} + +impl Drop for MustUnmapGuard { + fn drop(&mut self) { + if self.armed.get() { + kernel::pr_warn!("MappedRange dropped without calling unmap_pages()\n"); + } + } +} + +/// Virtual Memory Manager for a GPU address space. +/// +/// Each [`Vmm`] instance manages a single address space identified by its Page +/// Directory Base (`PDB`) address. Used for Channel, BAR1 and BAR2 mappings. +pub(crate) struct Vmm { + /// Page Directory Base address for this address space. + #[expect(dead_code)] + pdb_addr: VramAddress, + /// Page table walker for reading existing mappings. + pt_walk: PtWalk, + /// Page table mapper for prepare/execute operations. + pt_map: PtMap, + /// Page table allocations required for mappings. + page_table_allocs: KVec<Pin<KBox<AllocatedBlocks>>>, + /// Maple tree allocator for virtual address range tracking. + virt_alloc: Pin<KBox<MapleTreeAlloc<()>>>, + /// Total number of pages in the virtual address space. + va_pages: usize, + /// Prepared PT pages pending PDE installation, keyed by `install_addr`. + /// + /// Populated during prepare phase and drained in execute phase. Shared by all + /// pending maps, preventing races on the same PDE slot. + pt_pages: RBTree<VramAddress, super::pagetable::map::PreparedPtPage>, +} + +impl Vmm { + /// Create a new [`Vmm`] for the given Page Directory Base address. + /// + /// The [`Vmm`] will manage a virtual address space of `va_size` bytes. + pub(crate) fn new( + pdb_addr: VramAddress, + mmu_version: MmuVersion, + va_size: u64, + ) -> Result<Self> { + let page_size: u64 = PAGE_SIZE.into_safe_cast(); + let va_pages: usize = (va_size / page_size).into_safe_cast(); + let virt_alloc = KBox::pin_init(MapleTreeAlloc::<()>::new(), GFP_KERNEL)?; + + Ok(Self { + pdb_addr, + pt_walk: PtWalk::new(pdb_addr, mmu_version), + pt_map: PtMap::new(pdb_addr, mmu_version), + page_table_allocs: KVec::new(), + virt_alloc, + va_pages, + pt_pages: RBTree::new(), + }) + } + + /// Allocate a contiguous virtual frame number range. + fn alloc_vfn_range(&self, num_pages: usize, va_range: Option<Range<u64>>) -> Result<Vfn> { + let page_size: u64 = PAGE_SIZE.into_safe_cast(); + + let start_vfn = match va_range { + Some(r) => { + let num_pages_u64: u64 = num_pages.into_safe_cast(); + let size = num_pages_u64.checked_mul(page_size).ok_or(EOVERFLOW)?; + let range_size = r.end.checked_sub(r.start).ok_or(EOVERFLOW)?; + if range_size != size { + return Err(EINVAL); + } + let start_vfn: usize = (r.start / page_size).into_safe_cast(); + let end_vfn: usize = (r.end / page_size).into_safe_cast(); + self.virt_alloc + .insert_range(start_vfn..end_vfn, (), GFP_KERNEL)?; + start_vfn + } + None => self + .virt_alloc + .alloc_range(num_pages, (), ..self.va_pages, GFP_KERNEL)?, + }; + + Ok(Vfn::new(start_vfn.into_safe_cast())) + } + + /// Free a virtual frame number range back to the maple tree. + fn free_vfn(&self, vfn: Vfn) { + let vfn_index: usize = vfn.raw().into_safe_cast(); + if self.virt_alloc.erase(vfn_index).is_none() { + kernel::pr_warn!("free_vfn: VFN {} not found in maple tree\n", vfn_index); + } + } + + /// Read the [`Pfn`] for a mapped [`Vfn`] if one is mapped. + pub(super) fn read_mapping(&self, mm: &mut GpuMm<'_>, vfn: Vfn) -> Result<Option<Pfn>> { + match self.pt_walk.walk_to_pte(mm, vfn)? { + WalkResult::Mapped { pfn, .. } => Ok(Some(pfn)), + WalkResult::Unmapped { .. } | WalkResult::PageTableMissing => Ok(None), + } + } + + /// Prepare resources for mapping `num_pages` pages. + /// + /// Allocates a contiguous VA range, then walks the hierarchy per-VFN to prepare pages + /// for all missing PDEs. Returns a [`PreparedMapping`] with the VA allocation. + /// + /// If `va_range` is not `None`, the VA range is constrained to the given range. Safe + /// to call outside the fence signalling critical path. + pub(crate) fn prepare_map( + &mut self, + mm: &mut GpuMm<'_>, + num_pages: usize, + va_range: Option<Range<u64>>, + ) -> Result<PreparedMapping> { + if num_pages == 0 { + return Err(EINVAL); + } + + // Allocate contiguous VA range. + let vfn_start = self.alloc_vfn_range(num_pages, va_range)?; + + if let Err(e) = self.pt_map.prepare_map( + mm, + vfn_start, + num_pages, + &mut self.page_table_allocs, + &mut self.pt_pages, + ) { + self.free_vfn(vfn_start); + return Err(e); + } + + Ok(PreparedMapping { + vfn_start, + num_pages, + _drop_guard: MustExecuteGuard::new(), + }) + } + + /// Execute a prepared multi-page mapping. + /// + /// Installs all prepared PDEs and writes PTEs into the page table, then flushes TLB. + pub(crate) fn execute_map( + &mut self, + mm: &mut GpuMm<'_>, + prepared: PreparedMapping, + pfns: &[Pfn], + writable: bool, + ) -> Result<MappedRange> { + if pfns.len() != prepared.num_pages { + self.free_vfn(prepared.vfn_start); + return Err(EINVAL); + } + + let PreparedMapping { + vfn_start, + num_pages, + _drop_guard, + } = prepared; + _drop_guard.disarm(); + + if let Err(e) = self.pt_map.install_mappings( + mm, + &mut self.pt_pages, + &mut self.page_table_allocs, + vfn_start, + pfns, + writable, + ) { + self.free_vfn(vfn_start); + return Err(e); + } + + Ok(MappedRange { + vfn_start, + num_pages, + _drop_guard: MustUnmapGuard::new(), + }) + } + + /// Map pages doing prepare and execute in the same call. + /// + /// This is a convenience wrapper for callers outside the fence signalling critical + /// path (e.g., BAR mappings). For DRM usecases, [`Vmm::prepare_map()`] and + /// [`Vmm::execute_map()`] will be called separately. + pub(crate) fn map_pages( + &mut self, + mm: &mut GpuMm<'_>, + pfns: &[Pfn], + va_range: Option<Range<u64>>, + writable: bool, + ) -> Result<MappedRange> { + if pfns.is_empty() { + return Err(EINVAL); + } + + // Check if provided VA range is sufficient (if provided). + if let Some(ref range) = va_range { + let required: u64 = pfns + .len() + .checked_mul(PAGE_SIZE) + .ok_or(EOVERFLOW)? + .into_safe_cast(); + let available = range.end.checked_sub(range.start).ok_or(EINVAL)?; + if available < required { + return Err(EINVAL); + } + } + + let prepared = self.prepare_map(mm, pfns.len(), va_range)?; + self.execute_map(mm, prepared, pfns, writable) + } + + /// Unmap all pages in a [`MappedRange`] with a single TLB flush. + pub(crate) fn unmap_pages(&mut self, mm: &mut GpuMm<'_>, range: MappedRange) -> Result { + let result = self + .pt_map + .invalidate_ptes(mm, range.vfn_start, range.num_pages); + + // TODO: Internal page table pages (PDE, PTE pages) are still kept around. + // This is by design as repeated maps/unmaps will be fast. As a future TODO, + // we can add a reclaimer here to reclaim if VRAM is short. For now, the PT + // pages are dropped once the `Vmm` is dropped. + + // Free the VA range regardless of PTE invalidation success, so that the VA + // range is recovered even on failure (PTEs may be stale, but that is better + // than leaking both PTEs and VA range). + self.free_vfn(range.vfn_start); + + // Unmap complete, safe to drop `MappedRange`. + range._drop_guard.disarm(); + result + } +} diff --git a/drivers/gpu/nova-core/nova_core.rs b/drivers/gpu/nova-core/nova_core.rs index 35a8b1214b0e..1133c6ce5c55 100644 --- a/drivers/gpu/nova-core/nova_core.rs +++ b/drivers/gpu/nova-core/nova_core.rs @@ -18,10 +18,13 @@ mod fsp; mod gpu; mod gsp; mod mctp; +mod mm; #[macro_use] mod num; mod regs; mod sbuffer; +#[cfg(CONFIG_NOVA_CORE_SELFTESTS)] +mod selftest; mod vbios; mod vgpu; diff --git a/drivers/gpu/nova-core/regs.rs b/drivers/gpu/nova-core/regs.rs index caeef4d85874..9978fb2803b0 100644 --- a/drivers/gpu/nova-core/regs.rs +++ b/drivers/gpu/nova-core/regs.rs @@ -4,110 +4,37 @@ use kernel::{ io::{ register, - register::WithBase, - Io, // + Io, + Mmio, // }, - prelude::*, sizes::SizeConstants, time, // }; +use pin_init::Zeroable; use crate::{ - driver::Bar0, + driver::NovaRegisters, falcon::{ DmaTrfCmdSize, FalconCoreRev, FalconCoreRevSubversion, - FalconEngine, FalconFbifMemType, FalconFbifTarget, FalconMem, FalconModSelAlgo, FalconSecurityModel, - PFalcon2Base, - PFalconBase, + PFalcon2Registers, + PFalconRegisters, PeregrineCoreSelect, // }, - gpu::{ - Architecture, - Chipset, // - }, + mm::tlb::TlbAckMode, // }; -// PMC - -register! { - /// Basic revision information about the GPU. - pub(crate) NV_PMC_BOOT_0(u32) @ 0x00000000 { - /// Lower bits of the architecture. - 28:24 architecture_0; - /// Implementation version of the architecture. - 23:20 implementation; - /// MSB of the architecture. - 8:8 architecture_1; - /// Major revision of the chip. - 7:4 major_revision; - /// Minor revision of the chip. - 3:0 minor_revision; - } - - /// Extended architecture information. - pub(crate) NV_PMC_BOOT_42(u32) @ 0x00000a00 { - /// Architecture value. - 29:24 architecture ?=> Architecture; - /// Implementation version of the architecture. - 23:20 implementation; - /// Major revision of the chip. - 19:16 major_revision; - /// Minor revision of the chip. - 15:12 minor_revision; - } -} - -impl NV_PMC_BOOT_0 { - pub(crate) fn is_older_than_fermi(self) -> bool { - // From https://github.com/NVIDIA/open-gpu-doc/tree/master/manuals : - const NV_PMC_BOOT_0_ARCHITECTURE_GF100: u32 = 0xc; - - // Older chips left arch1 zeroed out. That, combined with an arch0 value that is less than - // GF100, means "older than Fermi". - self.architecture_1() == 0 && self.architecture_0() < NV_PMC_BOOT_0_ARCHITECTURE_GF100 - } -} - -impl NV_PMC_BOOT_42 { - /// Combines `architecture` and `implementation` to obtain a code unique to the chipset. - pub(crate) fn chipset(self) -> Result<Chipset> { - self.architecture() - .map(|arch| { - ((arch as u32) << Self::IMPLEMENTATION_RANGE.len()) - | u32::from(self.implementation()) - }) - .and_then(Chipset::try_from) - } - - /// Returns the raw architecture value from the register. - fn architecture_raw(self) -> u8 { - ((self.into_raw() >> Self::ARCHITECTURE_RANGE.start()) - & ((1 << Self::ARCHITECTURE_RANGE.len()) - 1)) as u8 - } -} - -impl kernel::fmt::Display for NV_PMC_BOOT_42 { - fn fmt(&self, f: &mut kernel::fmt::Formatter<'_>) -> kernel::fmt::Result { - write!( - f, - "boot42 = 0x{:08x} (architecture 0x{:x}, implementation 0x{:x})", - self.inner, - self.architecture_raw(), - self.implementation() - ) - } -} - // PBUS register! { + base: NovaRegisters; + pub(crate) NV_PBUS_SW_SCRATCH(u32)[64] @ 0x00001400 {} } @@ -121,6 +48,8 @@ register! { // number. register! { + base: NovaRegisters; + /// Boot Sequence Interface (BSI) register used to determine /// if GSP reload/resume has completed during the boot process. pub(crate) NV_PGC6_BSI_SECURE_SCRATCH_14(u32) @ 0x001180f8 { @@ -175,6 +104,8 @@ impl NV_USABLE_FB_SIZE_IN_MB { pub(crate) const NV_FUSE_OPT_FPF_SIZE: usize = 16; register! { + base: NovaRegisters; + pub(crate) NV_FUSE_OPT_FPF_NVDEC_UCODE1_VERSION(u32)[NV_FUSE_OPT_FPF_SIZE] @ 0x00824100 { 15:0 data => u16; } @@ -191,30 +122,32 @@ register! { // PFALCON register! { - pub(crate) NV_PFALCON_FALCON_IRQSCLR(u32) @ PFalconBase + 0x00000004 { + base: PFalconRegisters; + + pub(crate) NV_PFALCON_FALCON_IRQSCLR(u32) @ 0x00000004 { 6:6 swgen0 => bool; 4:4 halt => bool; } - pub(crate) NV_PFALCON_FALCON_MAILBOX0(u32) @ PFalconBase + 0x00000040 { + pub(crate) NV_PFALCON_FALCON_MAILBOX0(u32) @ 0x00000040 { 31:0 value => u32; } - pub(crate) NV_PFALCON_FALCON_MAILBOX1(u32) @ PFalconBase + 0x00000044 { + pub(crate) NV_PFALCON_FALCON_MAILBOX1(u32) @ 0x00000044 { 31:0 value => u32; } /// Used to store version information about the firmware running /// on the Falcon processor. - pub(crate) NV_PFALCON_FALCON_OS(u32) @ PFalconBase + 0x00000080 { + pub(crate) NV_PFALCON_FALCON_OS(u32) @ 0x00000080 { 31:0 value => u32; } - pub(crate) NV_PFALCON_FALCON_RM(u32) @ PFalconBase + 0x00000084 { + pub(crate) NV_PFALCON_FALCON_RM(u32) @ 0x00000084 { 31:0 value => u32; } - pub(crate) NV_PFALCON_FALCON_HWCFG2(u32) @ PFalconBase + 0x000000f4 { + pub(crate) NV_PFALCON_FALCON_HWCFG2(u32) @ 0x000000f4 { /// Signal indicating that reset is completed (GA102+). 31:31 reset_ready => bool; /// RISC-V branch privilege lockdown bit. @@ -224,17 +157,17 @@ register! { 10:10 riscv => bool; } - pub(crate) NV_PFALCON_FALCON_CPUCTL(u32) @ PFalconBase + 0x00000100 { + pub(crate) NV_PFALCON_FALCON_CPUCTL(u32) @ 0x00000100 { 6:6 alias_en => bool; 4:4 halted => bool; 1:1 startcpu => bool; } - pub(crate) NV_PFALCON_FALCON_BOOTVEC(u32) @ PFalconBase + 0x00000104 { + pub(crate) NV_PFALCON_FALCON_BOOTVEC(u32) @ 0x00000104 { 31:0 value => u32; } - pub(crate) NV_PFALCON_FALCON_DMACTL(u32) @ PFalconBase + 0x0000010c { + pub(crate) NV_PFALCON_FALCON_DMACTL(u32) @ 0x0000010c { 7:7 secure_stat => bool; 6:3 dmaq_num; 2:2 imem_scrubbing => bool; @@ -242,15 +175,15 @@ register! { 0:0 require_ctx => bool; } - pub(crate) NV_PFALCON_FALCON_DMATRFBASE(u32) @ PFalconBase + 0x00000110 { + pub(crate) NV_PFALCON_FALCON_DMATRFBASE(u32) @ 0x00000110 { 31:0 base => u32; } - pub(crate) NV_PFALCON_FALCON_DMATRFMOFFS(u32) @ PFalconBase + 0x00000114 { + pub(crate) NV_PFALCON_FALCON_DMATRFMOFFS(u32) @ 0x00000114 { 23:0 offs; } - pub(crate) NV_PFALCON_FALCON_DMATRFCMD(u32) @ PFalconBase + 0x00000118 { + pub(crate) NV_PFALCON_FALCON_DMATRFCMD(u32) @ 0x00000118 { 16:16 set_dmtag; 14:12 ctxdma; 10:8 size ?=> DmaTrfCmdSize; @@ -261,15 +194,15 @@ register! { 0:0 full => bool; } - pub(crate) NV_PFALCON_FALCON_DMATRFFBOFFS(u32) @ PFalconBase + 0x0000011c { + pub(crate) NV_PFALCON_FALCON_DMATRFFBOFFS(u32) @ 0x0000011c { 31:0 offs => u32; } - pub(crate) NV_PFALCON_FALCON_DMATRFBASE1(u32) @ PFalconBase + 0x00000128 { + pub(crate) NV_PFALCON_FALCON_DMATRFBASE1(u32) @ 0x00000128 { 8:0 base; } - pub(crate) NV_PFALCON_FALCON_HWCFG1(u32) @ PFalconBase + 0x0000012c { + pub(crate) NV_PFALCON_FALCON_HWCFG1(u32) @ 0x0000012c { /// Core revision subversion. 7:6 core_rev_subversion => FalconCoreRevSubversion; /// Security model. @@ -278,12 +211,12 @@ register! { 3:0 core_rev ?=> FalconCoreRev; } - pub(crate) NV_PFALCON_FALCON_CPUCTL_ALIAS(u32) @ PFalconBase + 0x00000130 { + pub(crate) NV_PFALCON_FALCON_CPUCTL_ALIAS(u32) @ 0x00000130 { 1:1 startcpu => bool; } /// IMEM access control register. Up to 4 ports are available for IMEM access. - pub(crate) NV_PFALCON_FALCON_IMEMC(u32)[4, stride = 16] @ PFalconBase + 0x00000180 { + pub(crate) NV_PFALCON_FALCON_IMEMC(u32)[4, stride = 16] @ 0x00000180 { /// Access secure IMEM. 28:28 secure => bool; /// Auto-increment on write. @@ -294,17 +227,17 @@ register! { /// IMEM data register. Reading/writing this register accesses IMEM at the address /// specified by the corresponding IMEMC register. - pub(crate) NV_PFALCON_FALCON_IMEMD(u32)[4, stride = 16] @ PFalconBase + 0x00000184 { + pub(crate) NV_PFALCON_FALCON_IMEMD(u32)[4, stride = 16] @ 0x00000184 { 31:0 data; } /// IMEM tag register. Used to set the tag for the current IMEM block. - pub(crate) NV_PFALCON_FALCON_IMEMT(u32)[4, stride = 16] @ PFalconBase + 0x00000188 { + pub(crate) NV_PFALCON_FALCON_IMEMT(u32)[4, stride = 16] @ 0x00000188 { 15:0 tag; } /// DMEM access control register. Up to 8 ports are available for DMEM access. - pub(crate) NV_PFALCON_FALCON_DMEMC(u32)[8, stride = 8] @ PFalconBase + 0x000001c0 { + pub(crate) NV_PFALCON_FALCON_DMEMC(u32)[8, stride = 8] @ 0x000001c0 { /// Auto-increment on write. 24:24 aincw => bool; /// DMEM block and word offset. @@ -313,29 +246,29 @@ register! { /// DMEM data register. Reading/writing this register accesses DMEM at the address /// specified by the corresponding DMEMC register. - pub(crate) NV_PFALCON_FALCON_DMEMD(u32)[8, stride = 8] @ PFalconBase + 0x000001c4 { + pub(crate) NV_PFALCON_FALCON_DMEMD(u32)[8, stride = 8] @ 0x000001c4 { 31:0 data; } /// Actually known as `NV_PSEC_FALCON_ENGINE` and `NV_PGSP_FALCON_ENGINE` depending on the /// falcon instance. - pub(crate) NV_PFALCON_FALCON_ENGINE(u32) @ PFalconBase + 0x000003c0 { + pub(crate) NV_PFALCON_FALCON_ENGINE(u32) @ 0x000003c0 { 0:0 reset => bool; } - pub(crate) NV_PFALCON_FBIF_TRANSCFG(u32)[8] @ PFalconBase + 0x00000600 { + pub(crate) NV_PFALCON_FBIF_TRANSCFG(u32)[8] @ 0x00000600 { 2:2 mem_type => FalconFbifMemType; 1:0 target ?=> FalconFbifTarget; } - pub(crate) NV_PFALCON_FBIF_CTL(u32) @ PFalconBase + 0x00000624 { + pub(crate) NV_PFALCON_FBIF_CTL(u32) @ 0x00000624 { 7:7 allow_phys_no_ctx => bool; } // Falcon EMEM PIO registers (used by FSP on Hopper/Blackwell). // These provide the falcon external memory communication interface. - pub(crate) NV_PFALCON_FALCON_EMEMC(u32) @ PFalconBase + 0x00000ac0 { + pub(crate) NV_PFALCON_FALCON_EMEMC(u32) @ 0x00000ac0 { /// EMEM byte offset (4-byte aligned) within the block. 7:2 offs; /// EMEM block to access. @@ -346,7 +279,7 @@ register! { 25:25 aincr => bool; } - pub(crate) NV_PFALCON_FALCON_EMEMD(u32) @ PFalconBase + 0x00000ac4 { + pub(crate) NV_PFALCON_FALCON_EMEMD(u32) @ 0x00000ac4 { 31:0 data => u32; } } @@ -372,13 +305,13 @@ impl NV_PFALCON_FALCON_DMATRFCMD { impl NV_PFALCON_FALCON_ENGINE { /// Resets the falcon - pub(crate) fn reset_engine<E: FalconEngine>(bar: Bar0<'_>) { - bar.update(Self::of::<E>(), |r| r.with_reset(true)); + pub(crate) fn reset_engine(pfalcon: Mmio<'_, PFalconRegisters>) { + pfalcon.update(NV_PFALCON_FALCON_ENGINE, |r| r.with_reset(true)); // TIMEOUT: falcon engine should not take more than 10us to reset. time::delay::fsleep(time::Delta::from_micros(10)); - bar.update(Self::of::<E>(), |r| r.with_reset(false)); + pfalcon.update(NV_PFALCON_FALCON_ENGINE, |r| r.with_reset(false)); } } @@ -392,21 +325,23 @@ impl NV_PFALCON_FALCON_HWCFG2 { /* PFALCON2 */ register! { - pub(crate) NV_PFALCON2_FALCON_MOD_SEL(u32) @ PFalcon2Base + 0x00000180 { + base: PFalcon2Registers; + + pub(crate) NV_PFALCON2_FALCON_MOD_SEL(u32) @ 0x00000180 { 7:0 algo ?=> FalconModSelAlgo; } - pub(crate) NV_PFALCON2_FALCON_BROM_CURR_UCODE_ID(u32) @ PFalcon2Base + 0x00000198 { + pub(crate) NV_PFALCON2_FALCON_BROM_CURR_UCODE_ID(u32) @ 0x00000198 { 7:0 ucode_id => u8; } - pub(crate) NV_PFALCON2_FALCON_BROM_ENGIDMASK(u32) @ PFalcon2Base + 0x0000019c { + pub(crate) NV_PFALCON2_FALCON_BROM_ENGIDMASK(u32) @ 0x0000019c { 31:0 value => u32; } /// OpenRM defines this as a register array, but doesn't specify its size and only uses its /// first element. Be conservative until we know the actual size or need to use more registers. - pub(crate) NV_PFALCON2_FALCON_BROM_PARAADDR(u32)[1] @ PFalcon2Base + 0x00000210 { + pub(crate) NV_PFALCON2_FALCON_BROM_PARAADDR(u32)[1] @ 0x00000210 { 31:0 value => u32; } } @@ -414,21 +349,23 @@ register! { // PRISCV register! { + base: PFalcon2Registers; + /// RISC-V status register for debug (Turing and GA100 only). /// Reflects current RISC-V core status. - pub(crate) NV_PRISCV_RISCV_CORE_SWITCH_RISCV_STATUS(u32) @ PFalcon2Base + 0x00000240 { + pub(crate) NV_PRISCV_RISCV_CORE_SWITCH_RISCV_STATUS(u32) @ 0x00000240 { /// RISC-V core active/inactive status. 0:0 active_stat => bool; } /// GA102 and later. - pub(crate) NV_PRISCV_RISCV_CPUCTL(u32) @ PFalcon2Base + 0x00000388 { + pub(crate) NV_PRISCV_RISCV_CPUCTL(u32) @ 0x00000388 { 7:7 active_stat => bool; 4:4 halted => bool; } /// GA102 and later. - pub(crate) NV_PRISCV_RISCV_BCR_CTRL(u32) @ PFalcon2Base + 0x00000668 { + pub(crate) NV_PRISCV_RISCV_BCR_CTRL(u32) @ 0x00000668 { 8:8 br_fetch => bool; 4:4 core_select => PeregrineCoreSelect; 0:0 valid => bool; @@ -439,6 +376,8 @@ register! { // These registers manage falcon EMEM communication queues. register! { + base: NovaRegisters; + pub(crate) NV_PFSP_QUEUE_HEAD(u32)[8] @ 0x008f2c00 { 31:0 address => u32; } @@ -462,9 +401,13 @@ register! { pub(crate) mod gm107 { use kernel::io::register; + use crate::driver::NovaRegisters; + // FUSE register! { + base: NovaRegisters; + pub(crate) NV_FUSE_STATUS_OPT_DISPLAY(u32) @ 0x00021c04 { 0:0 display_disabled => bool; } @@ -474,9 +417,13 @@ pub(crate) mod gm107 { pub(crate) mod ga100 { use kernel::io::register; + use crate::driver::NovaRegisters; + // FUSE register! { + base: NovaRegisters; + pub(crate) NV_FUSE_STATUS_OPT_DISPLAY(u32) @ 0x00820c04 { 0:0 display_disabled => bool; } @@ -488,9 +435,13 @@ pub(crate) const NV_THERM_I2CS_SCRATCH_FSP_BOOT_COMPLETE_STATUS_SUCCESS: u32 = 0 pub(crate) mod gh100 { use kernel::io::register; + use crate::driver::NovaRegisters; + // PTHERM register! { + base: NovaRegisters; + pub(crate) NV_THERM_I2CS_SCRATCH(u32) @ 0x000200bc { 31:0 data; } @@ -505,9 +456,13 @@ pub(crate) mod gh100 { pub(crate) mod gb202 { use kernel::io::register; + use crate::driver::NovaRegisters; + // PTHERM register! { + base: NovaRegisters; + pub(crate) NV_THERM_I2CS_SCRATCH(u32) @ 0x00ad00bc { 31:0 data; } @@ -518,3 +473,69 @@ pub(crate) mod gb202 { } } } + +// MMU TLB + +register! { + base: NovaRegisters; + + /// TLB flush register: PDB address lower bits. + pub(crate) NV_TLB_FLUSH_PDB_LO(u32) @ 0x00b830a0 { + /// PDB address bits [39:8]. + 31:0 pdb_lo => u32; + } + + /// TLB flush register: PDB address higher bits. + pub(crate) NV_TLB_FLUSH_PDB_HI(u32) @ 0x00b830a4 { + /// PDB address bits [47:40]. + 7:0 pdb_hi => u8; + } + + /// TLB flush control register. + pub(crate) NV_TLB_FLUSH_CTRL(u32) @ 0x00b830b0 { + /// Invalidate every VA in the PDB selected by `NV_TLB_FLUSH_PDB_LO/HI`. + 0:0 all_va => bool; + /// Invalidate TLBs for all PDBs (ignores `NV_TLB_FLUSH_PDB_LO/HI`). + 1:1 all_pdb => bool; + /// Restrict the flush to the HUB MMU's TLBs; skip broadcasting to the + /// per-GPC L2 TLBs. + /// + /// The GPU MMU has a two-level TLB hierarchy: + /// 1. The *HUB MMU* sits at the top and serves memory requests from + /// "host-side" engines: the host/channel interface, copy engines, + /// display, and BAR1/BAR2 accesses. + /// 2. Each GPC (Graphics Processing Cluster — the block that houses + /// shader cores / SMs) has its own L2 TLB that serves requests from + /// the compute and graphics engines inside the cluster. + /// + /// When set, only the HUB TLBs are invalidated. This is a performance + /// optimization for flushes that only affect HUB-side mappings (e.g. + /// BAR1/BAR2 windows), where fanning the invalidation out to every + /// GPC's L2 TLB would be wasted work. Must be false when flushing + /// mappings that may be cached by compute/graphics engines. + 2:2 hubtlb_only => bool; + /// Invalidation acknowledgment scope. See [`TlbAckMode`] for details. + 8:7 ack ?=> TlbAckMode; + /// Write 1 to kick off the flush. Hardware clears this bit when the + /// flush completes; reads as 1 while the flush is in progress. + 31:31 trigger => bool; + } +} + +impl NV_TLB_FLUSH_PDB_LO { + /// Create a register value from a PDB address. + /// + /// Extracts bits [39:8] of the address and shifts it right by 8 bits. + pub(crate) fn from_pdb_addr(addr: u64) -> Self { + Self::zeroed().with_pdb_lo(((addr >> 8) & 0xFFFF_FFFF) as u32) + } +} + +impl NV_TLB_FLUSH_PDB_HI { + /// Create a register value from a PDB address. + /// + /// Extracts bits [47:40] of the address and shifts it right by 40 bits. + pub(crate) fn from_pdb_addr(addr: u64) -> Self { + Self::zeroed().with_pdb_hi(((addr >> 40) & 0xFF) as u8) + } +} diff --git a/drivers/gpu/nova-core/selftest.rs b/drivers/gpu/nova-core/selftest.rs new file mode 100644 index 000000000000..f5b5965b7e6a --- /dev/null +++ b/drivers/gpu/nova-core/selftest.rs @@ -0,0 +1,64 @@ +// SPDX-License-Identifier: GPL-2.0 +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + +//! Assertion macros for driver self-tests. +//! +//! Self-tests run against live hardware during probe, so a failed assertion should not panic. These +//! macros log the failure on the device and fail the enclosing test by returning +//! [`EIO`](kernel::error::code::EIO) instead. + +/// Like [`assert!`], but logs the failure via `dev` and fails the enclosing test instead of +/// panicking. +/// +/// As with [`assert!`], a custom message with format arguments can follow the condition. +#[macro_export] +macro_rules! selftest_assert { + ($dev:expr, $cond:expr $(,)?) => { + $crate::selftest_assert!($dev, $cond, "assertion failed: {}", ::core::stringify!($cond)) + }; + ($dev:expr, $cond:expr, $($arg:tt)+) => {{ + if !$cond { + ::kernel::dev_err!( + $dev, + "Selftest: {}:{}: {}\n", + ::core::file!(), + ::core::line!(), + ::kernel::prelude::fmt!($($arg)+) + ); + return Err(::kernel::error::code::EIO); + } + }}; +} + +/// Like [`assert_eq!`], but logs the failure via `dev` and fails the enclosing test instead of +/// panicking. +/// +/// As with [`assert_eq!`], a custom message with format arguments can follow the compared values. +#[macro_export] +macro_rules! selftest_assert_eq { + ($dev:expr, $left:expr, $right:expr $(,)?) => { + match (&$left, &$right) { + (left, right) => $crate::selftest_assert!( + $dev, + left == right, + "assertion `{} == {}` failed: left {:?}, right {:?}", + ::core::stringify!($left), + ::core::stringify!($right), + left, + right + ), + } + }; + ($dev:expr, $left:expr, $right:expr, $($arg:tt)+) => { + match (&$left, &$right) { + (left, right) => $crate::selftest_assert!( + $dev, + left == right, + "assertion `left == right` failed: {}: left {:?}, right {:?}", + ::kernel::prelude::fmt!($($arg)+), + left, + right + ), + } + }; +} diff --git a/drivers/gpu/nova-core/vbios.rs b/drivers/gpu/nova-core/vbios.rs index c03650ee5226..9c214b9f4dd9 100644 --- a/drivers/gpu/nova-core/vbios.rs +++ b/drivers/gpu/nova-core/vbios.rs @@ -16,7 +16,10 @@ use kernel::{ }; use crate::{ - driver::Bar0, + driver::{ + Bar0, + NovaRegisters, // + }, firmware::{ fwsec::Bcrt30Rsa3kSignature, FalconUCodeDesc, @@ -92,12 +95,16 @@ impl<'a> VbiosIterator<'a> { fn rom_offset(dev: &device::Device, bar0: Bar0<'_>) -> Result<usize> { // IFR Header in VBIOS. register! { + base: NovaRegisters; + NV_PBUS_IFR_FMT_FIXED0(u32) @ 0x300000 { 31:0 signature; } } register! { + base: NovaRegisters; + NV_PBUS_IFR_FMT_FIXED1(u32) @ 0x300004 { 30:16 fixed_data_size; 15:8 version => u8; @@ -105,6 +112,8 @@ impl<'a> VbiosIterator<'a> { } register! { + base: NovaRegisters; + NV_PBUS_IFR_FMT_FIXED2(u32) @ 0x300008 { 19:0 total_data_size; } diff --git a/rust/bindings/bindings_helper.h b/rust/bindings/bindings_helper.h index 1075b26e53ac..c4b03c33d1cb 100644 --- a/rust/bindings/bindings_helper.h +++ b/rust/bindings/bindings_helper.h @@ -53,6 +53,7 @@ #include <linux/debugfs.h> #include <linux/device/faux.h> #include <linux/dma-direction.h> +#include <linux/dma-fence.h> #include <linux/dma-mapping.h> #include <linux/dma-resv.h> #include <linux/errname.h> diff --git a/rust/helpers/dma_fence.c b/rust/helpers/dma_fence.c new file mode 100644 index 000000000000..549f6b6a7171 --- /dev/null +++ b/rust/helpers/dma_fence.c @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <linux/dma-fence.h> + +__rust_helper void rust_helper_dma_fence_get(struct dma_fence *f) +{ + dma_fence_get(f); +} + +__rust_helper void rust_helper_dma_fence_put(struct dma_fence *f) +{ + dma_fence_put(f); +} + +__rust_helper bool rust_helper_dma_fence_begin_signalling(void) +{ + return dma_fence_begin_signalling(); +} + +__rust_helper void rust_helper_dma_fence_end_signalling(bool cookie) +{ + dma_fence_end_signalling(cookie); +} + +__rust_helper bool rust_helper_dma_fence_is_signaled(struct dma_fence *f) +{ + return dma_fence_is_signaled(f); +} + +__rust_helper bool rust_helper_dma_fence_test_signaled_flag(struct dma_fence *f) +{ + return dma_fence_test_signaled_flag(f); +} + +__rust_helper void rust_helper_dma_fence_lock_irqsave(struct dma_fence *f, unsigned long *flags) +{ + dma_fence_lock_irqsave(f, *flags); +} + +__rust_helper void rust_helper_dma_fence_unlock_irqrestore(struct dma_fence *f, + unsigned long *flags) +{ + dma_fence_unlock_irqrestore(f, *flags); +} + +__rust_helper void rust_helper_dma_fence_set_error(struct dma_fence *f, int error) +{ + dma_fence_set_error(f, error); +} diff --git a/rust/helpers/helpers.c b/rust/helpers/helpers.c index 440fb7638e3c..2c5eecc091c4 100644 --- a/rust/helpers/helpers.c +++ b/rust/helpers/helpers.c @@ -58,6 +58,7 @@ #include "device.c" #include "dma.c" #ifdef CONFIG_DMA_SHARED_BUFFER +#include "dma_fence.c" #include "dma-resv.c" #endif #include "drm.c" diff --git a/rust/helpers/pci.c b/rust/helpers/pci.c index a714cc2bfb7a..3686e405160d 100644 --- a/rust/helpers/pci.c +++ b/rust/helpers/pci.c @@ -19,6 +19,12 @@ __rust_helper resource_size_t rust_helper_pci_resource_len(struct pci_dev *pdev, return pci_resource_len(pdev, bar); } +__rust_helper unsigned long rust_helper_pci_resource_flags(const struct pci_dev *pdev, + int bar) +{ + return pci_resource_flags(pdev, bar); +} + __rust_helper bool rust_helper_dev_is_pci(const struct device *dev) { return dev_is_pci(dev); diff --git a/rust/kernel/bitfield.rs b/rust/kernel/bitfield.rs index a0d089423f21..15c78790e151 100644 --- a/rust/kernel/bitfield.rs +++ b/rust/kernel/bitfield.rs @@ -346,6 +346,15 @@ macro_rules! bitfield { Self::from_raw(val) } } + + // SAFETY: `$name` is transparent over `$storage` and `$storage` has no interior mutability. + unsafe impl $crate::mem::AsRepr for $name { + // Normalize `$storage` to the canonical repr type in case it is signed. + type Repr = <$storage as $crate::mem::AsRepr>::Repr; + } + + // SAFETY: `$name` is transparent over `$storage`. + unsafe impl $crate::mem::AsReprMut for $name {} }; // Definitions requiring knowledge of individual fields: private and public field accessors, diff --git a/rust/kernel/debugfs.rs b/rust/kernel/debugfs.rs index d7b8014a6474..2beb55d444ca 100644 --- a/rust/kernel/debugfs.rs +++ b/rust/kernel/debugfs.rs @@ -538,7 +538,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { } } - fn create_file<T: Sync>(&self, name: &CStr, data: &'data T, vtable: &'static FileOps<T>) { + fn create_file<T: Sync>(&self, name: &CStr, data: &'data T, vtable: &FileOps<T>) { #[cfg(CONFIG_DEBUG_FS)] core::mem::forget(Entry::file(name, &self.entry, data, vtable)); } @@ -550,7 +550,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { /// This function does not produce an owning handle to the file. The created /// file is removed when the [`Scope`] that this directory belongs /// to is dropped. - pub fn read_only_file<T: Writer + Send + Sync + 'static>(&self, name: &CStr, data: &'data T) { + pub fn read_only_file<T: Writer + Send + Sync>(&self, name: &CStr, data: &'data T) { self.create_file(name, data, &T::FILE_OPS) } @@ -560,11 +560,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { /// /// This function does not produce an owning handle to the file. The created file is removed /// when the [`Scope`] that this directory belongs to is dropped. - pub fn read_binary_file<T: BinaryWriter + Send + Sync + 'static>( - &self, - name: &CStr, - data: &'data T, - ) { + pub fn read_binary_file<T: BinaryWriter + Send + Sync>(&self, name: &CStr, data: &'data T) { self.create_file(name, data, &T::FILE_OPS) } @@ -596,11 +592,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { /// This function does not produce an owning handle to the file. The created /// file is removed when the [`Scope`] that this directory belongs /// to is dropped. - pub fn read_write_file<T: Writer + Reader + Send + Sync + 'static>( - &self, - name: &CStr, - data: &'data T, - ) { + pub fn read_write_file<T: Writer + Reader + Send + Sync>(&self, name: &CStr, data: &'data T) { let vtable = &<T as ReadWriteFile<_>>::FILE_OPS; self.create_file(name, data, vtable) } @@ -612,7 +604,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { /// /// This function does not produce an owning handle to the file. The created file is removed /// when the [`Scope`] that this directory belongs to is dropped. - pub fn read_write_binary_file<T: BinaryWriter + BinaryReader + Send + Sync + 'static>( + pub fn read_write_binary_file<T: BinaryWriter + BinaryReader + Send + Sync>( &self, name: &CStr, data: &'data T, @@ -655,7 +647,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { /// This function does not produce an owning handle to the file. The created /// file is removed when the [`Scope`] that this directory belongs /// to is dropped. - pub fn write_only_file<T: Reader + Send + Sync + 'static>(&self, name: &CStr, data: &'data T) { + pub fn write_only_file<T: Reader + Send + Sync>(&self, name: &CStr, data: &'data T) { let vtable = &<T as WriteFile<_>>::FILE_OPS; self.create_file(name, data, vtable) } @@ -666,11 +658,7 @@ impl<'data, 'dir> ScopedDir<'data, 'dir> { /// /// This function does not produce an owning handle to the file. The created file is removed /// when the [`Scope`] that this directory belongs to is dropped. - pub fn write_binary_file<T: BinaryReader + Send + Sync + 'static>( - &self, - name: &CStr, - data: &'data T, - ) { + pub fn write_binary_file<T: BinaryReader + Send + Sync>(&self, name: &CStr, data: &'data T) { self.create_file(name, data, &T::FILE_OPS) } diff --git a/rust/kernel/debugfs/entry.rs b/rust/kernel/debugfs/entry.rs index 46aad64896ec..88a870d8c295 100644 --- a/rust/kernel/debugfs/entry.rs +++ b/rust/kernel/debugfs/entry.rs @@ -74,7 +74,7 @@ impl Entry<'static> { parent.as_ptr(), core::ptr::from_ref(data) as *mut c_void, core::ptr::null(), - &**file_ops, + file_ops.fops(), ) }; @@ -127,7 +127,7 @@ impl<'a> Entry<'a> { parent.as_ptr(), core::ptr::from_ref(data) as *mut c_void, core::ptr::null(), - &**file_ops, + file_ops.fops(), ) }; diff --git a/rust/kernel/debugfs/file_ops.rs b/rust/kernel/debugfs/file_ops.rs index f15908f71c4a..47acc4851d47 100644 --- a/rust/kernel/debugfs/file_ops.rs +++ b/rust/kernel/debugfs/file_ops.rs @@ -20,9 +20,6 @@ use crate::{ use core::marker::PhantomData; -#[cfg(CONFIG_DEBUG_FS)] -use core::ops::Deref; - /// # Invariant /// /// `FileOps<T>` will always contain an `operations` which is safe to use for a file backed @@ -30,7 +27,7 @@ use core::ops::Deref; /// into a reference. pub(super) struct FileOps<T> { #[cfg(CONFIG_DEBUG_FS)] - operations: bindings::file_operations, + operations: &'static bindings::file_operations, #[cfg(CONFIG_DEBUG_FS)] mode: u16, _phantom: PhantomData<T>, @@ -41,7 +38,7 @@ impl<T> FileOps<T> { /// /// The caller asserts that the provided `operations` is safe to use for a file whose /// inode has a pointer to `T` in its private data that is safe to convert into a reference. - const unsafe fn new(operations: bindings::file_operations, mode: u16) -> Self { + const unsafe fn new(operations: &'static bindings::file_operations, mode: u16) -> Self { Self { #[cfg(CONFIG_DEBUG_FS)] operations, @@ -65,11 +62,11 @@ impl<T: Adapter> FileOps<T> { } #[cfg(CONFIG_DEBUG_FS)] -impl<T> Deref for FileOps<T> { - type Target = bindings::file_operations; - - fn deref(&self) -> &Self::Target { - &self.operations +impl<T> FileOps<T> { + /// Returns a `'static` reference to the inner `file_operations`. + #[inline] + pub(crate) fn fops(&self) -> &'static bindings::file_operations { + self.operations } } @@ -130,7 +127,7 @@ pub(crate) trait ReadFile<T> { impl<T: Writer + Sync> ReadFile<T> for T { const FILE_OPS: FileOps<T> = { - let operations = bindings::file_operations { + let operations = &bindings::file_operations { read: Some(bindings::seq_read), llseek: Some(bindings::seq_lseek), release: Some(bindings::single_release), @@ -181,7 +178,7 @@ pub(crate) trait ReadWriteFile<T> { impl<T: Writer + Reader + Sync> ReadWriteFile<T> for T { const FILE_OPS: FileOps<T> = { - let operations = bindings::file_operations { + let operations = &bindings::file_operations { open: Some(writer_open::<T>), read: Some(bindings::seq_read), write: Some(write::<T>), @@ -238,7 +235,7 @@ pub(crate) trait WriteFile<T> { impl<T: Reader + Sync> WriteFile<T> for T { const FILE_OPS: FileOps<T> = { - let operations = bindings::file_operations { + let operations = &bindings::file_operations { open: Some(write_only_open), write: Some(write_only_write::<T>), llseek: Some(bindings::noop_llseek), @@ -290,7 +287,7 @@ pub(crate) trait BinaryReadFile<T> { impl<T: BinaryWriter + Sync> BinaryReadFile<T> for T { const FILE_OPS: FileOps<T> = { - let operations = bindings::file_operations { + let operations = &bindings::file_operations { read: Some(blob_read::<T>), llseek: Some(bindings::default_llseek), open: Some(bindings::simple_open), @@ -344,7 +341,7 @@ pub(crate) trait BinaryWriteFile<T> { impl<T: BinaryReader + Sync> BinaryWriteFile<T> for T { const FILE_OPS: FileOps<T> = { - let operations = bindings::file_operations { + let operations = &bindings::file_operations { write: Some(blob_write::<T>), llseek: Some(bindings::default_llseek), open: Some(bindings::simple_open), @@ -368,7 +365,7 @@ pub(crate) trait BinaryReadWriteFile<T> { impl<T: BinaryWriter + BinaryReader + Sync> BinaryReadWriteFile<T> for T { const FILE_OPS: FileOps<T> = { - let operations = bindings::file_operations { + let operations = &bindings::file_operations { read: Some(blob_read::<T>), write: Some(blob_write::<T>), llseek: Some(bindings::default_llseek), diff --git a/rust/kernel/device_id.rs b/rust/kernel/device_id.rs index c81fca5b4986..f0b9cb84e58e 100644 --- a/rust/kernel/device_id.rs +++ b/rust/kernel/device_id.rs @@ -146,8 +146,7 @@ impl<T: RawDeviceId, const N: usize> IdArray<T, (), N> { /// If the device implements [`RawDeviceIdIndex`], consider using [`IdArray::new`] instead. pub const fn new_without_index(ids: [T; N]) -> Self { // SAFETY: `T` is layout-wise compatible with `T::RawType`, so is the array of them. - let raw_ids: [MaybeUninit<T::RawType>; N] = unsafe { core::mem::transmute_copy(&ids) }; - core::mem::forget(ids); + let raw_ids: [MaybeUninit<T::RawType>; N] = unsafe { crate::mem::transmute(ids) }; Self { ids: raw_ids, diff --git a/rust/kernel/dma.rs b/rust/kernel/dma.rs index 2ce09f8e90c6..4ce914b7d1da 100644 --- a/rust/kernel/dma.rs +++ b/rust/kernel/dma.rs @@ -24,7 +24,6 @@ use crate::{ }, prelude::*, ptr::KnownSize, - sync::aref::ARef, transmute::{ AsBytes, FromBytes, // @@ -223,7 +222,7 @@ impl DmaMask { /// /// # fn test(dev: &Device<Bound>) -> Result { /// let attribs = DMA_ATTR_FORCE_CONTIGUOUS | DMA_ATTR_NO_WARN; -/// let c: Coherent<[u64]> = +/// let c: Coherent<'_, [u64]> = /// Coherent::zeroed_slice_with_attrs(dev, 4, GFP_KERNEL, attribs)?; /// # Ok::<(), Error>(()) } /// ``` @@ -390,9 +389,9 @@ impl From<DataDirection> for bindings::dma_data_direction { /// }; /// /// # fn test(dev: &Device<Bound>) -> Result { -/// let mut dmem: CoherentBox<u64> = CoherentBox::zeroed(dev, GFP_KERNEL)?; +/// let mut dmem: CoherentBox<'_, u64> = CoherentBox::zeroed(dev, GFP_KERNEL)?; /// *dmem = 42; -/// let dmem: Coherent<u64> = dmem.into(); +/// let dmem: Coherent<'_, u64> = dmem.into(); /// # Ok::<(), Error>(()) } /// ``` /// @@ -410,18 +409,18 @@ impl From<DataDirection> for bindings::dma_data_direction { /// }; /// /// # fn test(dev: &Device<Bound>) -> Result { -/// let mut dmem: CoherentBox<[u64]> = CoherentBox::zeroed_slice(dev, 4, GFP_KERNEL)?; +/// let mut dmem: CoherentBox<'_, [u64]> = CoherentBox::zeroed_slice(dev, 4, GFP_KERNEL)?; /// dmem.fill(42); -/// let dmem: Coherent<[u64]> = dmem.into(); +/// let dmem: Coherent<'_, [u64]> = dmem.into(); /// # Ok::<(), Error>(()) } /// ``` -pub struct CoherentBox<T: KnownSize + ?Sized>(Coherent<T>); +pub struct CoherentBox<'a, T: KnownSize + ?Sized>(Coherent<'a, T>); -impl<T: AsBytes + FromBytes> CoherentBox<[T]> { +impl<'a, T: AsBytes + FromBytes> CoherentBox<'a, [T]> { /// [`CoherentBox`] variant of [`Coherent::zeroed_slice_with_attrs`]. #[inline] pub fn zeroed_slice_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, count: usize, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, @@ -432,7 +431,7 @@ impl<T: AsBytes + FromBytes> CoherentBox<[T]> { /// Same as [CoherentBox::zeroed_slice_with_attrs], but with `dma::Attrs(0)`. #[inline] pub fn zeroed_slice( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, count: usize, gfp_flags: kernel::alloc::Flags, ) -> Result<Self> { @@ -480,14 +479,14 @@ impl<T: AsBytes + FromBytes> CoherentBox<[T]> { /// /// # fn test(dev: &Device<Bound>) -> Result { /// let data = [0u8, 1u8, 2u8, 3u8]; - /// let c: CoherentBox<[u8]> = + /// let c: CoherentBox<'_, [u8]> = /// CoherentBox::from_slice_with_attrs(dev, &data, GFP_KERNEL, DMA_ATTR_NO_WARN)?; /// /// assert_eq!(c.deref(), &data); /// # Ok::<(), Error>(()) } /// ``` pub fn from_slice_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, data: &[T], gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, @@ -512,7 +511,7 @@ impl<T: AsBytes + FromBytes> CoherentBox<[T]> { /// `dma_attrs` is 0 by default. #[inline] pub fn from_slice( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, data: &[T], gfp_flags: kernel::alloc::Flags, ) -> Result<Self> @@ -523,11 +522,11 @@ impl<T: AsBytes + FromBytes> CoherentBox<[T]> { } } -impl<T: AsBytes + FromBytes> CoherentBox<T> { +impl<'a, T: AsBytes + FromBytes> CoherentBox<'a, T> { /// Same as [`CoherentBox::zeroed_slice_with_attrs`], but for a single element. #[inline] pub fn zeroed_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, ) -> Result<Self> { @@ -536,12 +535,12 @@ impl<T: AsBytes + FromBytes> CoherentBox<T> { /// Same as [`CoherentBox::zeroed_slice`], but for a single element. #[inline] - pub fn zeroed(dev: &device::Device<Bound>, gfp_flags: kernel::alloc::Flags) -> Result<Self> { + pub fn zeroed(dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags) -> Result<Self> { Self::zeroed_with_attrs(dev, gfp_flags, Attrs(0)) } } -impl<T: KnownSize + ?Sized> Deref for CoherentBox<T> { +impl<T: KnownSize + ?Sized> Deref for CoherentBox<'_, T> { type Target = T; #[inline] @@ -554,7 +553,7 @@ impl<T: KnownSize + ?Sized> Deref for CoherentBox<T> { } } -impl<T: AsBytes + FromBytes + KnownSize + ?Sized> DerefMut for CoherentBox<T> { +impl<T: AsBytes + FromBytes + KnownSize + ?Sized> DerefMut for CoherentBox<'_, T> { #[inline] fn deref_mut(&mut self) -> &mut Self::Target { // SAFETY: @@ -565,9 +564,9 @@ impl<T: AsBytes + FromBytes + KnownSize + ?Sized> DerefMut for CoherentBox<T> { } } -impl<T: AsBytes + FromBytes + KnownSize + ?Sized> From<CoherentBox<T>> for Coherent<T> { +impl<'a, T: AsBytes + FromBytes + KnownSize + ?Sized> From<CoherentBox<'a, T>> for Coherent<'a, T> { #[inline] - fn from(value: CoherentBox<T>) -> Self { + fn from(value: CoherentBox<'a, T>) -> Self { value.0 } } @@ -588,26 +587,20 @@ impl<T: AsBytes + FromBytes + KnownSize + ?Sized> From<CoherentBox<T>> for Coher /// to an allocated region of coherent memory and `dma_addr` is the DMA address base of the /// region. /// - The size in bytes of the allocation is equal to size information via pointer. -// TODO // -// DMA allocations potentially carry device resources (e.g.IOMMU mappings), hence for soundness -// reasons DMA allocation would need to be embedded in a `Devres` container, in order to ensure -// that device resources can never survive device unbind. -// -// However, it is neither desirable nor necessary to protect the allocated memory of the DMA -// allocation from surviving device unbind; it would require RCU read side critical sections to -// access the memory, which may require subsequent unnecessary copies. -// -// Hence, find a way to revoke the device resources of a `Coherent`, but not the -// entire `Coherent` including the allocated memory itself. -pub struct Coherent<T: KnownSize + ?Sized> { - dev: ARef<device::Device>, +// The lifetime parameter ties DMA allocations to the device's bound scope, ensuring they are freed +// before the device is unbound under normal circumstances. However, if a `Coherent` is leaked (e.g. +// via `mem::forget`), device resources such as IOMMU mappings will not be released. Making all +// constructors `unsafe` to prevent this is considered too restrictive for the common case; this +// soundness hole is accepted for now. +pub struct Coherent<'a, T: KnownSize + ?Sized> { + dev: &'a device::Device<Bound>, dma_addr: DmaAddress, cpu_addr: NonNull<T>, dma_attrs: Attrs, } -impl<T: KnownSize + ?Sized> Coherent<T> { +impl<T: KnownSize + ?Sized> Coherent<'_, T> { /// Returns the size in bytes of this allocation. #[inline] pub fn size(&self) -> usize { @@ -663,10 +656,10 @@ impl<T: KnownSize + ?Sized> Coherent<T> { } } -impl<T: AsBytes + FromBytes> Coherent<T> { +impl<'a, T: AsBytes + FromBytes> Coherent<'a, T> { /// Allocates a region of `T` of coherent memory. fn alloc_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, ) -> Result<Self> { @@ -692,9 +685,9 @@ impl<T: AsBytes + FromBytes> Coherent<T> { // INVARIANT: // - We just successfully allocated a coherent region which is adequately sized for `T`, // hence the cpu address is valid. - // - We also hold a refcounted reference to the device. + // - `dev` is a valid reference to a bound device that outlives this allocation. Ok(Self { - dev: dev.into(), + dev, dma_addr, cpu_addr, dma_attrs, @@ -716,13 +709,13 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// }; /// /// # fn test(dev: &Device<Bound>) -> Result { - /// let c: Coherent<[u64; 4]> = + /// let c: Coherent<'_, [u64; 4]> = /// Coherent::zeroed_with_attrs(dev, GFP_KERNEL, DMA_ATTR_NO_WARN)?; /// # Ok::<(), Error>(()) } /// ``` #[inline] pub fn zeroed_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, ) -> Result<Self> { @@ -732,14 +725,14 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// Performs the same functionality as [`Coherent::zeroed_with_attrs`], except the /// `dma_attrs` is 0 by default. #[inline] - pub fn zeroed(dev: &device::Device<Bound>, gfp_flags: kernel::alloc::Flags) -> Result<Self> { + pub fn zeroed(dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags) -> Result<Self> { Self::zeroed_with_attrs(dev, gfp_flags, Attrs(0)) } /// Same as [`Coherent::zeroed_with_attrs`], but instead of a zero-initialization the memory is /// initialized with `init`. pub fn init_with_attrs<E>( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, init: impl Init<T, E>, @@ -764,7 +757,7 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// with `init`. #[inline] pub fn init<E>( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, gfp_flags: kernel::alloc::Flags, init: impl Init<T, E>, ) -> Result<Self> @@ -776,11 +769,11 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// Allocates a region of `[T; len]` of coherent memory. fn alloc_slice_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, len: usize, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, - ) -> Result<Coherent<[T]>> { + ) -> Result<Coherent<'a, [T]>> { const { assert!( core::mem::size_of::<T>() > 0, @@ -809,9 +802,9 @@ impl<T: AsBytes + FromBytes> Coherent<T> { // INVARIANT: // - We just successfully allocated a coherent region which is adequately sized for // `[T; len]`, hence the cpu address is valid. - // - We also hold a refcounted reference to the device. + // - `dev` is a valid reference to a bound device that outlives this allocation. Ok(Coherent { - dev: dev.into(), + dev, dma_addr, cpu_addr, dma_attrs, @@ -836,17 +829,17 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// }; /// /// # fn test(dev: &Device<Bound>) -> Result { - /// let c: Coherent<[u64]> = + /// let c: Coherent<'_, [u64]> = /// Coherent::zeroed_slice_with_attrs(dev, 4, GFP_KERNEL, DMA_ATTR_NO_WARN)?; /// # Ok::<(), Error>(()) } /// ``` #[inline] pub fn zeroed_slice_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, len: usize, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, - ) -> Result<Coherent<[T]>> { + ) -> Result<Coherent<'a, [T]>> { Coherent::alloc_slice_with_attrs(dev, len, gfp_flags | __GFP_ZERO, dma_attrs) } @@ -854,10 +847,10 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// `dma_attrs` is 0 by default. #[inline] pub fn zeroed_slice( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, len: usize, gfp_flags: kernel::alloc::Flags, - ) -> Result<Coherent<[T]>> { + ) -> Result<Coherent<'a, [T]>> { Self::zeroed_slice_with_attrs(dev, len, gfp_flags, Attrs(0)) } @@ -876,18 +869,18 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// # fn test(dev: &Device<Bound>) -> Result { /// let data = [0u8, 1u8, 2u8, 3u8]; /// // `c` has the same content as `data`. - /// let c: Coherent<[u8]> = + /// let c: Coherent<'_, [u8]> = /// Coherent::from_slice_with_attrs(dev, &data, GFP_KERNEL, DMA_ATTR_NO_WARN)?; /// /// # Ok::<(), Error>(()) } /// ``` #[inline] pub fn from_slice_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, data: &[T], gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, - ) -> Result<Coherent<[T]>> + ) -> Result<Coherent<'a, [T]>> where T: Copy, { @@ -898,10 +891,10 @@ impl<T: AsBytes + FromBytes> Coherent<T> { /// `dma_attrs` is 0 by default. #[inline] pub fn from_slice( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, data: &[T], gfp_flags: kernel::alloc::Flags, - ) -> Result<Coherent<[T]>> + ) -> Result<Coherent<'a, [T]>> where T: Copy, { @@ -909,7 +902,7 @@ impl<T: AsBytes + FromBytes> Coherent<T> { } } -impl<T> Coherent<[T]> { +impl<T> Coherent<'_, [T]> { /// Returns the number of elements `T` in this allocation. /// /// Note that this is not the size of the allocation in bytes, which is provided by @@ -922,10 +915,10 @@ impl<T> Coherent<[T]> { } /// Note that the device configured to do DMA must be halted before this object is dropped. -impl<T: KnownSize + ?Sized> Drop for Coherent<T> { +impl<T: KnownSize + ?Sized> Drop for Coherent<'_, T> { fn drop(&mut self) { let size = T::size(self.cpu_addr.as_ptr()); - // SAFETY: Device pointer is guaranteed as valid by the type invariant on `Device`. + // SAFETY: Device pointer is guaranteed as valid by the lifetime of this `Coherent`. // The cpu address, and the dma address are valid due to the type invariants on // `Coherent`. unsafe { @@ -942,15 +935,15 @@ impl<T: KnownSize + ?Sized> Drop for Coherent<T> { // SAFETY: It is safe to send a `Coherent` to another thread if `T` // can be sent to another thread. -unsafe impl<T: KnownSize + Send + ?Sized> Send for Coherent<T> {} +unsafe impl<T: KnownSize + Send + ?Sized> Send for Coherent<'_, T> {} // SAFETY: Sharing `&Coherent` across threads is safe if `T` is `Sync`, because all // methods that access the buffer contents (`field_read`, `field_write`, `as_slice`, // `as_slice_mut`) are `unsafe`, and callers are responsible for ensuring no data races occur. // The safe methods only return metadata or raw pointers whose use requires `unsafe`. -unsafe impl<T: KnownSize + ?Sized + AsBytes + FromBytes + Sync> Sync for Coherent<T> {} +unsafe impl<T: KnownSize + ?Sized + AsBytes + FromBytes + Sync> Sync for Coherent<'_, T> {} -impl<T: KnownSize + AsBytes + ?Sized> debugfs::BinaryWriter for Coherent<T> { +impl<T: KnownSize + AsBytes + ?Sized> debugfs::BinaryWriter for Coherent<'_, T> { fn write_to_slice( &self, writer: &mut UserSliceWriter, @@ -996,15 +989,15 @@ impl<T: KnownSize + AsBytes + ?Sized> debugfs::BinaryWriter for Coherent<T> { /// - `size` is the allocation size in bytes as passed to `dma_alloc_attrs`. /// - `dma_attrs` contains the attributes used for the allocation, always including /// `DMA_ATTR_NO_KERNEL_MAPPING`. -pub struct CoherentHandle { - dev: ARef<device::Device>, +pub struct CoherentHandle<'a> { + dev: &'a device::Device<Bound>, dma_addr: DmaAddress, cpu_handle: NonNull<c_void>, size: usize, dma_attrs: Attrs, } -impl CoherentHandle { +impl<'a> CoherentHandle<'a> { /// Allocates `size` bytes of coherent DMA memory without creating a kernel virtual mapping. /// /// Additional DMA attributes may be passed via `dma_attrs`; `DMA_ATTR_NO_KERNEL_MAPPING` is @@ -1012,7 +1005,7 @@ impl CoherentHandle { /// /// Returns `EINVAL` if `size` is zero, `ENOMEM` if the allocation fails. pub fn alloc_with_attrs( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, size: usize, gfp_flags: kernel::alloc::Flags, dma_attrs: Attrs, @@ -1038,9 +1031,9 @@ impl CoherentHandle { // INVARIANT: `cpu_handle` is the opaque handle from a successful `dma_alloc_attrs` call // with `DMA_ATTR_NO_KERNEL_MAPPING`, `dma_addr` is the corresponding DMA address, - // and we hold a refcounted reference to the device. + // and `dev` is a valid reference to a bound device that outlives this allocation. Ok(Self { - dev: dev.into(), + dev, dma_addr, cpu_handle, size, @@ -1051,7 +1044,7 @@ impl CoherentHandle { /// Allocates `size` bytes of coherent DMA memory without creating a kernel virtual mapping. #[inline] pub fn alloc( - dev: &device::Device<Bound>, + dev: &'a device::Device<Bound>, size: usize, gfp_flags: kernel::alloc::Flags, ) -> Result<Self> { @@ -1073,7 +1066,7 @@ impl CoherentHandle { } } -impl Drop for CoherentHandle { +impl Drop for CoherentHandle<'_> { fn drop(&mut self) { // SAFETY: All values are valid by the type invariants on `CoherentHandle`. // `cpu_handle` is the opaque handle from `dma_alloc_attrs` and is passed back unchanged. @@ -1091,12 +1084,12 @@ impl Drop for CoherentHandle { // SAFETY: `CoherentHandle` only holds a device reference, a DMA address, an opaque CPU handle, // and a size. None of these are tied to a specific thread. -unsafe impl Send for CoherentHandle {} +unsafe impl Send for CoherentHandle<'_> {} // SAFETY: `CoherentHandle` provides no CPU access to the underlying allocation. The only // operations on `&CoherentHandle` are reading the DMA address and size, both of which are // plain `Copy` values. -unsafe impl Sync for CoherentHandle {} +unsafe impl Sync for CoherentHandle<'_> {} /// View type for `Coherent`. /// @@ -1236,7 +1229,7 @@ impl<'a, T: ?Sized + KnownSize> IoBase<'a> for CoherentView<'a, T> { } } -impl<'a, T: ?Sized + KnownSize> IoBase<'a> for &'a Coherent<T> { +impl<'a, T: ?Sized + KnownSize> IoBase<'a> for &'a Coherent<'_, T> { type Backend = CoherentIoBackend; type Target = T; diff --git a/rust/kernel/dma_buf/dma_fence.rs b/rust/kernel/dma_buf/dma_fence.rs new file mode 100644 index 000000000000..18a43e1bb442 --- /dev/null +++ b/rust/kernel/dma_buf/dma_fence.rs @@ -0,0 +1,1022 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (C) 2025-2026 Red Hat Inc. + * Author: Philipp Stanner <pstanner@redhat.com> + */ + +//! DMA Fence support. +//! +//! Reference: <https://docs.kernel.org/driver-api/dma-buf.html#c.dma_fence> +//! +//! header: [`include/linux/dma-fence.h`](srctree/include/linux/dma-fence.h) + +use crate::{ + alloc::AllocError, + bindings, + container_of, + error::to_result, + prelude::*, + types::ForeignOwnable, + types::Opaque, // +}; + +use core::{ + marker::PhantomData, + mem::ManuallyDrop, + ops::Deref, + ptr, + ptr::{ + drop_in_place, + NonNull, // + }, // +}; + +use kernel::{ + str::CString, + sync::{ + aref::{ + ARef, + AlwaysRefCounted, // + }, + atomic::{ + Atomic, + Relaxed, // + }, + rcu::rcu_barrier, // + }, // +}; + +/// VTable for dma_fence backend_ops callbacks. +// +// Mandatory dma_fence backend_ops are implemented implicitly through +// [`FenceContext`]. Additional ones shall get implemented on this trait. +pub trait FenceContextOps { + /// The generic payload data for [`DriverFence`]s created on this fctx. + type FenceDataType: Send + Sync; +} + +/// A dma-fence context. A fence context takes care of associating related fences +/// with each other, providing each with raising sequence numbers and a common +/// identifier. +#[pin_data(PinnedDrop)] +pub struct FenceContext<T: FenceContextOps + Send + Sync> { + /// The fence context number. + nr: u64, + /// The sequence number for the next fence created. + seqno: Atomic<u64>, + // The name parameters can be accessed by the dma_fence backend_ops. UAF + // errors are prevented by the `call_rcu()` in `drop_driver_fence_data()`. + /// The name of the driver this FenceContext's fences belong to. + driver_name: CString, + /// The name of the timeline this FenceContext's fences belong to. + timeline_name: CString, + /// The number of all unsignaled fences on this context. + // Used to prevent bugs due to forgotten fences. + // + // The lifetime on `DriverFence`s should typically prevent this from + // happening. + // + // However, we cannot fully guarantee in Rust that `DriverFence`s will not + // be forgotten, e.g., through `core::mem::forget()`. This could circumvent + // the lifetime which intends to enforce that all fences disappear before + // their context. + nr_of_unsignaled_fences: Atomic<usize>, + /// The user's data. + #[pin] + data: T, +} + +impl<'a, T: Send + Sync + FenceContextOps> FenceContext<T> { + // This can later be extended as a vtable in case other parties need support + // for the more "exotic" callbacks. + const OPS: bindings::dma_fence_ops = bindings::dma_fence_ops { + get_driver_name: Some(Self::get_driver_name), + get_timeline_name: Some(Self::get_timeline_name), + enable_signaling: None, + signaled: None, + // Deprecated. + wait: None, + // Must never be implemented for these abstractions. + release: None, + set_deadline: None, + }; + + /// Create a new `FenceContext`. + pub fn new<E>( + initial_seqno: u64, + driver_name: &CStr, + timeline_name: &CStr, + data: impl PinInit<T, E>, + ) -> impl PinInit<Self, Error> + where + Error: From<E>, + { + let driver_name = CString::try_from(driver_name); + let timeline_name = CString::try_from(timeline_name); + try_pin_init!(Self { + // SAFETY: `dma_fence_context_alloc()` merely works on a global + // atomic. Parameter `1` is the number of contexts we want to + // allocate. + nr: unsafe { bindings::dma_fence_context_alloc(1) }, + seqno: Atomic::new(initial_seqno), + driver_name: driver_name?, + timeline_name: timeline_name?, + nr_of_unsignaled_fences: Atomic::new(0), + data <- data, + }) + } + + fn next_seqno(&self) -> u64 { + self.seqno.fetch_add(1, Relaxed) + } + + /// Allocate the memory for a [`DriverFence`] and already store `data` inside. + /// + /// This is needed because many times, creation of a [`DriverFence`] must not + /// fail, and allocating might deadlock in some situations. + /// + /// The `data` you pass here must not perform any operations that are illegal + /// in atomic context in its [`Drop`] implementation. + pub fn new_fence_allocation( + &self, + data: T::FenceDataType, + ) -> Result<DriverFenceAllocation<'_, T>> { + let fence_data = DriverFenceData { + rcu_head: Default::default(), + // `inner` remains uninitialized until a `DriverFence` takes over. + inner: Fence { + inner: Opaque::uninit(), + }, + fctx: self, + data, + }; + + // In order to support the C dma_fence callbacks, it is necessary for + // a `Fence` and a `DriverFence` to live in the same allocation, + // because the C backend passes a dma_fence, from which the driver most + // likely wants to be able to access its `data` in `DriverFence`. + // + // Hence, we need the manage the memory manually. It will be freed by the + // C backend automatically once the refcount within `Fence` drops to 0. + let data = KBox::new(fence_data, GFP_KERNEL | __GFP_ZERO)?; + + Ok(DriverFenceAllocation { + data, + ops: &Self::OPS, + }) + } + + extern "C" fn get_driver_name(ptr: *mut bindings::dma_fence) -> *const c_char { + // SAFETY: The C backend only invokes this callback with `ptr` pointing + // to a valid, unsignaled `bindings::dma_fence`. All fences created in + // this module always reside within `Fence` which always resides in a + // `DriverFenceData`, thus satisfying the function's safety + // requirements. + let fctx = unsafe { Self::from_raw_fence(ptr) }; + + fctx.driver_name.as_char_ptr() + } + + extern "C" fn get_timeline_name(ptr: *mut bindings::dma_fence) -> *const c_char { + // SAFETY: The C backend only invokes this callback with `ptr` pointing + // to a valid, unsignaled `bindings::dma_fence`. All fences created in + // this module always reside within `Fence` which always resides in a + // `DriverFenceData`, thus satisfying the function's safety + // requirements. + let fctx = unsafe { Self::from_raw_fence(ptr) }; + + fctx.timeline_name.as_char_ptr() + } + + /// Create a [`FenceContext`] from an associated [`bindings::dma_fence`]. + /// + /// # Safety + /// + /// `ptr` must be a valid pointer to a [`bindings::dma_fence`] which resides + /// within a [`Fence`], which in turn resides in a [`DriverFenceData`]. + unsafe fn from_raw_fence(ptr: *mut bindings::dma_fence) -> &'a Self { + let opaque_fence = Opaque::cast_from(ptr); + + // SAFETY: Safe due to the function's overall safety requirements. + let fence_ptr = unsafe { container_of!(opaque_fence, Fence, inner) }; + + // CAST: `DriverFenceData` is `repr(C)` and a `Fence` is its first member. + let fence_data_ptr: *const DriverFenceData<'a, T> = fence_ptr.cast(); + + // SAFETY: Safe because of the comments directly above. + let fence_data = unsafe { &*fence_data_ptr }; + + fence_data.fctx + } +} + +#[pinned_drop] +impl<T: FenceContextOps + Send + Sync> PinnedDrop for FenceContext<T> { + fn drop(self: Pin<&mut Self>) { + // Fence ops callbacks can be called on unsignaled fences. Since these + // callbacks can access the fence context and its data, it needs to be + // guaranteed that a context only drops after all associated + // `DriverFence`s have been dropped. This is unlikely to occur, but + // would result in silent UAF. Throw a panic to prevent that. + // + // TODO: + // It would be better if the fence context signals all forgotten fences + // itself. To do so, it would keep a list of unsignaled fences. That + // list's members would have to be pre-allocated (see + // `FenceCallback::new_fence_allocation()`). + if self.nr_of_unsignaled_fences.load(Relaxed) != 0 { + panic!("Forgotten fences in FenceContext."); + } + + // Ensure that the driver cannot unload while there are still dma_fence + // callbacks running. At the same time, the RCU barrier addresses the + // problem inherited by the C backend, in which backend ops callbacks + // might be accessing the fence while it is being signaled (or shortly + // after). This could cause UAF access on the fence context's + // `fctx.driver_name` and `fctx.timeline_name`. + // + // Wait for the RCU callbacks in `DriverFence::drop`. + rcu_barrier(); + } +} + +/// Error type for fence callback registration. +/// +/// Generic over `T` so that `AlreadySignaled` can return the callback to the +/// caller, allowing it to reclaim any resources owned by the callback (e.g., +/// a fence handle that needs to be signaled). +#[derive(Debug)] +pub enum CallbackError<T> { + /// The fence was already signaled. The callback is returned so the caller + /// can extract owned resources without losing them. + AlreadySignaled(T), + /// Some other error occurred during registration. + Other(Error), +} + +impl<T> From<CallbackError<T>> for Error { + #[inline] + fn from(err: CallbackError<T>) -> Self { + match err { + CallbackError::AlreadySignaled(_) => ENOENT, + CallbackError::Other(e) => e, + } + } +} + +impl<T> From<AllocError> for CallbackError<T> { + #[inline] + fn from(e: AllocError) -> Self { + CallbackError::Other(Error::from(e)) + } +} + +/// Trait for callbacks that can be registered on fences. +/// +/// When the fence signals, the callback will be invoked. +/// +/// # Example +/// +/// ```rust +/// use kernel::dma_buf::FenceCallback; +/// +/// struct MyCallback { +/// // Your callback state here +/// } +/// +/// impl FenceCallback for MyCallback { +/// fn on_signal(&mut self) { +/// pr_info!("Fence signaled!\n"); +/// // Handle fence completion +/// } +/// } +/// ``` +pub trait FenceCallback: Send + 'static { + /// Called when the fence is signaled. + /// + /// This is called from the fence signaling path, which may be in interrupt + /// context or with locks held, which is why `self` is only borrowed, so that + /// it cannot drop. Implementations must not sleep or perform + /// long-running operations. + /// + /// An implementation likely wants to inform itself (e.g., through a work item) + /// within this callback that the associated [`FenceCallbackRegistration`] + /// can now be dropped. + fn on_signal(&mut self); +} + +/// A callback registration on a fence. +/// +/// When this object is dropped, the callback is automatically removed if it +/// hasn't been called yet. +#[pin_data(PinnedDrop)] +pub struct FenceCallbackRegistration<T: FenceCallback + 'static> { + #[pin] + callback_foreign: Opaque<bindings::dma_fence_cb>, + callback: ManuallyDrop<T>, + fence: ARef<Fence>, +} + +impl<T: FenceCallback> FenceCallbackRegistration<T> { + /// Create a [`PinInit`] closure for registering a callback on a fence. + /// + /// The actual attempt at registering the callback will take place once you + /// call an allocator's `pin_init()` function. + /// + /// On success the callback is pinned in place and will fire when the fence + /// signals. On `AlreadySignaled` the callback is returned to the caller so + /// that owned resources can be reclaimed. + pub fn new<'a>(fence: &'a Fence, callback: T) -> impl PinInit<Self, CallbackError<T>> + 'a + where + T: 'a, + { + try_pin_init!(Self { + // We need to fully initialize the fence because after + // `dma_fence_add_callback()` ran, the callback might immediately + // get invoked. + callback: ManuallyDrop::new(callback), + fence: ARef::from(fence), + callback_foreign <- Opaque::try_ffi_init(|ptr| { + // SAFETY: `fence.inner.get()` is a valid, initialized `struct + // dma_fence`. `ptr` points to the `struct dma_fence_cb` field + // within the pinned allocation, so it remains valid until + // `dma_fence_remove_callback()` in `PinnedDrop` or until the + // callback fires. + let ret = unsafe { + to_result(bindings::dma_fence_add_callback( + fence.inner.get(), + ptr, + Some(Self::dma_fence_callback), + )) + }; + match ret { + Ok(()) => Ok(()), + Err(e) => { + // SAFETY: We could not register the callback. Thus, + // C will not use it. So we can just take it back + // and pass it to the user again. + let cb_back = unsafe { ManuallyDrop::take(callback) }; + if e == ENOENT { + Err(CallbackError::AlreadySignaled(cb_back)) + } else { + Err(CallbackError::Other(e)) + } + }, + } + }), + }? CallbackError<T>) + } + + /// Raw dma fence callback that is called by the C code. + /// + /// # Safety + /// + /// This is only called by the dma_fence subsystem with valid pointers. + unsafe extern "C" fn dma_fence_callback( + _fence: *mut bindings::dma_fence, + callback_foreign: *mut bindings::dma_fence_cb, + ) { + let ptr = Opaque::cast_from(callback_foreign).cast_mut(); + + // SAFETY: All callbacks we can receive here have been created in such a way that they are + // embedded into a `FenceCallbackRegistration`. + let reg: *mut Self = unsafe { container_of!(ptr, Self, callback_foreign) }; + + // SAFETY: `reg` is a valid `Self` pointer. + // + // The backend ensures synchronisation so whoever holds the registration object cannot drop + // it while this code is running. See `FenceCallbackRegistration::drop`. + unsafe { (*reg).callback.on_signal() }; + } + + /// Returns a reference to the fence this callback is registered on. + #[inline] + pub fn fence(&self) -> &Fence { + &self.fence + } +} + +#[pinned_drop] +impl<T: FenceCallback> PinnedDrop for FenceCallbackRegistration<T> { + fn drop(self: Pin<&mut Self>) { + // Always call `dma_fence_remove_callback()`, even if the callback + // already ran. This is necessary for synchronization: + // `dma_fence_remove_callback()` acquires `fence->lock`, which ensures + // that any in-flight `dma_fence_signal()` (which calls our callback + // while holding the same lock) has completed before we free the struct. + // + // Without this, Drop can race with a concurrent signal: + // CPU0 (signal, lock held): take() -> on_signal(fence_ref) (in progress) + // CPU1 (drop): skips lock -> frees struct + // CPU0: accesses fence_ref -> use-after-free + // + // When the callback has already fired, the signal path detached the + // list node via `INIT_LIST_HEAD()`, so dma_fence_remove_callback just + // sees an empty node and returns false — the lock acquisition is the + // only thing that matters. + // + // SAFETY: The fence pointer is valid and the cb was initialized by + // `dma_fence_add_callback()` during construction. + unsafe { + bindings::dma_fence_remove_callback(self.fence.as_raw(), self.callback_foreign.get()) + }; + + // SAFETY: This is literally the drop implementation, so no one has + // dropped this so far; so we can do it now. + unsafe { ManuallyDrop::<T>::drop(self.project().callback) }; + } +} + +// SAFETY: FenceCallbackRegistration can be sent between threads. +unsafe impl<T: FenceCallback> Send for FenceCallbackRegistration<T> {} + +// SAFETY: &FenceCallbackRegistration can be shared between threads if &T can. +unsafe impl<T: FenceCallback> Sync for FenceCallbackRegistration<T> where T: Sync {} + +/// The receiving counterpart of a [`DriverFence`]. +/// +/// The Rust DMA fence implementation has a dualistic design: [`DriverFence`]s +/// are the producer-side, intended to be always owned by only one party. That +/// party has the monopoly on signaling the fence. +/// +/// A [`Fence`] is the counterpart for consumers. Thus, [`Fence`]s are always +/// refcounted and can shared with an arbitrary number of parties, including +/// userspace. A [`Fence`] can only be used for actions such as checking the +/// fence's status or for registering callbacks on it. +/// +/// Once the associated [`DriverFence`] signals, all +/// [`FenceCallbackRegistration`]s registered on the [`Fence`] will be executed. +/// +/// A [`Fence`] can arbitrarily outlive its [`DriverFence`] and the +/// [`FenceContext`]. Signaling a [`DriverFence`] decouples it from its +/// [`Fence`]s. +#[repr(transparent)] +pub struct Fence { + /// The actual dma_fence passed to C. + inner: Opaque<bindings::dma_fence>, +} + +/// Guard helper for locking within this module. +/// +/// Its only purpose for now is to avoid a number of unsafe lock-unlock cycles. +/// It is never used outside of this module. +// TODO: This should be made more canonical, probably by basing it on a +// SpinLockIrqGuard once available. +struct FenceGuard<'a> { + inner: &'a Fence, + flags: usize, +} + +impl<'a> Deref for FenceGuard<'a> { + type Target = &'a Fence; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +impl Drop for FenceGuard<'_> { + fn drop(&mut self) { + // SAFETY: `fence` is valid because `self` is valid. `flag_ptr` is + // merely a pointer to an integer, which lives as long as this function. + // When a `FenceGuard` exists, the lock has been taken by definition. + unsafe { bindings::dma_fence_unlock_irqrestore(self.as_raw(), &raw mut self.flags) }; + } +} + +// SAFETY: Fences are literally designed to be shared between threads. +unsafe impl Send for Fence {} +// SAFETY: Fences are literally designed to be shared between threads. +unsafe impl Sync for Fence {} + +impl Fence { + /// Check whether the fence was signaled at the moment of the function call. + /// + /// Note that this can return `true` for a [`Fence`] whose [`DriverFence`] + /// has not yet been dropped. The reason is that the fence ops callbacks can + /// cause the fence to get signaled by the C backend. + #[inline] + pub fn is_signaled(&self) -> bool { + // We should not use `dma_fence_is_signaled_locked()` here, because + // according to the C backend's recommendations, that function is + // problematic and we should avoid calling that function with a lock + // held. + + // SAFETY: Inner `fence` is valid because `self` is valid. + let ret = unsafe { bindings::dma_fence_is_signaled(self.as_raw()) }; + + // To be as robust as possible for the future we guarantee that an API + // caller can 100% rely on the signaling being completed (i.e., all + // fence callbacks ran), so we have to take the lock. + // + // The reason is that the C dma_fence backend currently does not + // carefully synchronize the `dma_fence_is_signaled()` function with the + // proper spinlock. This can lead to the function returning `true` while + // fence callbacks are still being executed. This can be mitigated by + // guarding the entire function with the spinlock. + // + // The fundamental reason is that the C backend currently does guard + // setting of the fence's signaled-bit with the fence's spinlock, but + // reading is done locklessly. + // + // See commit c8a5d5ea3ba6a. + let _ = self.lock(); + + ret + } + + /// Lock the fence. A helper only to be used internally in this module. + fn lock(&self) -> FenceGuard<'_> { + let mut guard = FenceGuard { + inner: self, + flags: 0, + }; + + // SAFETY: `fence` is valid because `self` is valid. `flag_ptr` is + // merely a pointer to an integer, whose lifetime is tied to the guard + // object. + unsafe { bindings::dma_fence_lock_irqsave(self.as_raw(), &raw mut guard.flags) }; + + guard + } + + /// Get the fence's sequence number. + #[inline] + pub fn seqno(&self) -> u64 { + // SAFETY: Valid because `self` is valid. + unsafe { (*self.as_raw()).seqno } + } + + fn as_raw(&self) -> *mut bindings::dma_fence { + self.inner.get() + } + + /// Create a [`Fence`] from a raw C [`bindings::dma_fence`]. + /// + /// # Safety + /// + /// `ptr` must point to an initialized fence that is embedded into a [`Fence`]. + #[inline] + pub unsafe fn from_raw<'a>(ptr: *mut bindings::dma_fence) -> &'a Self { + // SAFETY: Safe as per the function's overall safety requirements. + unsafe { &*ptr.cast() } + } +} + +// SAFETY: These implement the C backends refcounting methods which are proven +// to work correctly. +unsafe impl AlwaysRefCounted for Fence { + fn inc_ref(&self) { + // SAFETY: `self.as_raw()` is a pointer to a valid `struct dma_fence`. + unsafe { bindings::dma_fence_get(self.as_raw()) } + } + + unsafe fn dec_ref(ptr: NonNull<Self>) { + // SAFETY: `ptr` is never a NULL pointer; and when `dec_ref()` is called + // the fence is by definition still valid. + let fence = unsafe { (*ptr.as_ptr()).inner.get() }; + + // SAFETY: `fence` was created validly above. When `dec_ref()` is called, + // there is by definition still a reference alive that can be put. + unsafe { bindings::dma_fence_put(fence) } + } +} + +// Necessary to guarantee that `inner` always comes first and can be freed by C. +// Also useful for using casts instead of container_of(). +#[repr(C)] +#[pin_data] +struct DriverFenceData<'a, T: Send + Sync + FenceContextOps> { + #[pin] + /// The inner fence. + // Must always be the first member so that unsafe casting works; but also + // necessary so that the C backend can free the allocation (coming from our + // Rust code) with kfree_rcu(). + inner: Fence, + /// Callback head for dropping this in a deferred manner through RCU. + rcu_head: bindings::callback_head, + /// Reference to access the FenceContext. + fctx: &'a FenceContext<T>, + /// The API user's data. It is essential that the data only performs + /// operations legal in atomic context in its [`Drop`] implementation. + #[pin] + data: T::FenceDataType, +} + +/// A synchronization primitive mainly for GPU drivers. +/// +/// The Rust DMA fence implementation has a dualistic design: [`DriverFence`]s +/// are the producer-side, intended to be always owned by only one party. That +/// party has the monopoly on signaling the fence. +/// +/// A [`Fence`] is the counterpart for consumers. Thus, [`Fence`]s are always +/// refcounted and can be shared with an arbitrary number of parties, including +/// userspace. A [`Fence`] can only be used for actions such as checking the +/// fence's status or for registering callbacks on it. +/// +/// Once the associated [`DriverFence`] signals, all +/// [`FenceCallbackRegistration`]s registered on a [`Fence`] will be executed. +/// +/// A [`Fence`] can arbitrarily outlive its [`DriverFence`] and the +/// [`FenceContext`]. Signaling a [`DriverFence`] decouples it from its +/// [`Fence`]s. +/// +/// It is crucial that a [`DriverFence`] always correctly represents the state +/// of the associated job on the hardware. Especially, it is strictly necessary +/// that the owner ensures that all [`DriverFence`]s eventually get signaled. +/// As a last resort, a [`DriverFence`] will signal itself if it drops +/// unsignaled and print a warning. +/// +/// This design intends to implement the [`bindings::dma_fence_ops`] in such a +/// way that the driver-data necessary to implement the callback's functionality +/// resides in the [`FenceContext`]. Thus, a [`DriverFence`] contains a +/// reference to the context, which can be accessed in the callbacks. The +/// implementation, therefore, ensures that a [`DriverFence`] cannot outlive its +/// [`FenceContext`]. Unfortunately, this can be circumvented under certain +/// circumstances in Rust (e.g., usage of [`core::mem::forget`]). +/// +/// In the unlikely case of such violations, a panic is thrown. +/// +/// # Examples +/// +/// ``` +/// use kernel::{ +/// dma_buf::{ +/// DriverFence, +/// FenceContext, +/// FenceContextOps, +/// FenceCallback, +/// FenceCallbackRegistration, +/// }, +/// str::CString, +/// sync::aref::ARef, // +/// }; +/// use core::fmt::Display; +/// +/// struct CallbackData { } +/// +/// impl FenceCallback for CallbackData { +/// fn on_signal(&mut self) { +/// pr_info!("DmaFence callback executed.\n"); +/// } +/// } +/// +/// #[pin_data] +/// struct FenceContextData {} +/// +/// impl FenceContextData { +/// fn new() -> impl PinInit<Self> { +/// pin_init!(Self {}) +/// } +/// } +/// +/// impl FenceContextOps for FenceContextData { +/// type FenceDataType = FenceData; +/// } +/// +/// let fctx_data = FenceContextData::new(); +/// +/// +/// let mut fctx = KBox::pin_init( +/// FenceContext::new(0, c"dummy_driver", c"dummy_timeline", fctx_data), +/// GFP_KERNEL +/// )?; +/// +/// struct FenceData { +/// data: CString, +/// } +/// +/// let fence_data = FenceData { data: c"dummy_data".try_into()? }; +/// +/// let fence_alloc = fctx.new_fence_allocation(fence_data)?; +/// let mut fence = fence_alloc.new_fence(); +/// +/// let cb_data = CallbackData { }; +/// let waiting_fence = ARef::from(fence.as_fence()); +/// let cb_reg = FenceCallbackRegistration::new(&waiting_fence, cb_data); +/// let cb_reg = KBox::pin_init(cb_reg, GFP_KERNEL)?; +/// +/// // TODO signalling guards +/// assert_eq!(waiting_fence.is_signaled(), false); +/// fence.signal(Ok(())); +/// assert_eq!(waiting_fence.is_signaled(), true); +/// +/// Ok::<(), Error>(()) +/// ``` +pub struct DriverFence<'a, T: Send + Sync + FenceContextOps> { + /// The actual content of the fence. Lives in a [`NonNull`] so that its + /// memory can be managed independently. Valid until both the [`DriverFence`] + /// and all associated [`Fence`]s have disappeared. + data: NonNull<DriverFenceData<'a, T>>, +} + +/// A pre-prepared DMA fence, carrying the user's data and the memory it and the +/// fence reside in. Only useful for creating a [`DriverFence`]. Splitting +/// allocation and full initialization is necessary because fences cannot be +/// allocated dynamically in some circumstances (deadlock). +pub struct DriverFenceAllocation<'a, T: Send + Sync + FenceContextOps> { + /// The memory for the actual content of the fence. + /// Handed over to a [`DriverFence`], or deallocated once the + /// [`DriverFenceAllocation`] drops. + data: KBox<DriverFenceData<'a, T>>, + /// Reference for the ops for the associated [`FenceContext`] + ops: &'static bindings::dma_fence_ops, +} + +impl<'a, T: Send + Sync + FenceContextOps> DriverFenceAllocation<'a, T> { + /// Create a new [`DriverFence`], the signalable counterpart of a [`Fence`]. + /// + /// This increments the sequence number in the associated [`FenceContext`]. + pub fn new_fence(self) -> DriverFence<'a, T> { + // We feed the C dma_fence backend a NULL for the spinlock so that it + // uses per-fence locks automatically. + let null_ptr: *mut bindings::spinlock = ptr::null_mut(); + let seqno = self.data.fctx.next_seqno(); + let fence_ptr = self.as_raw(); + // SAFETY: `fence_ptr` has been created directly above. It will live + // at least as long as `Self`. The same applies to `&Self::OPS`. + unsafe { + bindings::dma_fence_init(fence_ptr, self.ops, null_ptr, self.data.fctx.nr, seqno) + }; + + self.data.fctx.nr_of_unsignaled_fences.fetch_add(1, Relaxed); + + // A `DriverFenceAllocation`'s purpose is to carry allocated memory, so + // that `DriverFence`s can always be created without allocating. In this + // method, ownership over that memory is transferred to the new + // `DriverFence` and managed through refcounting. The C dma_fence + // backend will ultimately free the memory once the refcount reaches 0. + let ptr = KBox::into_raw(self.data); + // SAFETY: `ptr` was just created validly directly above. + let ptr = unsafe { NonNull::new_unchecked(ptr) }; + + DriverFence { data: ptr } + } + + fn as_raw(&self) -> *mut bindings::dma_fence { + self.data.inner.inner.get() + } +} + +impl<'a, T: Send + Sync + FenceContextOps> DriverFence<'a, T> { + fn as_raw(&self) -> *mut bindings::dma_fence { + // SAFETY: Valid because `self` is valid. + let fence_data = unsafe { &*self.data.as_ptr() }; + + fence_data.inner.inner.get() + } + + /// Create a [`DriverFence`] from a raw pointer to a [`bindings::dma_fence`]. + /// + /// # Safety + /// + /// `ptr` must be a valid pointer to a `dma_fence` that was obtained through + /// a [`DriverFence`] with matching generic data for both fence and associated + /// [`FenceContext`]. + unsafe fn from_raw(ptr: *mut bindings::dma_fence) -> Self { + let opaque_fence = Opaque::cast_from(ptr); + + // SAFETY: Safe due to the function's overall safety requirements. + let fence_ptr = unsafe { container_of!(opaque_fence, Fence, inner) }; + + // DriverFenceData is `repr(C)` and a Fence is its first member. + let fence_data_ptr = fence_ptr as *mut DriverFenceData<'a, T>; + + // SAFETY: `fence_data_ptr` was created validly above. + let data = unsafe { NonNull::new_unchecked(fence_data_ptr) }; + + Self { data } + } + + /// Return the underlying [`Fence`]. + #[inline] + pub fn as_fence(&self) -> &Fence { + // SAFETY: `self` is by definition still valid, and it cannot drop until + // this new reference is gone. + unsafe { Fence::from_raw(self.as_raw()) } + } + + /// Signal the fence. This will invoke all registered callbacks. + pub fn signal(self, res: Result) { + let fence = self.as_fence().lock(); + + // SAFETY: `fence` is valid because `self` is valid. The lock must be + // held, which we acquired directly above. + if !unsafe { bindings::dma_fence_test_signaled_flag(fence.as_raw()) } { + if let Err(err) = res { + // SAFETY: `fence` is valid because `self` is valid. The fence + // must not have been signaled yet, which we check directly above. + unsafe { bindings::dma_fence_set_error(fence.as_raw(), err.to_errno()) }; + } + // SAFETY: `fence` is valid because `self` is valid. The lock must + // be held, which we acquired above. + unsafe { bindings::dma_fence_signal_locked(fence.as_raw()) }; + } + + // SAFETY: `self.data` is valid because `self` is valid. + let fctx = unsafe { self.data.as_ref().fctx }; + let _ = fctx.nr_of_unsignaled_fences.fetch_sub(1, Relaxed); + } +} + +// SAFETY: Fences are literally designed to be shared between threads. +unsafe impl<'a, T: Send + Sync + FenceContextOps> Send for DriverFence<'a, T> {} +// SAFETY: Fences are literally designed to be shared between threads. +unsafe impl<'a, T: Send + Sync + FenceContextOps> Sync for DriverFence<'a, T> {} + +impl<'a, T: Send + Sync + FenceContextOps> Deref for DriverFence<'a, T> { + type Target = T::FenceDataType; + + fn deref(&self) -> &Self::Target { + // SAFETY: Thanks to refcounting, `data` is always valid as long as `self` is. + let data = unsafe { &*self.data.as_ptr() }; + + &data.data + } +} + +/// A borrow wrapper for [`DriverFence`]. Implements [`Deref`]. +pub struct DriverFenceBorrow<'a, T: Send + Sync + FenceContextOps> { + driver_fence: ManuallyDrop<DriverFence<'a, T>>, + _lifetime: PhantomData<&'a T>, +} + +impl<'a, T: Send + Sync + FenceContextOps> Deref for DriverFenceBorrow<'a, T> { + type Target = DriverFence<'a, T>; + + fn deref(&self) -> &Self::Target { + self.driver_fence.deref() + } +} + +// SAFETY: The Rust dma_fence abstractions are already designed around the inner +// C `dma_fence`, which can serve safely as the identification point when being +// owned by C. Moreover, safety is ensured by not dropping `DriverFence` and by +// only allowing operations without side effects on the Borrowed type. +unsafe impl<T: Send + Sync + FenceContextOps> ForeignOwnable for DriverFence<'_, T> { + type Borrowed<'a> + = DriverFenceBorrow<'a, T> + where + Self: 'a; + type BorrowedMut<'a> + = DriverFenceBorrow<'a, T> + where + Self: 'a; + + const FOREIGN_ALIGN: usize = core::mem::align_of::<bindings::dma_fence>(); + + fn into_foreign(self) -> *mut c_void { + let fence = self; + + let ptr = fence.as_raw(); + + // DriverFence must not drop. + let _ = ManuallyDrop::new(fence); + + ptr.cast() + } + + unsafe fn from_foreign(ptr: *mut c_void) -> Self { + // SAFETY: Safe because the trait implementation only invokes this with + // a valid `ptr`, associated to a `DriverFence` with matching generic data. + unsafe { Self::from_raw(ptr.cast()) } + } + + unsafe fn borrow<'a>(ptr: *mut c_void) -> Self::Borrowed<'a> + where + Self: 'a, + { + // SAFETY: The trait implementation ensures that `ptr` always resides + // within a [`Fence`] within a [`DriverFenceData`]. + let driver_fence = unsafe { Self::from_raw(ptr.cast()) }; + + let driver_fence = ManuallyDrop::new(driver_fence); + + DriverFenceBorrow { + driver_fence, + _lifetime: PhantomData, + } + } + + unsafe fn borrow_mut<'a>(ptr: *mut c_void) -> Self::BorrowedMut<'a> + // FIXME: The bound below and the one above in `borrow` should actually be + // unnecessary since the compiler should be able to completely derive all + // necessary information automatically. There is currently a compiler bug + // preventing that, though: + // + // https://github.com/rust-lang/rust/issues/155430. + // + // (Help to) fix the compiler bug and remove the bounds afterwards. + where + Self: 'a, + { + // SAFETY: The trait implementation ensures that `ptr` always resides + // within a [`Fence`] within a [`DriverFenceData`]. + let driver_fence = unsafe { Self::from_raw(ptr.cast()) }; + + let driver_fence = ManuallyDrop::new(driver_fence); + + DriverFenceBorrow { + driver_fence, + _lifetime: PhantomData, + } + } +} + +impl<'a, T: Send + Sync + FenceContextOps> Drop for DriverFence<'a, T> { + fn drop(&mut self) { + let guard = self.as_fence().lock(); + + // Use dma_fence_test_signaled_flag() instead of + // dma_fence_is_signaled_locked() because the C backend wants to get rid + // of the latter. + + // SAFETY: `guard` is valid until the `call_rcu()` below. + let signaled: bool = unsafe { bindings::dma_fence_test_signaled_flag(guard.as_raw()) }; + if !signaled { + pr_err!("DriverFence drops unsignaled. Danger of memory corruption!\n"); + // SAFETY: `guard` is valid until the `call_rcu()` below. The fence + // must not have been signaled yet, which we check directly above. + unsafe { bindings::dma_fence_set_error(guard.as_raw(), ECANCELED.to_errno()) }; + // SAFETY: `guard` is valid until the `call_rcu()` below. The lock + // must be held, which we acquired above. + unsafe { bindings::dma_fence_signal_locked(guard.as_raw()) }; + + // SAFETY: `self.data` is valid because `self` is valid. + let fctx = unsafe { self.data.as_ref().fctx }; + let _ = fctx.nr_of_unsignaled_fences.fetch_sub(1, Relaxed); + } + drop(guard); + + // `DriverFenceData` could be accessed through some dma_fence + // callbacks right now. Access is being revoked in principle above by + // signaling the fence, but since the C backend does not guarantee + // perfect full synchronization, we have to wait for one grace period to + // ensure that all accessors of `DriverFenceData` (through the + // dma_fence_ops accessible through a `Fence`) are gone. + + if !core::mem::needs_drop::<T::FenceDataType>() { + // SAFETY: Once a `DriverFence` is initialized, the inner `fence` is + // valid and initialized. It is valid until the refcount drops to 0, + // which can earliest happen once we drop the `DriverFence`'s + // reference here. + unsafe { bindings::dma_fence_put(self.as_raw()) }; + return; + } + + // SAFETY: Valid because `self` is valid. + let rcu_head_ptr = unsafe { &raw mut (*self.data.as_ptr()).rcu_head }; + + // SAFETY: `call_rcu()` is always safe to be called. `rcu_head_ptr` was + // created validly above. The module must perform a `synchronize_rcu()` + // or `rcu_barrier()` call to guard against module unload. + unsafe { bindings::call_rcu(rcu_head_ptr, Some(drop_driver_fence_data::<T>)) }; + } +} + +// TODO: +// The entire call_rcu() mechanism in the drop above and the code below would be +// unnecessary if C's dma_fence_signal() could be reworked in a way that after it +// ran, the caller knows that no fence_ops callbacks can be running anymore. +// In other words, if the dma_fence backend would use its spinlock for full +// synchronization. +// +// Then we could move the drop_in_place() and dma_fence_put() upwards into the +// drop() implementation and call it a day. + +/// Finally really drop this `DriverFence<T>` +/// +/// # Safety +/// +/// `head` references the `rcu_head` field of an `DriverFenceData<T>`. All +/// accessors to that `DriverFenceData<T>` must be gone by now. This must be +/// ensured by signalling the associated `DriverFence<T>` and then waiting +/// for a grace period until calling this function here. +unsafe extern "C" fn drop_driver_fence_data<T: Send + Sync + FenceContextOps>( + head: *mut bindings::callback_head, +) { + // SAFETY: Caller provides a pointer to the `rcu_head` field of a `DriverFenceData<C>`. + let fence_data = unsafe { container_of!(head, DriverFenceData<'_, T>, rcu_head) }; + + // SAFETY: `fence_data` was created validly above. All the fence's data will + // only drop below, but the raw pointer to the raw C `dma_fence` remains + // valid because the reference count is only decremented at the end of the + // function. + let fence = unsafe { (*fence_data).inner.inner.get() }; + + // SAFETY: `fence_data` was created validly above. The user has already + // dropped the only conventional accessor to the user data, the `DriverFence`, + // one grace period ago. All accessors are gone now. + unsafe { drop_in_place(&raw mut (*fence_data).data) }; + + // The inner `Fence` explicitly does not get dropped because there may be + // many more users / consumers, each holding their own reference. + + // SAFETY: Once a `DriverFence` is initialized, the inner `fence` is valid + // and initialized. It is valid until the refcount drops to 0, which can + // earliest happen once we drop the `DriverFence`'s reference here. + unsafe { bindings::dma_fence_put(fence) }; + + // The actual memory the data associated with a `DriverFence` lives in + // gets freed by the C dma_fence backend once the fence's refcount reaches 0. +} diff --git a/rust/kernel/dma_buf/mod.rs b/rust/kernel/dma_buf/mod.rs new file mode 100644 index 000000000000..4764a828642e --- /dev/null +++ b/rust/kernel/dma_buf/mod.rs @@ -0,0 +1,14 @@ +// SPDX-License-Identifier: GPL-2.0 OR MIT + +//! DMA-buf subsystem abstractions. + +pub mod dma_fence; + +pub use self::dma_fence::{ + DriverFence, + Fence, + FenceCallback, + FenceCallbackRegistration, + FenceContext, + FenceContextOps, // +}; diff --git a/rust/kernel/io.rs b/rust/kernel/io.rs index 5ce9fd129068..de8ef8e2aec4 100644 --- a/rust/kernel/io.rs +++ b/rust/kernel/io.rs @@ -11,6 +11,10 @@ use core::{ use crate::{ bindings, + mem::{ + AsRepr, + AsReprMut, // + }, prelude::*, ptr::{ Alignment, @@ -226,6 +230,17 @@ fn io_view<'a, IO: Io<'a>, U>( Ok(unsafe { IO::Backend::project_view(view, projected_ptr) }) } +/// Returns the primitive view of a I/O view. +#[inline] +fn io_view_as_repr<'a, IO: Io<'a, Target = T>, T: AsRepr>( + this: IO, +) -> <IO::Backend as IoBackend>::View<'a, T::Repr> { + let view = this.as_view(); + + // SAFETY: `AsRepr` guarantees layout compatibility. + unsafe { IO::Backend::project_view(view, IO::Backend::as_ptr(view).cast::<T::Repr>()) } +} + /// I/O backends. /// /// This is an abstract representation to be implemented by arbitrary I/O @@ -353,15 +368,12 @@ pub trait IoCopyable: IoBackend { /// /// - The valid `Base` to operate on. For most registers, this should be [`Region`]. /// - The offset to access (returned by [`IoLoc::offset`]), -/// - The width of the access (determined by [`IoLoc::IoType`]), -/// - The type `T` in which the raw data is returned or provided. +/// - The type `T` in which the data is returned or provided. /// -/// `T` and `IoLoc::IoType` may differ: for instance, a typed register has `T` = the register type -/// with its bitfields, and `IoType` = its backing primitive (e.g. `u32`). +/// `T` is not necessarily the type for underlying I/O operation. Methods that take `IoLoc` have `T: +/// AsRepr` bound and the `<T as AsRepr>::Repr` type would be used to perform I/O and converted to +/// `T` instead. pub trait IoLoc<Base: ?Sized, T> { - /// Size ([`u8`], [`u16`], etc) of the I/O performed on the returned [`offset`](IoLoc::offset). - type IoType: Into<T> + From<T>; - /// Consumes `self` and returns the offset of this location. fn offset(self) -> usize; } @@ -372,8 +384,6 @@ macro_rules! impl_usize_ioloc { ($($ty:ty),*) => { $( impl<const SIZE: usize> IoLoc<Region<SIZE>, $ty> for usize { - type IoType = $ty; - #[inline(always)] fn offset(self) -> usize { self @@ -437,6 +447,45 @@ pub trait Io<'a>: IoBase<'a> { self.len() == 0 } + /// Convert into a different typed I/O view. + /// + /// The target type must be known (statically) to be of the same or smaller size to current + /// type, and the current view must be properly aligned for the target type. + /// + /// # Examples + /// + /// ```no_run + /// use kernel::io::{ + /// io_project, + /// Mmio, + /// Io, + /// Region, + /// }; + /// #[derive(FromBytes, IntoBytes)] + /// #[repr(C)] + /// struct MyStruct { field: u32, } + /// + /// # fn test(mmio: &Mmio<'_, Region<0x1000>>) { + /// // let mmio: Mmio<'_, Region<0x1000>>; + /// let whole: Mmio<'_, MyStruct> = mmio.cast(); + /// # } + /// ``` + #[inline] + fn cast<U>(self) -> <Self::Backend as IoBackend>::View<'a, U> + where + Self::Target: FromBytes + IntoBytes, + U: FromBytes + IntoBytes, + { + let view = self.as_view(); + let ptr = Self::Backend::as_ptr(view); + + const_assert!(size_of::<U>() <= Self::Target::MIN_SIZE); + const_assert!(align_of::<U>() <= Self::Target::MIN_ALIGN.as_usize()); + + // SAFETY: We have checked bounds and alignment, so this is a valid projection. + unsafe { Self::Backend::project_view(view, ptr.cast()) } + } + /// Try to convert into a different typed I/O view. /// /// A runtime check is performed to ensure that the target type is of same or smaller size to @@ -498,10 +547,10 @@ pub trait Io<'a>: IoBase<'a> { #[inline] fn read_val(self) -> Self::Target where - Self::Backend: IoCapable<Self::Target>, - Self::Target: Sized, + Self::Target: AsReprMut, + Self::Backend: IoCapable<<Self::Target as AsRepr>::Repr>, { - Self::Backend::io_read(self.as_view()) + Self::Target::from_repr(Self::Backend::io_read(io_view_as_repr(self))) } /// Write a value to I/O. @@ -520,10 +569,10 @@ pub trait Io<'a>: IoBase<'a> { #[inline] fn write_val(self, value: Self::Target) where - Self::Backend: IoCapable<Self::Target>, - Self::Target: Sized, + Self::Target: AsRepr, + Self::Backend: IoCapable<<Self::Target as AsRepr>::Repr>, { - Self::Backend::io_write(self.as_view(), value) + Self::Backend::io_write(io_view_as_repr(self), Self::Target::into_repr(value)) } /// Copy-read from I/O memory. @@ -645,7 +694,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_read8(self, offset: usize) -> Result<u8> where - usize: IoLoc<Self::Target, u8, IoType = u8>, + usize: IoLoc<Self::Target, u8>, Self::Backend: IoCapable<u8>, { self.try_read(offset) @@ -655,7 +704,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_read16(self, offset: usize) -> Result<u16> where - usize: IoLoc<Self::Target, u16, IoType = u16>, + usize: IoLoc<Self::Target, u16>, Self::Backend: IoCapable<u16>, { self.try_read(offset) @@ -665,7 +714,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_read32(self, offset: usize) -> Result<u32> where - usize: IoLoc<Self::Target, u32, IoType = u32>, + usize: IoLoc<Self::Target, u32>, Self::Backend: IoCapable<u32>, { self.try_read(offset) @@ -675,7 +724,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_read64(self, offset: usize) -> Result<u64> where - usize: IoLoc<Self::Target, u64, IoType = u64>, + usize: IoLoc<Self::Target, u64>, Self::Backend: IoCapable<u64>, { self.try_read(offset) @@ -685,7 +734,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_write8(self, value: u8, offset: usize) -> Result where - usize: IoLoc<Self::Target, u8, IoType = u8>, + usize: IoLoc<Self::Target, u8>, Self::Backend: IoCapable<u8>, { self.try_write(offset, value) @@ -695,7 +744,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_write16(self, value: u16, offset: usize) -> Result where - usize: IoLoc<Self::Target, u16, IoType = u16>, + usize: IoLoc<Self::Target, u16>, Self::Backend: IoCapable<u16>, { self.try_write(offset, value) @@ -705,7 +754,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_write32(self, value: u32, offset: usize) -> Result where - usize: IoLoc<Self::Target, u32, IoType = u32>, + usize: IoLoc<Self::Target, u32>, Self::Backend: IoCapable<u32>, { self.try_write(offset, value) @@ -715,7 +764,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_write64(self, value: u64, offset: usize) -> Result where - usize: IoLoc<Self::Target, u64, IoType = u64>, + usize: IoLoc<Self::Target, u64>, Self::Backend: IoCapable<u64>, { self.try_write(offset, value) @@ -727,7 +776,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn read8(self, offset: usize) -> u8 where - usize: IoLoc<Self::Target, u8, IoType = u8>, + usize: IoLoc<Self::Target, u8>, Self::Backend: IoCapable<u8>, { self.read(offset) @@ -739,7 +788,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn read16(self, offset: usize) -> u16 where - usize: IoLoc<Self::Target, u16, IoType = u16>, + usize: IoLoc<Self::Target, u16>, Self::Backend: IoCapable<u16>, { self.read(offset) @@ -751,7 +800,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn read32(self, offset: usize) -> u32 where - usize: IoLoc<Self::Target, u32, IoType = u32>, + usize: IoLoc<Self::Target, u32>, Self::Backend: IoCapable<u32>, { self.read(offset) @@ -763,7 +812,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn read64(self, offset: usize) -> u64 where - usize: IoLoc<Self::Target, u64, IoType = u64>, + usize: IoLoc<Self::Target, u64>, Self::Backend: IoCapable<u64>, { self.read(offset) @@ -775,7 +824,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn write8(self, value: u8, offset: usize) where - usize: IoLoc<Self::Target, u8, IoType = u8>, + usize: IoLoc<Self::Target, u8>, Self::Backend: IoCapable<u8>, { self.write(offset, value) @@ -787,7 +836,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn write16(self, value: u16, offset: usize) where - usize: IoLoc<Self::Target, u16, IoType = u16>, + usize: IoLoc<Self::Target, u16>, Self::Backend: IoCapable<u16>, { self.write(offset, value) @@ -799,7 +848,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn write32(self, value: u32, offset: usize) where - usize: IoLoc<Self::Target, u32, IoType = u32>, + usize: IoLoc<Self::Target, u32>, Self::Backend: IoCapable<u32>, { self.write(offset, value) @@ -811,7 +860,7 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn write64(self, value: u64, offset: usize) where - usize: IoLoc<Self::Target, u64, IoType = u64>, + usize: IoLoc<Self::Target, u64>, Self::Backend: IoCapable<u64>, { self.write(offset, value) @@ -843,11 +892,11 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_read<T, L>(self, location: L) -> Result<T> where + T: AsReprMut, L: IoLoc<Self::Target, T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, { - let view = io_view::<Self, L::IoType>(self, location.offset())?; - Ok(Self::Backend::io_read(view).into()) + Ok(io_read!(self, try: location)) } /// Generic fallible write with runtime bounds check. @@ -876,12 +925,11 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_write<T, L>(self, location: L, value: T) -> Result where + T: AsRepr, L: IoLoc<Self::Target, T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, { - let view = io_view::<Self, L::IoType>(self, location.offset())?; - let io_value = value.into(); - Self::Backend::io_write(view, io_value); + io_write!(self, try: location, value); Ok(()) } @@ -900,6 +948,8 @@ pub trait Io<'a>: IoBase<'a> { /// }; /// /// register! { + /// base: Region; + /// /// VERSION(u32) @ 0x100 { /// 15:8 major; /// 7:0 minor; @@ -920,9 +970,10 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_write_reg<T, L, V>(self, value: V) -> Result where + T: AsRepr, L: IoLoc<Self::Target, T>, V: LocatedRegister<Self::Target, Location = L, Value = T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, { let (location, value) = value.into_io_op(); @@ -954,16 +1005,13 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn try_update<T, L, F>(self, location: L, f: F) -> Result where + T: AsReprMut, L: IoLoc<Self::Target, T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, F: FnOnce(T) -> T, { - let view = io_view::<Self, L::IoType>(self, location.offset())?; - - let value: T = Self::Backend::io_read(view).into(); - let io_value = f(value).into(); - Self::Backend::io_write(view, io_value); - + let view = io_project!(self, try: location); + view.write_val(f(view.read_val())); Ok(()) } @@ -991,11 +1039,11 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn read<T, L>(self, location: L) -> T where + T: AsReprMut, L: IoLoc<Self::Target, T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, { - let view = io_view_assert::<Self, L::IoType>(self, location.offset()); - Self::Backend::io_read(view).into() + io_read!(self, build: location) } /// Generic infallible write with compile-time bounds check. @@ -1022,12 +1070,11 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn write<T, L>(self, location: L, value: T) where + T: AsRepr, L: IoLoc<Self::Target, T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, { - let view = io_view_assert::<Self, L::IoType>(self, location.offset()); - let io_value = value.into(); - Self::Backend::io_write(view, io_value); + io_write!(self, build: location, value); } /// Generic infallible write of a fully-located register value. @@ -1045,6 +1092,8 @@ pub trait Io<'a>: IoBase<'a> { /// }; /// /// register! { + /// base: Region<0x1000>; + /// /// VERSION(u32) @ 0x100 { /// 15:8 major; /// 7:0 minor; @@ -1064,9 +1113,10 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn write_reg<T, L, V>(self, value: V) where + T: AsRepr, L: IoLoc<Self::Target, T>, V: LocatedRegister<Self::Target, Location = L, Value = T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, { let (location, value) = value.into_io_op(); @@ -1098,14 +1148,13 @@ pub trait Io<'a>: IoBase<'a> { #[inline(always)] fn update<T, L, F>(self, location: L, f: F) where + T: AsReprMut, L: IoLoc<Self::Target, T>, - Self::Backend: IoCapable<L::IoType>, + Self::Backend: IoCapable<<T as AsRepr>::Repr>, F: FnOnce(T) -> T, { - let view = io_view_assert::<Self, L::IoType>(self, location.offset()); - let value: T = Self::Backend::io_read(view).into(); - let io_value = f(value).into(); - Self::Backend::io_write(view, io_value); + let view = io_project!(self, build: location); + view.write_val(f(view.read_val())); } } @@ -1649,6 +1698,25 @@ where // SAFETY: Per safety requirement. unsafe { T::Backend::project_view::<T::Target, _>(self.0, ptr) } } + + #[inline(always)] + pub fn try_project_loc<U, L>( + self, + location: L, + ) -> Result<<T::Backend as IoBackend>::View<'a, U>> + where + L: IoLoc<T::Target, U>, + { + io_view::<_, U>(self.0, location.offset()) + } + + #[inline(always)] + pub fn project_loc<U, L>(self, location: L) -> <T::Backend as IoBackend>::View<'a, U> + where + L: IoLoc<T::Target, U>, + { + io_view_assert::<_, U>(self.0, location.offset()) + } } /// Project an I/O type to a subview of it. @@ -1656,26 +1724,54 @@ where /// The syntax is of form `io_project!(io, proj)` where `io` is an expression to a type that /// implements [`Io`] and `proj` is a [projection specification](kernel::ptr::project!). /// +/// `io_project!` can also project to a subview of registers defined with [`register!`] macro. +/// Register projection has syntax `io_project!(io, try: REGISTER)` for fallible projection and +/// `io_project!(io, build: REGISTER)` for infallible projection. +/// /// # Examples /// /// ``` /// use kernel::io::{ /// io_project, +/// register, /// Mmio, /// }; /// #[repr(C)] /// struct MyStruct { field: u32, } /// +/// register! { +/// base: MyStruct; +/// FIELD(u32) @ 0 { +/// 31:0 val; +/// } +/// } +/// /// # fn test(mmio: Mmio<'_, [MyStruct]>) -> Result { /// // let mmio: Mmio<[MyStruct]>; /// let field: Mmio<'_, u32> = io_project!(mmio, [try: 1].field); /// let whole: Mmio<'_, MyStruct> = io_project!(mmio, [try: 2]); /// let nested: Mmio<'_, u32> = io_project!(whole, .field); +/// let reg: Mmio<'_, FIELD> = io_project!(whole, build: FIELD); /// # Ok::<(), Error>(()) } /// ``` #[macro_export] #[doc(hidden)] macro_rules! io_project { + // Register projection + ($io:expr, try: $ioloc:expr) => {{ + #[allow(unused)] + use $crate::io::IoBase as _; + let view = $crate::io::ProjectHelper($io.as_view()); + view.try_project_loc($ioloc)? + }}; + ($io:expr, build: $ioloc:expr) => {{ + #[allow(unused)] + use $crate::io::IoBase as _; + let view = $crate::io::ProjectHelper($io.as_view()); + view.project_loc($ioloc) + }}; + + // Field or index projection ($io:expr, $($proj:tt)*) => {{ #[allow(unused)] use $crate::io::IoBase as _; @@ -1746,6 +1842,12 @@ macro_rules! io_write { (@parse [$io:expr] [$($proj:tt)*] [[$flavor:ident: $index:expr] $($rest:tt)*]) => { $crate::io_write!(@parse [$io] [$($proj)* [$flavor: $index]] [$($rest)*]) }; + (@parse [$io:expr] [] [try: $ioloc:expr, $($rest:tt)*]) => { + $crate::io_write!(@parse [$io] [try: $ioloc] [, $($rest)*]) + }; + (@parse [$io:expr] [] [build: $ioloc:expr, $($rest:tt)*]) => { + $crate::io_write!(@parse [$io] [build: $ioloc] [, $($rest)*]) + }; ($io:expr, $($rest:tt)*) => { $crate::io_write!(@parse [$io] [] [$($rest)*]) }; diff --git a/rust/kernel/io/register.rs b/rust/kernel/io/register.rs index 03dfd2ff48c7..b6513fa0f412 100644 --- a/rust/kernel/io/register.rs +++ b/rust/kernel/io/register.rs @@ -8,14 +8,19 @@ //! //! Note: most of the items in this module are public so they can be referenced by the macro, but //! most are not to be used directly by users. Outside of the `register!` macro itself, the only -//! items you might want to import from this module are [`WithBase`] and [`Array`]. +//! item you might want to import from this module is [`Array`]. //! //! # Simple example //! //! ```no_run -//! use kernel::io::register; +//! use kernel::io::{ +//! register, +//! Region, +//! }; //! //! register! { +//! base: Region<0x1000>; +//! //! /// Basic information about the chip. //! pub BOOT_0(u32) @ 0x00000100 { //! /// Vendor ID. @@ -55,11 +60,14 @@ //! register, //! Io, //! IoLoc, +//! Region, //! }, //! num::Bounded, //! }; -//! # use kernel::io::{Mmio, Region}; +//! # use kernel::io::Mmio; //! # register! { +//! # base: Region<0x1000>; +//! # //! # pub BOOT_0(u32) @ 0x00000100 { //! # 15:8 vendor_id; //! # 7:4 major_revision; @@ -113,149 +121,50 @@ use crate::{ io::IoLoc, // }; -use super::Region; - -/// Trait implemented by all registers. -pub trait Register: Sized { - /// Backing primitive type of the register. - type Storage: Into<Self> + From<Self>; - - /// Start offset of the register. - /// - /// The interpretation of this offset depends on the type of the register. - const OFFSET: usize; -} - -/// Trait implemented by registers with a fixed offset. -pub trait FixedRegister: Register {} - /// Allows `()` to be used as the `location` parameter of [`Io::write`](super::Io::write) when -/// passing a [`FixedRegister`] value. -impl<const SIZE: usize, T> IoLoc<Region<SIZE>, T> for () +/// passing a [`FixedIoLoc`] value. +impl<Base: ?Sized, T> IoLoc<Base, T> for () where - T: FixedRegister, + T: FixedIoLoc<Base>, { - type IoType = T::Storage; - #[inline(always)] fn offset(self) -> usize { - T::OFFSET + T::LOCATION.offset() } } -/// A [`FixedRegister`] carries its location in its type. Thus `FixedRegister` values can be used -/// as an [`IoLoc`]. -impl<const SIZE: usize, T> IoLoc<Region<SIZE>, T> for T -where - T: FixedRegister, -{ - type IoType = T::Storage; - - #[inline(always)] - fn offset(self) -> usize { - T::OFFSET - } -} - -/// Location of a fixed register. -pub struct FixedRegisterLoc<T: FixedRegister>(PhantomData<T>); - -impl<T: FixedRegister> FixedRegisterLoc<T> { - /// Returns the location of `T`. - #[inline(always)] - // We do not implement `Default` so we can be const. - #[expect(clippy::new_without_default)] - pub const fn new() -> Self { - Self(PhantomData) - } -} - -impl<const SIZE: usize, T> IoLoc<Region<SIZE>, T> for FixedRegisterLoc<T> -where - T: FixedRegister, -{ - type IoType = T::Storage; - - #[inline(always)] - fn offset(self) -> usize { - T::OFFSET - } -} - -/// Trait providing a base address to be added to the offset of a relative register to obtain -/// its actual offset. -/// -/// The `T` generic argument is used to distinguish which base to use, in case a type provides -/// several bases. It is given to the `register!` macro to restrict the use of the register to -/// implementors of this particular variant. -pub trait RegisterBase<T> { - /// Base address to which register offsets are added. - const BASE: usize; -} - -/// Trait implemented by all registers that are relative to a base. -pub trait WithBase { - /// Family of bases applicable to this register. - type BaseFamily; +// Provides a `IoLoc` impl that for a fixed offset. +#[doc(hidden)] +pub struct OffsetLoc<Base: ?Sized, T>(usize, PhantomData<(T, Base)>); - /// Returns the absolute location of this type when using `B` as its base. - #[inline(always)] - fn of<B: RegisterBase<Self::BaseFamily>>() -> RelativeRegisterLoc<Self, B> - where - Self: Register, - { - RelativeRegisterLoc::new() - } -} - -/// Trait implemented by relative registers. -pub trait RelativeRegister: Register + WithBase {} - -/// Location of a relative register. -/// -/// This can either be an immediately accessible regular [`RelativeRegister`], or a -/// [`RelativeRegisterArray`] that needs one additional resolution through -/// [`RelativeRegisterLoc::at`]. -pub struct RelativeRegisterLoc<T: WithBase, B: ?Sized>(PhantomData<T>, PhantomData<B>); - -impl<T, B> RelativeRegisterLoc<T, B> -where - T: Register + WithBase, - B: RegisterBase<T::BaseFamily> + ?Sized, -{ - /// Returns the location of a relative register or register array. - #[inline(always)] - // We do not implement `Default` so we can be const. - #[expect(clippy::new_without_default)] - pub const fn new() -> Self { - Self(PhantomData, PhantomData) +impl<Base: ?Sized, T> OffsetLoc<Base, T> { + #[inline] + pub const fn new(offset: usize) -> Self { + Self(offset, PhantomData) } - // Returns the absolute offset of the relative register using base `B`. - // - // This is implemented as a private const method so it can be reused by the [`IoLoc`] - // implementations of both [`RelativeRegisterLoc`] and [`RelativeRegisterArrayLoc`]. #[inline] - const fn offset(self) -> usize { - B::BASE + T::OFFSET + pub const fn const_offset(self) -> usize { + self.0 } } -impl<const SIZE: usize, T, B> IoLoc<Region<SIZE>, T> for RelativeRegisterLoc<T, B> -where - T: RelativeRegister, - B: RegisterBase<T::BaseFamily> + ?Sized, -{ - type IoType = T::Storage; - +impl<Base: ?Sized, T> IoLoc<Base, T> for OffsetLoc<Base, T> { #[inline(always)] fn offset(self) -> usize { - RelativeRegisterLoc::offset(self) + self.0 } } /// Trait implemented by arrays of registers. -pub trait RegisterArray: Register { +pub trait RegisterArray: Sized { + /// Base type for this register. + type Base: ?Sized; + + /// Start offset of the register. + /// + /// The interpretation of this offset depends on the type of the register. + const OFFSET: usize; /// Number of elements in the registers array. const SIZE: usize; /// Number of bytes between the start of elements in the registers array. @@ -285,12 +194,10 @@ impl<T: RegisterArray> RegisterArrayLoc<T> { } } -impl<const SIZE: usize, T> IoLoc<Region<SIZE>, T> for RegisterArrayLoc<T> +impl<Base: ?Sized, T> IoLoc<Base, T> for RegisterArrayLoc<T> where - T: RegisterArray, + T: RegisterArray<Base = Base>, { - type IoType = T::Storage; - #[inline(always)] fn offset(self) -> usize { T::OFFSET + self.0 * T::STRIDE @@ -318,71 +225,15 @@ pub trait Array { } } -/// Trait implemented by arrays of relative registers. -pub trait RelativeRegisterArray: RegisterArray + WithBase {} - -/// Location of a relative array register. -pub struct RelativeRegisterArrayLoc< - T: RelativeRegisterArray, - B: RegisterBase<T::BaseFamily> + ?Sized, ->(RelativeRegisterLoc<T, B>, usize); - -impl<T, B> RelativeRegisterArrayLoc<T, B> -where - T: RelativeRegisterArray, - B: RegisterBase<T::BaseFamily> + ?Sized, -{ - /// Returns the location of register `T` from the base `B` at index `idx`, with build-time - /// validation. - #[inline(always)] - pub fn new(idx: usize) -> Self { - build_assert!(idx < T::SIZE); - - Self(RelativeRegisterLoc::new(), idx) - } - - /// Attempts to return the location of register `T` from the base `B` at index `idx`, with - /// runtime validation. - #[inline(always)] - pub fn try_new(idx: usize) -> Option<Self> { - if idx < T::SIZE { - Some(Self(RelativeRegisterLoc::new(), idx)) - } else { - None - } - } -} - -/// Methods exclusive to [`RelativeRegisterLoc`]s created with a [`RelativeRegisterArray`]. -impl<T, B> RelativeRegisterLoc<T, B> -where - T: RelativeRegisterArray, - B: RegisterBase<T::BaseFamily> + ?Sized, -{ - /// Returns the location of the register at position `idx`, with build-time validation. - #[inline(always)] - pub fn at(self, idx: usize) -> RelativeRegisterArrayLoc<T, B> { - RelativeRegisterArrayLoc::new(idx) - } - - /// Returns the location of the register at position `idx`, with runtime validation. - #[inline(always)] - pub fn try_at(self, idx: usize) -> Option<RelativeRegisterArrayLoc<T, B>> { - RelativeRegisterArrayLoc::try_new(idx) - } -} - -impl<const SIZE: usize, T, B> IoLoc<Region<SIZE>, T> for RelativeRegisterArrayLoc<T, B> -where - T: RelativeRegisterArray, - B: RegisterBase<T::BaseFamily> + ?Sized, -{ - type IoType = T::Storage; +/// Trait implemented by types that indicate there is a fixed I/O location for this given type. +/// +/// Implementors can be used with [`Io::write_reg`](super::Io::write_reg). +pub trait FixedIoLoc<Base: ?Sized>: Sized { + /// Type of [`FixedIoLoc::LOCATION`]. + type Location: IoLoc<Base, Self>; - #[inline(always)] - fn offset(self) -> usize { - self.0.offset() + self.1 * T::STRIDE - } + /// Location of this type within given base. + const LOCATION: Self::Location; } /// Trait implemented by items that contain both a register value and the absolute I/O location at @@ -390,8 +241,8 @@ where /// /// Implementors can be used with [`Io::write_reg`](super::Io::write_reg). pub trait LocatedRegister<Base: ?Sized> { - /// Register value to write. - type Value: Register; + /// Value to write. + type Value; /// Full location information at which to write the value. type Location: IoLoc<Base, Self::Value>; @@ -400,27 +251,38 @@ pub trait LocatedRegister<Base: ?Sized> { fn into_io_op(self) -> (Self::Location, Self::Value); } -impl<const SIZE: usize, T> LocatedRegister<Region<SIZE>> for T +impl<Base: ?Sized, T> LocatedRegister<Base> for T where - T: FixedRegister, + T: FixedIoLoc<Base>, { - type Location = FixedRegisterLoc<Self::Value>; + type Location = T::Location; type Value = T; #[inline(always)] - fn into_io_op(self) -> (FixedRegisterLoc<T>, T) { - (FixedRegisterLoc::new(), self) + fn into_io_op(self) -> (T::Location, T) { + (T::LOCATION, self) } } +/// Helper function for register element alias implementation. +/// +/// This is used to enforce base matching and provide bounds checking. +#[doc(hidden)] +#[inline(always)] // for const eval only +pub const fn element_alias_offset<Base: ?Sized, Alias: RegisterArray<Base = Base>>( + idx: usize, +) -> usize { + assert!(idx < Alias::SIZE); + Alias::OFFSET + idx * Alias::STRIDE +} + /// Defines a dedicated type for a register, including getter and setter methods for its fields and /// methods to read and write it from an [`Io`](kernel::io::Io) region. /// /// This documentation focuses on how to declare registers. See the [module-level /// documentation](mod@kernel::io::register) for examples of how to access them. /// -/// There are 4 possible kinds of registers: fixed offset registers, relative registers, arrays of -/// registers, and relative arrays of registers. +/// Registers can either be fixed offset registers or arrays of registers. /// /// ## Fixed offset registers /// @@ -444,11 +306,14 @@ where /// io::{ /// register, /// Io, +/// Region, /// }, /// }; -/// # use kernel::io::{Mmio, Region}; +/// # use kernel::io::Mmio; /// /// register! { +/// base: Region<0x1000>; +/// /// FIXED_REG(u32) @ 0x100 { /// 15:8 high_byte; /// 7:0 low_byte; @@ -479,9 +344,14 @@ where /// the context: /// /// ```no_run -/// use kernel::io::register; +/// use kernel::io::{ +/// register, +/// Region, +/// }; /// /// register! { +/// base: Region<0x1000>; +/// /// /// Scratch register. /// pub SCRATCH(u32) @ 0x00000200 { /// 31:0 value; @@ -497,113 +367,45 @@ where /// In this example, `SCRATCH_BOOT_STATUS` uses the same I/O address as `SCRATCH`, while providing /// its own `completed` field. /// -/// ## Relative registers -/// -/// Relative registers can be instantiated several times at a relative offset of a group of bases. -/// For instance, imagine the following I/O space: +/// If you do not wish to have a bitfield defined, you can also create a register using an existing +/// type. /// -/// ```text -/// +-----------------------------+ -/// | ... | -/// | | -/// 0x100--->+------------CPU0-------------+ -/// | | -/// 0x110--->+-----------------------------+ -/// | CPU_CTL | -/// +-----------------------------+ -/// | ... | -/// | | -/// | | -/// 0x200--->+------------CPU1-------------+ -/// | | -/// 0x210--->+-----------------------------+ -/// | CPU_CTL | -/// +-----------------------------+ -/// | ... | -/// +-----------------------------+ -/// ``` -/// -/// `CPU0` and `CPU1` both have a `CPU_CTL` register that starts at offset `0x10` of their I/O -/// space segment. Since both instances of `CPU_CTL` share the same layout, we don't want to define -/// them twice and would prefer a way to select which one to use from a single definition. -/// -/// This can be done using the `Base + Offset` syntax when specifying the register's address: -/// -/// ```ignore +/// ```no_run +/// # use kernel::io::*; /// register! { -/// pub RELATIVE_REG(u32) @ Base + 0x80 { -/// ... -/// } +/// base: Region<0x1000>; +/// +/// /// UART RX register. +/// pub UART_RX: u8 @ 0x100; /// } /// ``` /// -/// This creates a register with an offset of `0x80` from a given base. +/// In case there is a fixed register associated with a specific type in the base, you can apply +/// `#[unique]` attribute which enables `write_reg` shorthand. This is automatically applied to +/// bitfields instantiated via the `register!` macro. /// -/// `Base` is an arbitrary type (typically a ZST) to be used as a generic parameter of the -/// [`RegisterBase`] trait to provide the base as a constant, i.e. each type providing a base for -/// this register needs to implement `RegisterBase<Base>`. -/// -/// The location of relative registers can be built using the [`WithBase::of`] method to specify -/// its base. All relative registers implement [`WithBase`]. -/// -/// Here is the above layout translated into code: +/// This should only be used when types meaningfully represent a register. For example, in the +/// previous `UART_RX` example, even if only a single register is defined with `u8` type, it is a +/// bad idea to annotate it with `#[unique]`. /// /// ```no_run -/// use kernel::{ -/// io::{ -/// register, -/// register::{ -/// RegisterBase, -/// WithBase, -/// }, -/// Io, -/// }, -/// }; -/// # use kernel::io::{Mmio, Region}; -/// -/// // Type used to identify the base. -/// pub struct CpuCtlBase; -/// -/// // ZST describing `CPU0`. -/// struct Cpu0; -/// impl RegisterBase<CpuCtlBase> for Cpu0 { -/// const BASE: usize = 0x100; -/// } -/// -/// // ZST describing `CPU1`. -/// struct Cpu1; -/// impl RegisterBase<CpuCtlBase> for Cpu1 { -/// const BASE: usize = 0x200; -/// } +/// # use kernel::{bitfield, io::*}; /// -/// // This makes `CPU_CTL` accessible from all implementors of `RegisterBase<CpuCtlBase>`. -/// register! { -/// /// CPU core control. -/// pub CPU_CTL(u32) @ CpuCtlBase + 0x10 { -/// 0:0 start; +/// bitfield! { +/// pub struct Reset(u32) { +/// 0:0 reset; /// } /// } /// -/// # fn test(io: Mmio<'_, Region<0x1000>>) { -/// // Read the status of `Cpu0`. -/// let cpu0_started = io.read(CPU_CTL::of::<Cpu0>()); -/// -/// // Stop `Cpu0`. -/// io.write(WithBase::of::<Cpu0>(), CPU_CTL::zeroed()); -/// # } -/// -/// // Aliases can also be defined for relative register. /// register! { -/// /// Alias to CPU core control. -/// pub CPU_CTL_ALIAS(u32) => CpuCtlBase + CPU_CTL { -/// /// Start the aliased CPU core. -/// 1:1 alias_start; -/// } +/// base: Region<0x1000>; +/// +/// pub RESET: #[unique] Reset @ 0x100; /// } /// -/// # fn test2(io: Mmio<'_, Region<0x1000>>) { -/// // Start the aliased `CPU0`, leaving its other fields untouched. -/// io.update(CPU_CTL_ALIAS::of::<Cpu0>(), |r| r.with_alias_start(true)); +/// # fn test(mmio: Mmio<'_, Region<0x1000>>) { +/// // let mmio: Mmio<'_, Region<0x1000>>; +/// mmio.write_reg(Reset::zeroed().with_const_reset::<1>()); /// # } /// ``` /// @@ -636,15 +438,18 @@ where /// register, /// register::Array, /// Io, +/// Region, /// }, /// }; -/// # use kernel::io::{Mmio, Region}; +/// # use kernel::io::Mmio; /// # fn get_scratch_idx() -> usize { /// # 0x15 /// # } /// /// // Array of 64 consecutive registers with the same layout starting at offset `0x80`. /// register! { +/// base: Region<0x1000>; +/// /// /// Scratch registers. /// pub SCRATCH(u32)[64] @ 0x00000080 { /// 31:0 value; @@ -670,6 +475,8 @@ where /// // Alias to a specific register in an array. /// // Here `SCRATCH[8]` is used to convey the firmware exit code. /// register! { +/// base: Region<0x1000>; +/// /// /// Firmware exit status code. /// pub FIRMWARE_STATUS(u32) => SCRATCH[8] { /// 7:0 status; @@ -682,6 +489,8 @@ where /// // Here, each of the 16 registers of the array is separated by 8 bytes, meaning that the /// // registers of the two declarations below are interleaved. /// register! { +/// base: Region<0x1000>; +/// /// /// Scratch registers bank 0. /// pub SCRATCH_INTERLEAVED_0(u32)[16, stride = 8] @ 0x000000c0 { /// 31:0 value; @@ -696,332 +505,88 @@ where /// # } /// ``` /// -/// ## Relative arrays of registers +/// ## Relative registers /// -/// Combining the two features described in the sections above, arrays of registers accessible from -/// a base can also be defined: +/// There are cases where a register region is subdivided into small subregions, and you may wish to +/// have your register definition be relative to these subregions. This may be needed, for example, +/// if these subregions are instantiated several times, or you just want it for encapsulation +/// purpose. /// -/// ```ignore -/// register! { -/// pub RELATIVE_REGISTER_ARRAY(u8)[10, stride = 4] @ Base + 0x100 { -/// ... -/// } -/// } +/// For instance, imagine the following I/O space: +/// +/// ```text +/// +-----------------------------+ +/// | ... | +/// | | +/// 0x100--->+------------CPU0-------------+ +/// | | +/// 0x110--->+-----------------------------+ +/// | CPU_CTL | +/// +-----------------------------+ +/// | ... | +/// | | +/// | | +/// 0x200--->+------------CPU1-------------+ +/// | | +/// 0x210--->+-----------------------------+ +/// | CPU_CTL | +/// +-----------------------------+ +/// | ... | +/// +-----------------------------+ /// ``` /// -/// Like relative registers, they implement the [`WithBase`] trait. However the return value of -/// [`WithBase::of`] cannot be used directly as a location and must be further specified using the -/// [`at`](RelativeRegisterLoc::at) method. +/// `CPU0` and `CPU1` both have a `CPU_CTL` register that starts at offset `0x10` of their I/O +/// space segment. Since both instances of `CPU_CTL` share the same layout, we don't want to define +/// them twice and would prefer a way to select which one to use from a single definition. +/// +/// This can be done by defining a new type for the subregion, and then defining registers that use +/// the new type as the base: /// /// ```no_run /// use kernel::{ /// io::{ +/// io_project, /// register, -/// register::{ -/// RegisterBase, -/// WithBase, -/// }, /// Io, +/// Region, /// }, /// }; -/// # use kernel::io::{Mmio, Region}; -/// # fn get_scratch_idx() -> usize { -/// # 0x15 -/// # } +/// # use kernel::io::Mmio; /// -/// // Type used as parameter of `RegisterBase` to specify the base. -/// pub struct CpuCtlBase; -/// -/// // ZST describing `CPU0`. -/// struct Cpu0; -/// impl RegisterBase<CpuCtlBase> for Cpu0 { -/// const BASE: usize = 0x100; -/// } -/// -/// // ZST describing `CPU1`. -/// struct Cpu1; -/// impl RegisterBase<CpuCtlBase> for Cpu1 { -/// const BASE: usize = 0x200; -/// } +/// // Subregion type. Make sure it has adequate size and alignment. +/// #[repr(align(4))] +/// #[derive(FromBytes, IntoBytes)] +/// pub struct CpuCtl([u8; 0x100]); /// -/// // 64 per-cpu scratch registers, arranged as a contiguous array. /// register! { -/// /// Per-CPU scratch registers. -/// pub CPU_SCRATCH(u32)[64] @ CpuCtlBase + 0x00000080 { -/// 31:0 value; -/// } -/// } -/// -/// # fn test(io: Mmio<'_, Region<0x1000>>) -> Result<(), Error> { -/// // Read scratch register 0 of CPU0. -/// let scratch = io.read(CPU_SCRATCH::of::<Cpu0>().at(0)); +/// base: Region<0x1000>; /// -/// // Write the retrieved value into scratch register 15 of CPU1. -/// io.write(WithBase::of::<Cpu1>().at(15), scratch); -/// -/// // This won't build. -/// // let cpu0_scratch_128 = io.read(CPU_SCRATCH::of::<Cpu0>().at(128)).value(); -/// -/// // Runtime-obtained array index. -/// let scratch_idx = get_scratch_idx(); -/// // Access on a runtime index returns an error if it is out-of-bounds. -/// let cpu0_scratch = io.read( -/// CPU_SCRATCH::of::<Cpu0>().try_at(scratch_idx).ok_or(EINVAL)? -/// ).value(); -/// # Ok(()) -/// # } -/// -/// // Alias to `SCRATCH[8]` used to convey the firmware exit code. -/// register! { -/// /// Per-CPU firmware exit status code. -/// pub CPU_FIRMWARE_STATUS(u32) => CpuCtlBase + CPU_SCRATCH[8] { -/// 7:0 status; -/// } +/// // Subregions can just be defined like normal registers. +/// CPU0: CpuCtl @ 0x100; +/// CPU1: CpuCtl @ 0x200; /// } /// -/// // Non-contiguous relative register arrays can be defined by adding a stride parameter. -/// // Here, each of the 16 registers of the array is separated by 8 bytes, meaning that the -/// // registers of the two declarations below are interleaved. +/// // Then you can define new registers on the subregion. /// register! { -/// /// Scratch registers bank 0. -/// pub CPU_SCRATCH_INTERLEAVED_0(u32)[16, stride = 8] @ CpuCtlBase + 0x00000d00 { -/// 31:0 value; -/// } +/// base: CpuCtl; /// -/// /// Scratch registers bank 1. -/// pub CPU_SCRATCH_INTERLEAVED_1(u32)[16, stride = 8] @ CpuCtlBase + 0x00000d04 { -/// 31:0 value; +/// /// CPU core control. +/// pub CPU_CTL(u32) @ 0x10 { +/// 0:0 start; /// } /// } /// -/// # fn test2(io: Mmio<'_, Region<0x1000>>) -> Result<(), Error> { -/// let cpu0_status = io.read(CPU_FIRMWARE_STATUS::of::<Cpu0>()).status(); -/// # Ok(()) +/// # fn test(io: Mmio<'_, Region<0x1000>>) { +/// // Read the status of `Cpu0`. +/// let cpu0_started = io_project!(io, build: CPU0).read(CPU_CTL); +/// +/// // Stop `Cpu0`. +/// io_project!(io, build: CPU0).write_reg(CPU_CTL::zeroed()); /// # } /// ``` #[macro_export] macro_rules! register { - // Entry point for the macro, allowing multiple registers to be defined in one call. - // It matches all possible register declaration patterns to dispatch them to corresponding - // `@reg` rule that defines a single register. - // - // TODO: change `alias:ident` to `alias:path` once relative registers are replaced by I/O - // projections. - ( - $( - $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) - $([ $size:expr $(, stride = $stride:expr)? ])? - $(@ $($base:ident +)? $offset:literal)? - $(=> $alias:ident $(+ $alias_offset:ident)? $([$alias_idx:expr])? )? - { $($fields:tt)* } - )* - ) => { - $( - $crate::register!( - @reg $(#[$attr])* $vis $name ($storage) $([$size $(, stride = $stride)?])? - $(@ $($base +)? $offset)? - $(=> $alias $(+ $alias_offset)? $([$alias_idx])? )? - { $($fields)* } - ); - )* - }; - - // All the rules below are private helpers. - - // Creates a register at a fixed offset of the MMIO space. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) @ $offset:literal - { $($fields:tt)* } - ) => { - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!(@io_base $name($storage) @ $offset); - $crate::register!(@io_fixed $(#[$attr])* $vis $name); - }; - - // Creates an alias register of fixed offset register `alias` with its own fields. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) => $alias:path - { $($fields:tt)* } - ) => { - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!( - @io_base $name($storage) @ - <$alias as $crate::io::register::Register>::OFFSET - ); - $crate::register!(@io_fixed $(#[$attr])* $vis $name); - }; - - // Creates a register at a relative offset from a base address provider. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) @ $base:ident + $offset:literal - { $($fields:tt)* } - ) => { - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!(@io_base $name($storage) @ $offset); - $crate::register!(@io_relative $name @ $base); - }; - - // Creates an alias register of relative offset register `alias` with its own fields. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) => $base:ident + $alias:ident - { $($fields:tt)* } - ) => { - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!( - @io_base $name($storage) @ <$alias as $crate::io::register::Register>::OFFSET - ); - $crate::register!(@io_relative $name @ $base); - }; - - // Creates an array of registers at a fixed offset of the MMIO space. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) - [ $size:expr, stride = $stride:expr ] @ $offset:literal { $($fields:tt)* } - ) => { - $crate::build_assert::static_assert!(::core::mem::size_of::<$storage>() <= $stride); - - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!(@io_base $name($storage) @ $offset); - $crate::register!(@io_array $name [ $size, stride = $stride ]); - }; - - // Shortcut for contiguous array of registers (stride == size of element). - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) [ $size:expr ] @ $offset:literal - { $($fields:tt)* } - ) => { - $crate::register!( - @reg $(#[$attr])* $vis $name($storage) - [ $size, stride = ::core::mem::size_of::<$storage>() ] - @ $offset { $($fields)* } - ); - }; - - // Creates an alias of register `idx` of array of registers `alias` with its own fields. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) => $alias:path [ $idx:expr ] - { $($fields:tt)* } - ) => { - $crate::build_assert::static_assert!( - $idx < <$alias as $crate::io::register::RegisterArray>::SIZE - ); - - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!( - @io_base $name($storage) @ - <$alias as $crate::io::register::Register>::OFFSET - + $idx * <$alias as $crate::io::register::RegisterArray>::STRIDE - ); - $crate::register!(@io_fixed $(#[$attr])* $vis $name); - }; - - // Creates an array of registers at a relative offset from a base address provider. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) - [ $size:expr, stride = $stride:expr ] - @ $base:ident + $offset:literal { $($fields:tt)* } - ) => { - $crate::build_assert::static_assert!(::core::mem::size_of::<$storage>() <= $stride); - - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!(@io_base $name($storage) @ $offset); - $crate::register!(@io_relative_array $name [ $size, stride = $stride ] @ $base); - }; - - // Shortcut for contiguous array of relative registers (stride == size of element). - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) [ $size:expr ] - @ $base:ident + $offset:literal { $($fields:tt)* } - ) => { - $crate::register!( - @reg $(#[$attr])* $vis $name($storage) - [ $size, stride = ::core::mem::size_of::<$storage>() ] - @ $base + $offset { $($fields)* } - ); - }; - - // Creates an alias of register `idx` of relative array of registers `alias` with its own - // fields. - ( - @reg $(#[$attr:meta])* $vis:vis $name:ident ($storage:ty) - => $base:ident + $alias:ident [ $idx:expr ] { $($fields:tt)* } - ) => { - $crate::build_assert::static_assert!( - $idx < <$alias as $crate::io::register::RegisterArray>::SIZE - ); - - $crate::register!(@bitfield $(#[$attr])* $vis struct $name($storage) { $($fields)* }); - $crate::register!( - @io_base $name($storage) @ - <$alias as $crate::io::register::Register>::OFFSET + - $idx * <$alias as $crate::io::register::RegisterArray>::STRIDE - ); - $crate::register!(@io_relative $name @ $base); - }; - - // Generates the bitfield for the register. - // - // `#[allow(non_camel_case_types)]` is added since register names typically use - // `SCREAMING_CASE`. - ( - @bitfield $(#[$attr:meta])* $vis:vis struct $name:ident($storage:ty) { $($fields:tt)* } - ) => { - $crate::bitfield!( - #[allow(non_camel_case_types)] - $(#[$attr])* $vis struct $name($storage) { $($fields)* } - ); - }; - - // Implementations shared by all registers types. - (@io_base $name:ident($storage:ty) @ $offset:expr) => { - impl $crate::io::register::Register for $name { - type Storage = $storage; - - const OFFSET: usize = $offset; - } - }; - - // Implementations of fixed registers. - (@io_fixed $(#[$attr:meta])* $vis:vis $name:ident) => { - impl $crate::io::register::FixedRegister for $name {} - - $(#[$attr])* - $vis const $name: $crate::io::register::FixedRegisterLoc<$name> = - $crate::io::register::FixedRegisterLoc::<$name>::new(); - }; - - // Implementations of relative registers. - (@io_relative $name:ident @ $base:ident) => { - impl $crate::io::register::WithBase for $name { - type BaseFamily = $base; - } - - impl $crate::io::register::RelativeRegister for $name {} - }; - - // Implementations of register arrays. - (@io_array $name:ident [ $size:expr, stride = $stride:expr ]) => { - impl $crate::io::register::Array for $name {} - - impl $crate::io::register::RegisterArray for $name { - const SIZE: usize = $size; - const STRIDE: usize = $stride; - } - }; - - // Implementations of relative array registers. - ( - @io_relative_array $name:ident [ $size:expr, stride = $stride:expr ] @ $base:ident - ) => { - impl $crate::io::register::WithBase for $name { - type BaseFamily = $base; - } - - impl $crate::io::register::RegisterArray for $name { - const SIZE: usize = $size; - const STRIDE: usize = $stride; - } - - impl $crate::io::register::RelativeRegisterArray for $name {} + ($($tt:tt)*) => { + $crate::macros::register!($($tt)*); }; } diff --git a/rust/kernel/io/resource.rs b/rust/kernel/io/resource.rs index 17b0c174cfc5..0d3b34f83334 100644 --- a/rust/kernel/io/resource.rs +++ b/rust/kernel/io/resource.rs @@ -226,10 +226,18 @@ impl Flags { /// Resource represents a memory region that must be ioremaped using `ioremap_np`. pub const IORESOURCE_MEM_NONPOSTED: Flags = Flags::new(bindings::IORESOURCE_MEM_NONPOSTED); + /// Memory region uses a 64-bit address (consumes two consecutive PCI resource slots). + pub const IORESOURCE_MEM_64: Flags = Flags::new(bindings::IORESOURCE_MEM_64); + // Always inline to optimize out error path of `build_assert`. #[inline(always)] const fn new(value: u32) -> Self { build_assert!(value as u64 <= c_ulong::MAX as u64); Flags(value as c_ulong) } + + /// Wrap a raw `c_ulong` value returned by a C API into [`Flags`]. + pub(crate) const fn from_raw(value: c_ulong) -> Self { + Flags(value) + } } diff --git a/rust/kernel/lib.rs b/rust/kernel/lib.rs index 4d5c96ddc49c..f9ef36217bb5 100644 --- a/rust/kernel/lib.rs +++ b/rust/kernel/lib.rs @@ -67,6 +67,8 @@ pub mod device; pub mod device_id; pub mod devres; pub mod dma; +#[cfg(CONFIG_DMA_SHARED_BUFFER)] +pub mod dma_buf; pub mod driver; #[cfg(CONFIG_DRM = "y")] pub mod drm; @@ -98,6 +100,7 @@ pub mod jump_label; pub mod kunit; pub mod list; pub mod maple_tree; +pub mod mem; pub mod miscdevice; pub mod mm; pub mod module; diff --git a/rust/kernel/maple_tree.rs b/rust/kernel/maple_tree.rs index 265d6396a78a..7abe41228cb7 100644 --- a/rust/kernel/maple_tree.rs +++ b/rust/kernel/maple_tree.rs @@ -16,7 +16,11 @@ use kernel::{ alloc::Flags, error::to_result, prelude::*, - types::{ForeignOwnable, Opaque}, + types::{ + ForeignOwnable, + NotThreadSafe, + Opaque, // + }, }; /// A maple tree optimized for storing non-overlapping ranges. @@ -240,7 +244,10 @@ impl<T: ForeignOwnable> MapleTree<T> { unsafe { bindings::spin_lock(self.ma_lock()) }; // INVARIANT: We just took the spinlock. - MapleGuard(self) + MapleGuard { + tree: self, + _not_send: NotThreadSafe, + } } #[inline] @@ -302,19 +309,30 @@ impl<T: ForeignOwnable> PinnedDrop for MapleTree<T> { } } +// SAFETY: `MapleTree<T>` is `Send` if `T` is `Send` because `MapleTree` owns its elements. +unsafe impl<T: ForeignOwnable + Send> Send for MapleTree<T> {} + +// SAFETY: `&MapleTree<T>` allows inserting and erasing entries from any thread, so `T: Send` is +// required, and shared borrows of entries require `T: Sync`. +unsafe impl<T: ForeignOwnable + Send + Sync> Sync for MapleTree<T> {} + /// A reference to a [`MapleTree`] that owns the inner lock. /// /// # Invariants /// /// This guard owns the inner spinlock. #[must_use = "if unused, the lock will be immediately unlocked"] -pub struct MapleGuard<'tree, T: ForeignOwnable>(&'tree MapleTree<T>); +pub struct MapleGuard<'tree, T: ForeignOwnable> { + tree: &'tree MapleTree<T>, + // A held spinlock must be released on the same CPU that acquired it. + _not_send: NotThreadSafe, +} impl<'tree, T: ForeignOwnable> Drop for MapleGuard<'tree, T> { #[inline] fn drop(&mut self) { // SAFETY: By the type invariants, we hold this spinlock. - unsafe { bindings::spin_unlock(self.0.ma_lock()) }; + unsafe { bindings::spin_unlock(self.tree.ma_lock()) }; } } @@ -323,7 +341,7 @@ impl<'tree, T: ForeignOwnable> MapleGuard<'tree, T> { pub fn ma_state(&mut self, first: usize, end: usize) -> MaState<'_, T> { // SAFETY: The `MaState` borrows this `MapleGuard`, so it can also borrow the `MapleGuard`s // read/write permissions to the maple tree. - unsafe { MaState::new_raw(self.0, first, end) } + unsafe { MaState::new_raw(self.tree, first, end) } } /// Load the value at the given index. @@ -375,7 +393,7 @@ impl<'tree, T: ForeignOwnable> MapleGuard<'tree, T> { #[inline] pub fn load(&mut self, index: usize) -> Option<T::BorrowedMut<'_>> { // SAFETY: `self.tree` contains a valid maple tree. - let ret = unsafe { bindings::mtree_load(self.0.tree.get(), index) }; + let ret = unsafe { bindings::mtree_load(self.tree.tree.get(), index) }; if ret.is_null() { return None; } diff --git a/rust/kernel/mem.rs b/rust/kernel/mem.rs new file mode 100644 index 000000000000..f2d4cdf87d00 --- /dev/null +++ b/rust/kernel/mem.rs @@ -0,0 +1,234 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! Basic utilities for dealing with memory, values, and types. + +use crate::prelude::*; + +/// Transmute between two types. +/// +/// Use this instead of [`core::mem::transmute`] when it is known that sizes are identical but this +/// cannot be proven by the compiler. +/// +/// This is equivalent to Rust's `transmute_unchecked` intrinsics. +/// +/// # Safety +/// +/// All safety requirements of [`core::mem::transmute`] apply, plus that the size `Src` and `Dst` +/// must match. +/// +/// # Examples +/// +/// This can be used when types are known to have the same size, but only at runtime. +/// +/// ```no_run +/// # use core::any::TypeId; +/// fn to_u32<T: 'static>(v: T) -> Option<u32> { +/// if TypeId::of::<T>() != TypeId::of::<u32>() { +/// return None; +/// } +/// +/// // `core::mem::transmute` won't work here. +/// // SAFETY: We've checked that `T` is `u32`! +/// Some(unsafe { kernel::mem::transmute_unchecked(v) }) +/// } +/// +/// to_u32(1u32); +/// ``` +#[inline(always)] +pub const unsafe fn transmute_unchecked<Src, Dst>(val: Src) -> Dst { + // SAFETY: This is identical to `transmute` except that we bypassed the size check; which is + // true per safety requirement. + unsafe { core::mem::transmute_copy(&core::mem::ManuallyDrop::new(val)) } +} + +/// Version of `transmute` that performs size check at monomorphization-time. +/// +/// Use this instead of [`core::mem::transmute`] when it is known that sizes are identical but this +/// cannot be proven by the compiler during type checking and can be proven during monomorphization. +/// +/// The signature is equivalent to Rust standard library's unstable `transmute_neo` and that of +/// [RFC 3844](https://github.com/rust-lang/rfcs/pull/3844). +/// +/// # Safety +/// +/// Same as [`core::mem::transmute`]. +/// +/// # Examples +/// +/// This is typically used in generic code where it's known that type will have the same size, but +/// the compiler cannot prove it generically. +/// +/// ```no_run +/// trait IsU32 {} +/// impl IsU32 for u32 {} +/// +/// fn to_u32<T: IsU32>(v: T) -> u32 { +/// // `core::mem::transmute` won't work here. +/// // SAFETY: We know that `v` is u32! +/// unsafe { kernel::mem::transmute(v) } +/// } +/// +/// to_u32(1u32); +/// ``` +#[inline(always)] +pub const unsafe fn transmute<Src, Dst>(val: Src) -> Dst { + const_assert!(size_of::<Src>() == size_of::<Dst>()); + + // SAFETY: Size is checked above. Other safety requirements follow those of the function. + unsafe { transmute_unchecked(val) } +} + +/// Safely transmutes a value of one type to a value of another type of the same size. +/// +/// The sizes are checked during monomorphization. +/// +/// This can be considered as generic version of [`zerocopy::transmute!`] macro that defers the size +/// check and thus can be used in more cases. +/// +/// # Examples +/// +/// ```no_run +/// fn to_u32<T: FromBytes + IntoBytes>(v: T) -> u32 { +/// // `zerocopy::transmute!` won't work here. +/// kernel::mem::safe_transmute(v) +/// } +/// +/// to_u32(1i32); +/// ``` +#[inline(always)] +pub const fn safe_transmute<Src: IntoBytes, Dst: FromBytes>(val: Src) -> Dst { + // SAFETY: `transmute` is safe with `IntoBytes` and `FromBytes` bounds. + unsafe { transmute(val) } +} + +/// Type that is layout-compatible with a primitive representation. +/// +/// # Safety +/// +/// - [`Self`] must have the same size and alignment as [`Self::Repr`]. +/// - [`Self`] must be [transmutable] to [`Self::Repr`]. +/// - Neither [`Self`] nor [`Self::Repr`] contains interior mutability. +/// +/// The above basically says that `&Self` can be transmuted to `&Self::Repr`. +/// +/// [transmutable]: core::mem::transmute +pub unsafe trait AsRepr: Sized { + /// Primitive representation of this type. + type Repr; + + /// Convert from [`&Self`](Self) to [`&Self::Repr`](AsRepr::Repr). + #[inline(always)] + fn as_repr(this: &Self) -> &Self::Repr { + // SAFETY: Per safety requirement of the trait. + unsafe { core::mem::transmute(this) } + } + + /// Convert from [`Self`] to [`Self::Repr`]. + #[inline(always)] + fn into_repr(this: Self) -> Self::Repr { + // SAFETY: Per safety requirement of the trait. + unsafe { transmute(this) } + } + + /// Convert from [`Self::Repr`] to [`Self`]. + /// + /// # Safety + /// + /// `repr` must be a valid bit pattern of [`Self`] and satisfy type-specific invariants of it. + /// + /// Alternatively, if `repr` is previously obtained using [`Self::into_repr`], and each + /// `from_repr_unchecked` should correspond to a unique `into_repr` call, then it is safe to + /// call as well (this means that we're undoing a `into_repr` call getting the exact bytes + /// back). + /// + /// No guarantee is made if the result of a `into_repr` is passed to multiple + /// `from_repr_unchecked` (i.e. copies are made), to allow for cases where `Repr` is a pointer + /// and the user of the API wants ownership transfer. Users that want the ability to call + /// `from_repr_unchecked` after copying can require `Copy` bound explicitly. + #[inline(always)] + unsafe fn from_repr_unchecked(repr: Self::Repr) -> Self { + // SAFETY: Per safety requirement, `repr` is valid repr of `Self`, or it is previously from + // `into_repr`, in which case we're undoing the transmute so it is also safe. + unsafe { transmute(repr) } + } +} + +/// Type that is bi-directionally transmutable with a primitive representation. +/// +/// # Safety +/// +/// - [`Self`] must be [transmutable] from [`Self::Repr`]. +/// +/// [transmutable]: core::mem::transmute +/// [`Self::Repr`]: AsRepr::Repr +pub unsafe trait AsReprMut: AsRepr { + /// Convert from `&mut Self` to [`&mut Self::Repr`](AsRepr::Repr). + #[inline(always)] + fn as_repr_mut(this: &mut Self) -> &mut Self::Repr { + // SAFETY: Per safety requirement of the trait. + unsafe { core::mem::transmute(this) } + } + + /// Convert from [`Self::Repr`](AsRepr::Repr) to `Self`. + #[inline(always)] + fn from_repr(repr: Self::Repr) -> Self { + // SAFETY: Per safety requirement of the trait. + unsafe { transmute(repr) } + } +} + +// SAFETY: `bool` has the same size and alignment as `u8`, and Rust guarantees that `bool` has +// only two valid bit patterns: 0 (`false`) and 1 (`true`). Thus `bool` can be transmuted to `u8`. +// Neither types contain interior mutability. +unsafe impl AsRepr for bool { + type Repr = u8; +} + +// SAFETY: `*mut T` has the same size and alignment with `*const c_void`, and thus `*mut T` is +// transmutable to `*const c_void`. Neither types contain interior mutability. +unsafe impl<T> AsRepr for *mut T { + type Repr = *const c_void; +} + +// SAFETY: `*mut T` is transmutable from `*const c_void`. +unsafe impl<T> AsReprMut for *mut T {} + +// SAFETY: `*const T` has the same size and alignment with `*const c_void`, and is transmutable to +// `*const c_void`. Neither types contain interior mutability. +unsafe impl<T> AsRepr for *const T { + type Repr = *const c_void; +} + +// SAFETY: `*const T` is transmutable from `*const c_void`. +unsafe impl<T> AsReprMut for *const T {} + +macro_rules! int_impl { + ($($unsigned:ident $signed:ident ,)*) => {$( + // SAFETY: `$unsigned` has the same size and alignment with itself, and is transmutable to + // itself. It does not contain interior mutability. + unsafe impl AsRepr for $unsigned { + type Repr = $unsigned; + } + + // SAFETY: `$unsigned` is transmutable from itself. + unsafe impl AsReprMut for $unsigned {} + + // SAFETY: `$signed` has the same size and alignment with `$unsigned`, and is transmutable + // to it Neither types contain interior mutability. + unsafe impl AsRepr for $signed { + type Repr = $unsigned; + } + + // SAFETY: `$signed` is transmutable from `$unsigned`. + unsafe impl AsReprMut for $signed {} + )*}; +} + +int_impl! { + u8 i8, + u16 i16, + u32 i32, + u64 i64, + // `usize` is not normalized to particular integer for portability. + usize isize, +} diff --git a/rust/kernel/pci.rs b/rust/kernel/pci.rs index 3ec897709e89..19a219847c17 100644 --- a/rust/kernel/pci.rs +++ b/rust/kernel/pci.rs @@ -17,6 +17,7 @@ use crate::{ from_result, to_result, // }, + io::resource, prelude::*, str::CStr, types::Opaque, @@ -439,6 +440,19 @@ impl Device { Ok(unsafe { bindings::pci_resource_len(self.as_raw(), bar.try_into()?) }) } + /// Returns the resource flags (`IORESOURCE_*`) of the given PCI BAR. + pub fn resource_flags(&self, bar: u32) -> Result<resource::Flags> { + if !Bar::index_is_valid(bar) { + return Err(EINVAL); + } + + // SAFETY: + // - `bar` is a valid bar number, as guaranteed by the above call to `Bar::index_is_valid`, + // - by its type invariant `self.as_raw` is always a valid pointer to a `struct pci_dev`. + let raw = unsafe { bindings::pci_resource_flags(self.as_raw(), bar.try_into()?) }; + Ok(resource::Flags::from_raw(raw)) + } + /// Returns the PCI class as a `Class` struct. #[inline] pub fn pci_class(&self) -> Class { diff --git a/rust/kernel/sync/atomic.rs b/rust/kernel/sync/atomic.rs index 9cd009d57e35..6d27898add42 100644 --- a/rust/kernel/sync/atomic.rs +++ b/rust/kernel/sync/atomic.rs @@ -140,7 +140,7 @@ pub unsafe trait AtomicAdd<Rhs = Self>: AtomicType { const fn into_repr<T: AtomicType>(v: T) -> T::Repr { // SAFETY: Per the safety requirement of `AtomicType`, `T` is round-trip transmutable to // `T::Repr`, therefore the transmute operation is sound. - unsafe { core::mem::transmute_copy(&v) } + unsafe { crate::mem::transmute(v) } } /// # Safety @@ -149,7 +149,7 @@ const fn into_repr<T: AtomicType>(v: T) -> T::Repr { #[inline(always)] const unsafe fn from_repr<T: AtomicType>(r: T::Repr) -> T { // SAFETY: Per the safety requirement of the function, the transmute operation is sound. - unsafe { core::mem::transmute_copy(&r) } + unsafe { crate::mem::transmute(r) } } impl<T: AtomicType> Atomic<T> { diff --git a/rust/kernel/uaccess.rs b/rust/kernel/uaccess.rs index 5f6c4d7a1a51..f09078228c53 100644 --- a/rust/kernel/uaccess.rs +++ b/rust/kernel/uaccess.rs @@ -520,14 +520,14 @@ impl UserSliceWriter { /// /// fn copy_dma_to_user( /// mut writer: UserSliceWriter, - /// alloc: &Coherent<[u8]>, + /// alloc: &Coherent<'_, [u8]>, /// ) -> Result { /// writer.write_dma(alloc, 0, 256) /// } /// ``` pub fn write_dma<T: KnownSize + AsBytes + ?Sized>( &mut self, - alloc: &Coherent<T>, + alloc: &Coherent<'_, T>, offset: usize, count: usize, ) -> Result { diff --git a/rust/macros/io/mod.rs b/rust/macros/io/mod.rs new file mode 100644 index 000000000000..87f7742f4619 --- /dev/null +++ b/rust/macros/io/mod.rs @@ -0,0 +1,3 @@ +// SPDX-License-Identifier: GPL-2.0 + +pub(crate) mod register; diff --git a/rust/macros/io/register.rs b/rust/macros/io/register.rs new file mode 100644 index 000000000000..420c3ad052b8 --- /dev/null +++ b/rust/macros/io/register.rs @@ -0,0 +1,296 @@ +// SPDX-License-Identifier: GPL-2.0 + +//! Documentation and usage example of the macro can be found at `rust/kernel/io/register.rs`. + +use proc_macro2::{ + Group, + Literal, + Span, + TokenStream, // +}; +use quote::{ + quote, + quote_spanned, // +}; +use syn::{ + bracketed, + parenthesized, + parse::Parse, + parse_quote, + spanned::Spanned, + token, + Attribute, + Error, + Expr, + Ident, + Path, + Result, + Token, + Type, + Visibility, // +}; + +mod kw { + syn::custom_keyword!(base); + syn::custom_keyword!(stride); +} + +/// Definition of a register array. +/// +/// Specify a size, and optionally a stride. Syntax is of form `[EXPR $(, stride = EXPR)?]`. +struct RegArrayDef { + size: Expr, + stride: Option<Expr>, +} + +/// Offset of a register. +/// +/// Can be either of form +/// * `@ offset` for fixed offset +/// * `=> alias` for alias of register `alias`. +/// * `=> alias[idx]` for alias of register array `alias`'s `idx`-th element. +enum RegOffset { + /// Register is located at fixed address. + Fixed { offset: Literal }, + /// Register is an alias of a fixed register. + Alias { alias: Path }, + /// Register is an alias of an element of a register array. + ElementAlias { alias: Path, idx: Expr }, +} + +/// Definition of a single register. +struct Reg { + attrs: Vec<Attribute>, + vis: Visibility, + name: Ident, + unique: bool, + ty: Type, + array: Option<RegArrayDef>, + offset: RegOffset, + bitfield: Option<(Type, Group)>, +} + +impl Parse for Reg { + fn parse(input: syn::parse::ParseStream<'_>) -> Result<Self> { + let attrs = input.call(Attribute::parse_outer)?; + let vis = input.parse()?; + let name = input.parse()?; + + let lh = input.lookahead1(); + let (unique, ty, bitfield_storage) = if lh.peek(Token![:]) { + let _: Token![:] = input.parse()?; + + let mut attrs = input.call(Attribute::parse_outer)?; + let unique = attrs + .extract_if(.., |attr| attr.path().is_ident("unique")) + .count() + != 0; + if !attrs.is_empty() { + Err(Error::new_spanned(&attrs[0], "unexpected attributes"))? + } + + (unique, input.parse()?, None) + } else if lh.peek(token::Paren) { + let content; + parenthesized!(content in input); + let bitfield_storage = Some(content.parse()?); + + // For bitfields, bitfield macro will generate a type with the same name as `name`. + (true, parse_quote!(#name), bitfield_storage) + } else { + Err(lh.error())? + }; + + let array = if input.peek(token::Bracket) { + let content; + bracketed!(content in input); + let size = content.parse()?; + let stride = if content.peek(Token![,]) { + let _: Token![,] = content.parse()?; + let _: kw::stride = content.parse()?; + let _: Token![=] = content.parse()?; + Some(content.parse()?) + } else { + None + }; + Some(RegArrayDef { size, stride }) + } else { + None + }; + + // Parse offset and the base it's relative to. + let lh = input.lookahead1(); + let offset = if lh.peek(Token![@]) { + let _: Token![@] = input.parse()?; + + RegOffset::Fixed { + offset: input.parse()?, + } + } else if lh.peek(Token![=>]) { + let _: Token![=>] = input.parse()?; + let alias: Path = input.parse()?; + + if input.peek(token::Bracket) { + let content; + bracketed!(content in input); + RegOffset::ElementAlias { + alias, + idx: content.parse()?, + } + } else { + RegOffset::Alias { alias } + } + } else { + Err(lh.error())? + }; + + let bitfield = if let Some(storage) = bitfield_storage { + let lh = input.lookahead1(); + let args = if lh.peek(token::Brace) { + input.parse()? + } else { + Err(lh.error())? + }; + Some((storage, args)) + } else { + let _: Token![;] = input.parse()?; + None + }; + + Ok(Self { + attrs, + vis, + name, + unique, + ty, + array, + offset, + bitfield, + }) + } +} + +pub(crate) struct RegDef { + base: Type, + regs: Vec<Reg>, +} + +impl Parse for RegDef { + fn parse(input: syn::parse::ParseStream<'_>) -> Result<Self> { + let _: kw::base = input.parse().map_err(|e| { + Error::new( + e.span(), + "a base type needs to be specified for `register!` invocation with `base: ty;`", + ) + })?; + + let _: Token![:] = input.parse()?; + let base = input.parse()?; + let _: Token![;] = input.parse()?; + + let mut regs = Vec::new(); + while !input.is_empty() { + regs.push(input.parse()?); + } + Ok(RegDef { base, regs }) + } +} + +pub(crate) fn register(def: RegDef) -> Result<TokenStream> { + let mut outputs = TokenStream::new(); + + let base = &def.base; + for reg in def.regs { + let Reg { + attrs, + vis, + name, + unique, + ty, + array, + offset, + bitfield, + } = reg; + + // Use register name's span for generated code, so error messages (if any) can point to it + // instead of the entire register allocation. + let span = name.span().resolved_at(Span::mixed_site()); + + let offset = match offset { + RegOffset::Fixed { offset } => quote!(#offset), + RegOffset::Alias { alias } => { + quote_spanned!(alias.span().resolved_at(span) => + ::kernel::io::register::OffsetLoc::<#base, _>::const_offset(#alias) + ) + } + RegOffset::ElementAlias { alias, idx } => { + quote_spanned!(alias.span().resolved_at(span) => + ::kernel::io::register::element_alias_offset::<#base, #alias>(#idx) + ) + } + }; + + if let Some((storage, args)) = &bitfield { + outputs.extend(quote_spanned!(span => + ::kernel::bitfield!( + // `#[allow(non_camel_case_types)]` is added since register names typically use + // `SCREAMING_CASE`. + #[allow(non_camel_case_types)] + #(#attrs)* #vis struct #name(#storage) #args + ); + )); + } + + match array { + None => { + if unique { + outputs.extend(quote!( + impl ::kernel::io::register::FixedIoLoc<#base> for #ty { + type Location = ::kernel::io::register::OffsetLoc<#base, #ty>; + const LOCATION: Self::Location = #name; + } + )) + } + + outputs.extend(quote_spanned!(span => + #(#attrs)* #vis const #name: ::kernel::io::register::OffsetLoc<#base, #ty> = + ::kernel::io::register::OffsetLoc::new(#offset); + )); + } + + Some(def) => { + if bitfield.is_none() { + Err(Error::new_spanned( + &ty, + "defining without bitfield is not yet supported for this type of register", + ))? + } + + let size = &def.size; + let stride = if let Some(stride) = &def.stride { + outputs.extend(quote_spanned!(stride.span().resolved_at(span) => + ::kernel::build_assert::static_assert!( + ::core::mem::size_of::<#ty>() <= #stride + ); + )); + quote!(#stride) + } else { + quote_spanned!(span => ::core::mem::size_of::<#ty>()) + }; + + outputs.extend(quote_spanned!(span => + impl ::kernel::io::register::Array for #name {} + + impl ::kernel::io::register::RegisterArray for #name { + type Base = #base; + const OFFSET: usize = #offset; + const SIZE: usize = #size; + const STRIDE: usize = #stride; + } + )); + } + }; + } + + Ok(outputs) +} diff --git a/rust/macros/lib.rs b/rust/macros/lib.rs index 24f96feaeb34..9b76efe1476f 100644 --- a/rust/macros/lib.rs +++ b/rust/macros/lib.rs @@ -19,6 +19,7 @@ mod export; mod fmt; mod for_lt; mod helpers; +mod io; mod kunit; mod module; mod paste; @@ -481,6 +482,14 @@ pub fn paste(input: TokenStream) -> TokenStream { .into() } +#[doc(hidden)] // Documented in `kernel` crate. +#[proc_macro] +pub fn register(input: TokenStream) -> TokenStream { + io::register::register(parse_macro_input!(input)) + .unwrap_or_else(|e| e.into_compile_error()) + .into() +} + /// Registers a KUnit test suite and its test cases using a user-space like syntax. /// /// This macro should be used on modules. If `CONFIG_KUNIT` (in `.config`) is `n`, the target module diff --git a/samples/rust/rust_dma.rs b/samples/rust/rust_dma.rs index bd60034ded23..ffb693544673 100644 --- a/samples/rust/rust_dma.rs +++ b/samples/rust/rust_dma.rs @@ -5,7 +5,10 @@ //! To make this driver probe, QEMU must be run with `-device pci-testdev`. use kernel::{ - device::Core, + device::{ + Bound, + Core, // + }, dma::{ Coherent, DataDirection, @@ -23,14 +26,15 @@ use kernel::{ scatterlist::{ Owned, SGTable, // - }, - sync::aref::ARef, // + }, // }; +struct DmaSampleDriver; + #[pin_data(PinnedDrop)] -struct DmaSampleDriver { - pdev: ARef<pci::Device>, - ca: Coherent<[MyStruct]>, +struct DmaSampleData<'bound> { + pdev: &'bound pci::Device<Bound>, + ca: Coherent<'bound, [MyStruct]>, #[pin] sgt: SGTable<Owned<VVec<u8>>>, } @@ -67,13 +71,13 @@ kernel::pci_device_table!( impl pci::Driver for DmaSampleDriver { type IdInfo = (); - type Data<'bound> = Self; + type Data<'bound> = DmaSampleData<'bound>; const ID_TABLE: pci::IdTable<Self::IdInfo> = &PCI_TABLE; fn probe<'bound>( pdev: &'bound pci::Device<Core<'_>>, _info: Option<&'bound Self::IdInfo>, - ) -> impl PinInit<Self, Error> + 'bound { + ) -> impl PinInit<Self::Data<'bound>, Error> + 'bound { pin_init::pin_init_scope(move || { dev_info!(pdev, "Probe DMA test driver.\n"); @@ -82,7 +86,7 @@ impl pci::Driver for DmaSampleDriver { // SAFETY: There are no concurrent calls to DMA allocation and mapping primitives. unsafe { pdev.dma_set_mask_and_coherent(mask)? }; - let ca: Coherent<[MyStruct]> = + let ca: Coherent<'_, [MyStruct]> = Coherent::zeroed_slice(pdev.as_ref(), TEST_VALUES.len(), GFP_KERNEL)?; for (i, value) in TEST_VALUES.into_iter().enumerate() { @@ -94,8 +98,8 @@ impl pci::Driver for DmaSampleDriver { let sgt = SGTable::new(pdev.as_ref(), pages, DataDirection::ToDevice, GFP_KERNEL); - Ok(try_pin_init!(Self { - pdev: pdev.into(), + Ok(try_pin_init!(Self::Data { + pdev, ca, sgt <- sgt, })) @@ -103,7 +107,7 @@ impl pci::Driver for DmaSampleDriver { } } -impl DmaSampleDriver { +impl DmaSampleData<'_> { fn check_dma(&self) { for (i, value) in TEST_VALUES.into_iter().enumerate() { let val0 = io_read!(self.ca, [panic: i].h); @@ -116,7 +120,7 @@ impl DmaSampleDriver { } #[pinned_drop] -impl PinnedDrop for DmaSampleDriver { +impl PinnedDrop for DmaSampleData<'_> { fn drop(self: Pin<&mut Self>) { dev_info!(self.pdev, "Unload DMA test driver.\n"); diff --git a/samples/rust/rust_driver_pci.rs b/samples/rust/rust_driver_pci.rs index 2282191e6292..13b035a95756 100644 --- a/samples/rust/rust_driver_pci.rs +++ b/samples/rust/rust_driver_pci.rs @@ -23,6 +23,8 @@ mod regs { use super::*; register! { + base: kernel::io::Region<END>; + pub(super) TEST(u8) @ 0x0 { 7:0 index => TestIndex; } @@ -102,6 +104,8 @@ impl SampleDriverData<'_> { // Some PCI configuration space registers. register! { + base: pci::Normal; + VENDOR_ID(u16) @ 0x0 { 15:0 vendor_id; } |
