mirror of
https://github.com/torvalds/linux.git
synced 2026-10-06 10:36:03 +02:00
perf: - export perf_allow_ APIs for xe udmabuf: - remove default size limit of 64MB rust: - i/o rework (signed tag from driver-core tree) - add registration guard and registration data - fix unbounded lifetimes in ioctl handler args - fix a drm_dev_register race - gem_shmem: add DmaResvGuard helper - gpuvm: require send/sync for driver data - implement send/sync for GpuVaAlloc and GpuVmBo - add SmContext lifetime - rename dma_handle to dma_address - change pci_sriov_get_totalvfs return to unsigned int core: - create drm_of_get_panel_orientation - send per-connector hotplug events - add thunderbolt UBHR tunneling support connector: - add color format property dmem: - introduce a peak file - accept one region per limit - add dmemcg support for eviction gpusvm: - reorg code to give drivers more flexibility atomic: - add create_state callback and helper - add documentation on atomic commit lifetime buddy: - add per-order free - add used block scoreboard - fix UAF - test buffer clearance on resume - add phys_addr->block helper gem: - drop DRIVER_GEM_GPUVA flag ttm: - be more aggressive allocating below protection limit sched: - add test suite for concurrent job submissions hdmi: - hook the color format property in helpers mipi-dsi: - add MIPI_DSI_MODE_DSC_ALL_SLICES_IN_PKT bridge: - add atomic create callbacks - drop atomic reset - display-connector: don't autoenable HPD IRQ - trigger initial HPD for DP - ti-sn65dsi83: remove NO_HFP and NO_HBP mode flags - analogix_dp: switch to DP link training helpers dp: - add support for DSC max delta BPP edid: - parse panel type from DisplayID 2.x Display Parameters sysfb: - improve panel, stride, framebuffer size validation panel: - implement ref counting for struct drm_panel - himax-hx83121a: add backlight regulator support - novatek-nt36672a: Inline panel init sequences - visionox-vtdr6130: enable DSC - novatek-nt37801: Use mipi_dsi_*_multi() functions - samsung-s6d16d0: Fix prepare error handling - support Novatek NT36536 plus DT bindings - sofef00: fix backlight updates - osd101t2587: use mipi_dsi_*_multi interface - panel-edp: adjust timing for AUO displays - panel-lvds: support Opto Logic SCX1001511GGC49 - panel-simple: support Kyocera tcg070wvlq - panel-edp: quirks - AUO B116XAT04.3, CMN N116BCP-EA2, CSW MNB601LS1-8 - BOE NV116WH2-M30, BOE NT116WHM-N21, BOE NV116FH1-M31 - BOE NV116FH1-M30, NV140FHM-N5B, TM156VDXP25 - BOE NE160QDM-NY1, MB116AS01 - new: - Samsung ATNA40HQ08-0, Anbernic TD4310 - Chipone ICNA35XX, Ilitek ILI9488 - Ilitek ILI7807S, Renesas R63419 - MNE001BS6-2, MNF601BS4-1, Sharp LQ120P1JX51 virtio: - add support for save/restore virtio_gpu_objects - abort vq wait on device removal amdgpu: - add color format DRM property - initial compute pipe reset support - add GFX 6-8 modifier support - initial DCN 6.0.0 support - dmemcg eviction support - improved boundary checking for bios parsing - RAS updates and rework - VCN secure submission fixes - 8K panel fix - Display KUNIT tests - parse panel type from DisplayID - Align IP discovery to pci device lifetime - SOC15 register macro cleanups - UVD memory placement fixes - GFX9 mode2 reset fixes - drop unnecessary BUG/BUG_ON - GFX8 soft reset rework - enable soft reset on GFX8 - PSP/SMU 15.0.9 update - VI ASPM fix - userq fixes - amdgpu_vm_get_task_info_pasid lifetime fix - DC CACP support - change system_unbound_wq with system_dfl_wq - Loosen VFCT bios parsing to deal with pci=realloc - SI/SMU7 AC/DC switch fix - VM fence handling fix - GEM close optimisation - Apple Studio Display fixes - DC FRL fixes amdkfd: - initial compute pipe reset support - allow applications to opt out of sigbus on fatal errors - improve CRIU boundary checks - MQD handling rework - move TBA/TMA from system to device memory - avoid topology-lock in kfd_mmap - SVM eviction fixes radeon: - fix unset CONFIG_ACPI build i915: - Novalake (NVL display version 35) timing generator enabling - NVL DC3CO enabling - enable UBHR link rates on thunderbolt tunnels - Reduce Xe3+ PM demand peak bandwidth - enable pipe DMC error interrupts for display 30+ - add kunit tests for DP link config selection - refactor and document DP link recovery - i915/xe driver display probe/remove/suspend/resume/shutdown cleanup and unification - i915/xe display runtime PM unified - Break i915 and xe panic dependency on struct intel_framebuffer - Streamline Pre/Post-CSC LUT loops - drop TGL DC3DO support - CDCLK santization - fix HDMI scrambling enable - fix phys bo pread/pwrite with offset - add missing nospec on parallel submit slot - fix some NULL derefs xe: - drop force_execlist module param - gate observation streams with perf_allow_cpu - skip FORCE_WC and vm_bound check for external dma-bufs - dmemcg eviction support - remove unused NVL-S GuC - TLB invalidation improvements - NVL-S updated PCI-IDs and w/a - madvise: optimise invalidation path - fix infinite gt-reset loop in timeout recovery - update TTM device benefical_order - wait on external BO kernel fences in exec ioctl - add/use more KLV helpers - sriov: disable display in admin only PF mode - add RAS GPU health indicator - optimise TTM populate for DONTNEED BO - drop force_probe for NVL-s - add debugfs for pcode info amdxdna: - disable device buffer export nova: - build nova-core/nova-drm from drivers/gpu - export nova-core rust symbols (workaround) - GSP boot process consolidation - Boot GSP with vGPU enabled - TLV firmware image format support - Hopper/Blackwell fixes and cleanups - I/O projection adoption tyr: - firmware loading and MCU boot - add generic slot manager + MMU - GPU VM support ARM64 LPAE page tables - add kernel buffer object for internal allocations - add parser for Mali CSF - add MCU booting nouveau: - race fixes - check instmem iomapping at first use - add dmemcg support - expose NVDEC channels - add scanline position/head state support for GSP qxl: - convert simple encoder to regular ethosu: - add perf counter support etnaviv: - force flush on power register ops msm: - support DSC configuration with slice_per_pkt > 1 mxsfb: - fix disable sequence panthor: - support sparse mappings rockchip: - switch away from simple helpers - support YUV background color - fix layer config timeout - add edp support for rk3576 - add batch command submission function rocket: - error handling and NULL ptr deref fixes sun4i: - switch away from simple helpers imagination: - mark BXM-4-64 MC1 as support host1x: - support tegra264 tegra: - add DSI for tegra 20/30 v3d: - reduce PM runtime autosuspend delay - scheduler fixes and refactoring - deprecate v3d 3.3 and 4.1 - validate CPU job query boundaries hibmc: - improve plane format handling - switch to gem shmem mediatek: - cec: correct compat for mt7623-8167? exynos: - remove simple dependency - add error handling to encoder paths - take i2c adapter module reference -----BEGIN PGP SIGNATURE----- iQIzBAABCgAdFiEEEKbZHaGwW9KfbeusDHTzWXnEhr4FAmqGlb0ACgkQDHTzWXnE hr7l9A//TnfntEghigEEFobfJX+p9FzaOTPPia8paooAj52OBK8Z86WpbYwEo4K9 X+vPXPYpqgKSiGkkC33swAlylWs2v3JoZQ+CESERBk176Ql3ZKhicBINH+k8jIcX uaFoDgpgMoV1JCcvF/m48de8YRcejSN43rIucS0aIH5/r/YEyRsE4d4dzCXw/qD8 92tjbmH20mChfeo8MUNatZx+t8ssSOrVdqouLmmFB8tYTcca6qwN60uA+9VESVtd nZLCEiZD0FUI73oT7fmK/zL2rTb2pZRPFNdz0mb6f7UUpu7f8RYnroHrNsGa/FHl K5RD1/gSVpfc6CbrhPnePaRKvIGeEC4ief8YRRyeoNVT3cmkf+citpOoKN2JajF1 bub/ni2z1FGA3y1ckJb4Z6HmGHt5gki/KoAKCmZkJ7bb7WJq/JHMWEFfq4LuNjTA FSSxPozM4pb69DL02wwRJIEe8cYcc/gVTgrSkzR/tsVUoE6XI4AwZJ/Exa43+jVe hjNkAOMrl/+ma/WGQ4BPUVeTRPZP6RlNM4cSWvG1YA2Pf6+O+XLjukac6t+Ozj0X Fz1ePqYxELKSOYkZKxdpxRDyzY6SxvtyDHfHblo4p+BUvaXKP4wfNRexXETD6Qow P7rqYN6riDMRPI9CoPd25V7cbfYbvoV0iJzv3EQwU3pvB9Vu1ho= =TGZ1 -----END PGP SIGNATURE----- Merge tag 'drm-next-2026-08-20' of https://gitlab.freedesktop.org/drm/kernel Pull drm updates from Dave Airlie: "Highlights: - dmemcg eviction support is good for low VRAM things like Steam Machine - AMD adds gfx6-8 modifier support for older GPUs that enables a bunch of wayland stuff - i915/xe has some new hw support but also a lot of display refactoring Everything: perf: - export perf_allow_ APIs for xe udmabuf: - remove default size limit of 64MB rust: - i/o rework (signed tag from driver-core tree) - add registration guard and registration data - fix unbounded lifetimes in ioctl handler args - fix a drm_dev_register race - gem_shmem: add DmaResvGuard helper - gpuvm: require send/sync for driver data - implement send/sync for GpuVaAlloc and GpuVmBo - add SmContext lifetime - rename dma_handle to dma_address - change pci_sriov_get_totalvfs return to unsigned int core: - create drm_of_get_panel_orientation - send per-connector hotplug events - add thunderbolt UBHR tunneling support connector: - add color format property dmem: - introduce a peak file - accept one region per limit - add dmemcg support for eviction gpusvm: - reorg code to give drivers more flexibility atomic: - add create_state callback and helper - add documentation on atomic commit lifetime buddy: - add per-order free - add used block scoreboard - fix UAF - test buffer clearance on resume - add phys_addr->block helper gem: - drop DRIVER_GEM_GPUVA flag ttm: - be more aggressive allocating below protection limit sched: - add test suite for concurrent job submissions hdmi: - hook the color format property in helpers mipi-dsi: - add MIPI_DSI_MODE_DSC_ALL_SLICES_IN_PKT bridge: - add atomic create callbacks - drop atomic reset - display-connector: don't autoenable HPD IRQ - trigger initial HPD for DP - ti-sn65dsi83: remove NO_HFP and NO_HBP mode flags - analogix_dp: switch to DP link training helpers dp: - add support for DSC max delta BPP edid: - parse panel type from DisplayID 2.x Display Parameters sysfb: - improve panel, stride, framebuffer size validation panel: - implement ref counting for struct drm_panel - himax-hx83121a: add backlight regulator support - novatek-nt36672a: Inline panel init sequences - visionox-vtdr6130: enable DSC - novatek-nt37801: Use mipi_dsi_*_multi() functions - samsung-s6d16d0: Fix prepare error handling - support Novatek NT36536 plus DT bindings - sofef00: fix backlight updates - osd101t2587: use mipi_dsi_*_multi interface - panel-edp: adjust timing for AUO displays - panel-lvds: support Opto Logic SCX1001511GGC49 - panel-simple: support Kyocera tcg070wvlq - panel-edp: quirks - AUO B116XAT04.3, CMN N116BCP-EA2, CSW MNB601LS1-8 - BOE NV116WH2-M30, BOE NT116WHM-N21, BOE NV116FH1-M31 - BOE NV116FH1-M30, NV140FHM-N5B, TM156VDXP25 - BOE NE160QDM-NY1, MB116AS01 - new: - Samsung ATNA40HQ08-0, Anbernic TD4310 - Chipone ICNA35XX, Ilitek ILI9488 - Ilitek ILI7807S, Renesas R63419 - MNE001BS6-2, MNF601BS4-1, Sharp LQ120P1JX51 virtio: - add support for save/restore virtio_gpu_objects - abort vq wait on device removal amdgpu: - add color format DRM property - initial compute pipe reset support - add GFX 6-8 modifier support - initial DCN 6.0.0 support - dmemcg eviction support - improved boundary checking for bios parsing - RAS updates and rework - VCN secure submission fixes - 8K panel fix - Display KUNIT tests - parse panel type from DisplayID - Align IP discovery to pci device lifetime - SOC15 register macro cleanups - UVD memory placement fixes - GFX9 mode2 reset fixes - drop unnecessary BUG/BUG_ON - GFX8 soft reset rework - enable soft reset on GFX8 - PSP/SMU 15.0.9 update - VI ASPM fix - userq fixes - amdgpu_vm_get_task_info_pasid lifetime fix - DC CACP support - change system_unbound_wq with system_dfl_wq - Loosen VFCT bios parsing to deal with pci=realloc - SI/SMU7 AC/DC switch fix - VM fence handling fix - GEM close optimisation - Apple Studio Display fixes - DC FRL fixes amdkfd: - initial compute pipe reset support - allow applications to opt out of sigbus on fatal errors - improve CRIU boundary checks - MQD handling rework - move TBA/TMA from system to device memory - avoid topology-lock in kfd_mmap - SVM eviction fixes radeon: - fix unset CONFIG_ACPI build i915: - Novalake (NVL display version 35) timing generator enabling - NVL DC3CO enabling - enable UBHR link rates on thunderbolt tunnels - Reduce Xe3+ PM demand peak bandwidth - enable pipe DMC error interrupts for display 30+ - add kunit tests for DP link config selection - refactor and document DP link recovery - i915/xe driver display probe/remove/suspend/resume/shutdown cleanup and unification - i915/xe display runtime PM unified - Break i915 and xe panic dependency on struct intel_framebuffer - Streamline Pre/Post-CSC LUT loops - drop TGL DC3DO support - CDCLK santization - fix HDMI scrambling enable - fix phys bo pread/pwrite with offset - add missing nospec on parallel submit slot - fix some NULL derefs xe: - drop force_execlist module param - gate observation streams with perf_allow_cpu - skip FORCE_WC and vm_bound check for external dma-bufs - dmemcg eviction support - remove unused NVL-S GuC - TLB invalidation improvements - NVL-S updated PCI-IDs and w/a - madvise: optimise invalidation path - fix infinite gt-reset loop in timeout recovery - update TTM device benefical_order - wait on external BO kernel fences in exec ioctl - add/use more KLV helpers - sriov: disable display in admin only PF mode - add RAS GPU health indicator - optimise TTM populate for DONTNEED BO - drop force_probe for NVL-s - add debugfs for pcode info amdxdna: - disable device buffer export nova: - build nova-core/nova-drm from drivers/gpu - export nova-core rust symbols (workaround) - GSP boot process consolidation - Boot GSP with vGPU enabled - TLV firmware image format support - Hopper/Blackwell fixes and cleanups - I/O projection adoption tyr: - firmware loading and MCU boot - add generic slot manager + MMU - GPU VM support ARM64 LPAE page tables - add kernel buffer object for internal allocations - add parser for Mali CSF - add MCU booting nouveau: - race fixes - check instmem iomapping at first use - add dmemcg support - expose NVDEC channels - add scanline position/head state support for GSP qxl: - convert simple encoder to regular ethosu: - add perf counter support etnaviv: - force flush on power register ops msm: - support DSC configuration with slice_per_pkt > 1 mxsfb: - fix disable sequence panthor: - support sparse mappings rockchip: - switch away from simple helpers - support YUV background color - fix layer config timeout - add edp support for rk3576 - add batch command submission function rocket: - error handling and NULL ptr deref fixes sun4i: - switch away from simple helpers imagination: - mark BXM-4-64 MC1 as support host1x: - support tegra264 tegra: - add DSI for tegra 20/30 v3d: - reduce PM runtime autosuspend delay - scheduler fixes and refactoring - deprecate v3d 3.3 and 4.1 - validate CPU job query boundaries hibmc: - improve plane format handling - switch to gem shmem mediatek: - cec: correct compat for mt7623-8167? exynos: - remove simple dependency - add error handling to encoder paths - take i2c adapter module reference" * tag 'drm-next-2026-08-20' of https://gitlab.freedesktop.org/drm/kernel: (2074 commits) drm/xe/mcr: Take vcs1/vecs1 into account for first media slice drm/xe: Fix a bug in pc_adjust_freq_bounds() drm/xe: Fix xe_device_probe() failure drm/xe/drm_ras: Move has_drm_ras check to drm_ras layer drm/xe/ras: Fix boot-time ras error processing drm/amd/display: make DC_RUN_WITH_PREEMPTION_ENABLED misuse a build error drm/amd/pm: silence uninitialized variable warnings drm/amdgpu: skip BOs being torn down during GTT recovery drm/amdgpu: Reject UVD message with invalid number of h265 refs drm/amdgpu: keep PRT mappings off the vm_bo state lists drm/amdgpu: fix nbif 6.3.1 l1 low power not functional drm/amd/display: fix BT.2020 YCbCr output CSC matrices for DCE drm/amd/display: fix BT.2020 YCbCr limited output CSC matrix drm/amdgpu: Implement insert_end for VCE 3 drm/amdgpu: Fix UVD min buffer sizes drm/amdgpu: Fix UVD decode image min size calculation drm/amdgpu: Fix UVD dpb min size calculation for H264 drm/amdgpu: Reject UVD message with dimensions above 4096 drm/amdgpu: check ASPM on the dGPU host link drm/radeon: fix autosuspend cleanup during teardown ...
417 lines
12 KiB
Rust
417 lines
12 KiB
Rust
// SPDX-License-Identifier: GPL-2.0
|
|
|
|
use core::ops::Range;
|
|
|
|
use kernel::{
|
|
device,
|
|
dma::Device,
|
|
fmt,
|
|
io::Io,
|
|
num::Bounded,
|
|
pci,
|
|
prelude::*,
|
|
sizes::SizeConstants, //
|
|
};
|
|
|
|
use crate::{
|
|
bounded_enum,
|
|
driver::Bar0,
|
|
falcon::{
|
|
gsp::Gsp as GspFalcon,
|
|
sec2::Sec2 as Sec2Falcon,
|
|
Falcon, //
|
|
},
|
|
fb::SysmemFlush,
|
|
fsp::Fsp,
|
|
gsp::{
|
|
self,
|
|
commands::GetGspStaticInfoReply,
|
|
Gsp,
|
|
GspBootContext, //
|
|
},
|
|
regs,
|
|
vgpu::VgpuManager, //
|
|
};
|
|
|
|
mod hal;
|
|
|
|
macro_rules! define_chipset {
|
|
({ $($variant:ident = $value:expr),* $(,)* }) =>
|
|
{
|
|
/// Enum representation of the GPU chipset.
|
|
#[derive(fmt::Debug, Copy, Clone, PartialOrd, Ord, PartialEq, Eq)]
|
|
pub(crate) enum Chipset {
|
|
$($variant = $value),*,
|
|
}
|
|
|
|
impl Chipset {
|
|
pub(crate) const ALL: &'static [Chipset] = &[
|
|
$( Chipset::$variant, )*
|
|
];
|
|
|
|
::kernel::macros::paste!(
|
|
/// Returns the name of this chipset, in lowercase.
|
|
///
|
|
/// # Examples
|
|
///
|
|
/// ```
|
|
/// let chipset = Chipset::GA102;
|
|
/// assert_eq!(chipset.name(), "ga102");
|
|
/// ```
|
|
pub(crate) const fn name(&self) -> &'static str {
|
|
match *self {
|
|
$(
|
|
Chipset::$variant => stringify!([<$variant:lower>]),
|
|
)*
|
|
}
|
|
}
|
|
);
|
|
}
|
|
|
|
// TODO[FPRI]: replace with something like derive(FromPrimitive)
|
|
impl TryFrom<u32> for Chipset {
|
|
type Error = kernel::error::Error;
|
|
|
|
fn try_from(value: u32) -> Result<Self, Self::Error> {
|
|
match value {
|
|
$( $value => Ok(Chipset::$variant), )*
|
|
_ => Err(ENODEV),
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
define_chipset!({
|
|
// Turing
|
|
TU102 = 0x162,
|
|
TU104 = 0x164,
|
|
TU106 = 0x166,
|
|
TU117 = 0x167,
|
|
TU116 = 0x168,
|
|
// Ampere
|
|
GA100 = 0x170,
|
|
GA102 = 0x172,
|
|
GA103 = 0x173,
|
|
GA104 = 0x174,
|
|
GA106 = 0x176,
|
|
GA107 = 0x177,
|
|
// Hopper
|
|
GH100 = 0x180,
|
|
// Ada
|
|
AD102 = 0x192,
|
|
AD103 = 0x193,
|
|
AD104 = 0x194,
|
|
AD106 = 0x196,
|
|
AD107 = 0x197,
|
|
// Blackwell GB10x
|
|
GB100 = 0x1a0,
|
|
GB102 = 0x1a2,
|
|
// Blackwell GB20x
|
|
GB202 = 0x1b2,
|
|
GB203 = 0x1b3,
|
|
GB205 = 0x1b5,
|
|
GB206 = 0x1b6,
|
|
GB207 = 0x1b7,
|
|
});
|
|
|
|
impl Chipset {
|
|
pub(crate) const fn arch(self) -> Architecture {
|
|
match self {
|
|
Self::TU102 | Self::TU104 | Self::TU106 | Self::TU117 | Self::TU116 => {
|
|
Architecture::Turing
|
|
}
|
|
Self::GA100 | Self::GA102 | Self::GA103 | Self::GA104 | Self::GA106 | Self::GA107 => {
|
|
Architecture::Ampere
|
|
}
|
|
Self::GH100 => Architecture::Hopper,
|
|
Self::AD102 | Self::AD103 | Self::AD104 | Self::AD106 | Self::AD107 => {
|
|
Architecture::Ada
|
|
}
|
|
Self::GB100 | Self::GB102 => Architecture::BlackwellGB10x,
|
|
Self::GB202 | Self::GB203 | Self::GB205 | Self::GB206 | Self::GB207 => {
|
|
Architecture::BlackwellGB20x
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Returns the address range of the PCI config mirror space.
|
|
pub(crate) fn pci_config_mirror_range(self) -> Range<u32> {
|
|
hal::gpu_hal(self).pci_config_mirror_range()
|
|
}
|
|
}
|
|
|
|
// TODO
|
|
//
|
|
// The resulting strings are used to generate firmware paths, hence the
|
|
// generated strings have to be stable.
|
|
//
|
|
// Hence, replace with something like strum_macros derive(Display).
|
|
//
|
|
// For now, redirect to fmt::Debug for convenience.
|
|
impl fmt::Display for Chipset {
|
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
write!(f, "{self:?}")
|
|
}
|
|
}
|
|
|
|
bounded_enum! {
|
|
/// Enum representation of the GPU generation.
|
|
#[derive(fmt::Debug, Copy, Clone)]
|
|
pub(crate) enum Architecture with TryFrom<Bounded<u32, 6>> {
|
|
Turing = 0x16,
|
|
Ampere = 0x17,
|
|
Hopper = 0x18,
|
|
Ada = 0x19,
|
|
BlackwellGB10x = 0x1a,
|
|
BlackwellGB20x = 0x1b,
|
|
}
|
|
}
|
|
|
|
#[derive(Clone, Copy)]
|
|
pub(crate) struct Revision {
|
|
major: Bounded<u8, 4>,
|
|
minor: Bounded<u8, 4>,
|
|
}
|
|
|
|
impl From<regs::NV_PMC_BOOT_42> for Revision {
|
|
fn from(boot0: regs::NV_PMC_BOOT_42) -> Self {
|
|
Self {
|
|
major: boot0.major_revision().cast(),
|
|
minor: boot0.minor_revision().cast(),
|
|
}
|
|
}
|
|
}
|
|
|
|
impl fmt::Display for Revision {
|
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
write!(f, "{:x}.{:x}", self.major, self.minor)
|
|
}
|
|
}
|
|
|
|
/// Structure holding a basic description of the GPU: `Chipset` and `Revision`.
|
|
#[derive(Clone, Copy)]
|
|
pub(crate) struct Spec {
|
|
chipset: Chipset,
|
|
revision: Revision,
|
|
}
|
|
|
|
impl Spec {
|
|
fn new(dev: &device::Device, bar: Bar0<'_>) -> Result<Spec> {
|
|
// Some brief notes about boot0 and boot42, in chronological order:
|
|
//
|
|
// NV04 through NV50:
|
|
//
|
|
// Not supported by Nova. boot0 is necessary and sufficient to identify these GPUs.
|
|
// boot42 may not even exist on some of these GPUs.
|
|
//
|
|
// Fermi through Volta:
|
|
//
|
|
// Not supported by Nova. boot0 is still sufficient to identify these GPUs, but boot42
|
|
// is also guaranteed to be both present and accurate.
|
|
//
|
|
// Turing and later:
|
|
//
|
|
// Supported by Nova. Identified by first checking boot0 to ensure that the GPU is not
|
|
// from an earlier (pre-Fermi) era, and then using boot42 to precisely identify the GPU.
|
|
// Somewhere in the Rubin timeframe, boot0 will no longer have space to add new GPU IDs.
|
|
|
|
let boot0 = bar.read(regs::NV_PMC_BOOT_0);
|
|
|
|
if boot0.is_older_than_fermi() {
|
|
return Err(ENODEV);
|
|
}
|
|
|
|
let boot42 = bar.read(regs::NV_PMC_BOOT_42);
|
|
Spec::try_from(boot42).inspect_err(|_| {
|
|
dev_err!(dev, "Unsupported chipset: {}\n", boot42);
|
|
})
|
|
}
|
|
}
|
|
|
|
impl TryFrom<regs::NV_PMC_BOOT_42> for Spec {
|
|
type Error = Error;
|
|
|
|
fn try_from(boot42: regs::NV_PMC_BOOT_42) -> Result<Self> {
|
|
Ok(Self {
|
|
chipset: boot42.chipset()?,
|
|
revision: boot42.into(),
|
|
})
|
|
}
|
|
}
|
|
|
|
impl fmt::Display for Spec {
|
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
f.write_fmt(fmt!(
|
|
"Chipset: {}, Architecture: {:?}, Revision: {}",
|
|
self.chipset,
|
|
self.chipset.arch(),
|
|
self.revision
|
|
))
|
|
}
|
|
}
|
|
|
|
/// Self-contained resources to operate and drop the GSP.
|
|
#[pin_data(PinnedDrop)]
|
|
struct GspResources<'gpu> {
|
|
/// Device owning the GPU.
|
|
device: &'gpu pci::Device<device::Bound>,
|
|
/// Details about the chipset.
|
|
spec: Spec,
|
|
/// MMIO mapping of PCI BAR 0.
|
|
bar: Bar0<'gpu>,
|
|
/// GSP falcon instance, used for GSP boot up and cleanup.
|
|
gsp_falcon: Falcon<'gpu, GspFalcon>,
|
|
/// SEC2 falcon instance, used for GSP boot up and cleanup.
|
|
sec2_falcon: Falcon<'gpu, Sec2Falcon>,
|
|
/// FSP instance, if on an arch that supports it.
|
|
// TODO: use different resource types for each boot method, and make the relevant Gsp methods
|
|
// generic against them.
|
|
fsp: Option<Fsp<'gpu>>,
|
|
/// vGPU state detected before GSP boot.
|
|
vgpu: VgpuManager,
|
|
/// GSP runtime data.
|
|
#[pin]
|
|
gsp: Gsp,
|
|
/// GSP unload firmware bundle, if any.
|
|
unload_bundle: Option<gsp::UnloadBundle>,
|
|
}
|
|
|
|
/// Structure holding the resources required to operate the GPU.
|
|
#[pin_data]
|
|
pub(crate) struct Gpu<'gpu> {
|
|
spec: Spec,
|
|
/// Static GPU information as provided by the GSP.
|
|
gsp_static_info: GetGspStaticInfoReply,
|
|
/// GSP and its resources.
|
|
#[pin]
|
|
gsp_resources: GspResources<'gpu>,
|
|
/// System memory page required for flushing all pending GPU-side memory writes done through
|
|
/// PCIE into system memory, via sysmembar (A GPU-initiated HW memory-barrier operation).
|
|
///
|
|
/// Must be kept declared *after* `gsp_resources`, as the latter's `PinnedDrop` implementation
|
|
/// requires the sysmem flush page to be in place.
|
|
sysmem_flush: SysmemFlush<'gpu>,
|
|
}
|
|
|
|
#[pinned_drop]
|
|
impl PinnedDrop for GspResources<'_> {
|
|
fn drop(self: Pin<&mut Self>) {
|
|
let this = self.project();
|
|
let device = *this.device;
|
|
let bar = *this.bar;
|
|
let bundle = this.unload_bundle.take();
|
|
|
|
let _ = this
|
|
.gsp
|
|
.as_ref()
|
|
.get_ref()
|
|
.unload(
|
|
GspBootContext {
|
|
pdev: device,
|
|
bar,
|
|
chipset: this.spec.chipset,
|
|
gsp_falcon: &*this.gsp_falcon,
|
|
sec2_falcon: &*this.sec2_falcon,
|
|
fsp: this.fsp.as_mut(),
|
|
vgpu: &*this.vgpu,
|
|
},
|
|
bundle,
|
|
)
|
|
.inspect_err(|e| dev_err!(device, "failed to unload GSP: {:?}\n", e));
|
|
}
|
|
}
|
|
|
|
impl<'gpu> Gpu<'gpu> {
|
|
pub(crate) fn new<'a>(
|
|
pdev: &'gpu pci::Device<device::Core<'a>>,
|
|
bar: Bar0<'gpu>,
|
|
) -> impl PinInit<Self, Error> + use<'gpu, 'a> {
|
|
let dev = pdev.as_ref();
|
|
|
|
try_pin_init!(Self {
|
|
spec: Spec::new(dev, bar).inspect(|spec| {
|
|
dev_info!(dev,"NVIDIA ({})\n", spec);
|
|
})?,
|
|
|
|
// We must wait for GFW_BOOT completion before doing any significant setup on the GPU.
|
|
_: {
|
|
let hal = hal::gpu_hal(spec.chipset);
|
|
let dma_mask = hal.dma_mask();
|
|
|
|
// SAFETY: `Gpu` owns all DMA allocations for this device, and we are
|
|
// still constructing it, so no concurrent DMA allocations can exist.
|
|
unsafe { pdev.dma_set_mask_and_coherent(dma_mask)? };
|
|
|
|
hal.wait_gfw_boot_completion(bar)
|
|
.inspect_err(|_| dev_err!(dev, "GFW boot did not complete\n"))?;
|
|
},
|
|
|
|
// Initialize this early because `gsp_resources` depends on it.
|
|
sysmem_flush: SysmemFlush::register(dev, bar, spec.chipset)?,
|
|
|
|
gsp_resources <- try_pin_init!(GspResources {
|
|
device: pdev,
|
|
|
|
spec: *spec,
|
|
|
|
bar,
|
|
|
|
gsp_falcon: Falcon::new(
|
|
dev,
|
|
spec.chipset,
|
|
bar
|
|
)
|
|
.inspect(|falcon| falcon.clear_swgen0_intr())?,
|
|
|
|
sec2_falcon: Falcon::new(dev, spec.chipset, bar)?,
|
|
|
|
fsp: Fsp::try_new(dev, bar, spec.chipset)?,
|
|
|
|
vgpu: VgpuManager::new(pdev, spec.chipset, fsp.as_mut()),
|
|
|
|
gsp <- Gsp::new(pdev),
|
|
|
|
// This member must be initialized last, so the `UnloadBundle` can never be dropped
|
|
// from outside of the constructed `GspResources`, ensuring that the unload sequence
|
|
// is properly run in case of failure.
|
|
unload_bundle: gsp.boot(GspBootContext {
|
|
pdev,
|
|
bar,
|
|
chipset: spec.chipset,
|
|
gsp_falcon,
|
|
sec2_falcon,
|
|
fsp: fsp.as_mut(),
|
|
vgpu,
|
|
})?,
|
|
}),
|
|
|
|
gsp_static_info: {
|
|
// Obtain and display basic GPU information.
|
|
let info = gsp_resources.gsp.get_static_info(bar)?;
|
|
match info.gpu_name() {
|
|
Ok(name) => dev_info!(dev, "GPU name: {}\n", name),
|
|
Err(e) => dev_warn!(dev, "GPU name unavailable: {:?}\n", e),
|
|
}
|
|
|
|
if !info.usable_fb_regions.is_empty() {
|
|
dev_dbg!(dev, "Usable FB regions:\n");
|
|
for region in &info.usable_fb_regions {
|
|
dev_dbg!(dev, " - {:#x?}\n", region);
|
|
}
|
|
|
|
dev_dbg!(
|
|
dev,
|
|
"Total usable VRAM: {} MiB\n",
|
|
info.usable_fb_regions.iter().fold(0u64, |res, region| res
|
|
.saturating_add(region.end - region.start))
|
|
/ u64::SZ_1M
|
|
);
|
|
}
|
|
|
|
info
|
|
}
|
|
})
|
|
}
|
|
}
|