Re: [PATCH v9 7/7] drm/tyr: add Microcontroller Unit (MCU) booting
Daniel Almeida <[email protected]>
| Newsgroups | org.kernel.vger.rust-for-linux,org.freedesktop.lists.dri-devel,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
> On 22 Jul 2026, at 20:54, Deborah Brouwer <[email protected]> wrote: > > Add a firmware module to load, parse, and map the MCU firmware sections > into shared GEM memory at the required virtual addresses accessible by the > GPU. > > Create a firmware instance during probe and store it inside the > TyrDrmRegistrationData to keep it alive after probe. Use the firmware > instance to boot the MCU. > > Remove the dead-code annotations from the MMU, VM, slot manager, and > kernel BO code now that these paths are used by the firmware module. > > Update Kconfig to add the RUST_FW_LOADER_ABSTRACTIONS dependency > required by this module. > > Co-developed-by: Boris Brezillon <[email protected]> > Signed-off-by: Boris Brezillon <[email protected]> > Signed-off-by: Deborah Brouwer <[email protected]> > --- > drivers/gpu/drm/tyr/Kconfig | 1 + > drivers/gpu/drm/tyr/driver.rs | 23 ++- > drivers/gpu/drm/tyr/fw.rs | 321 ++++++++++++++++++++++++++++++++++++++++++ > drivers/gpu/drm/tyr/gem.rs | 3 - > drivers/gpu/drm/tyr/tyr.rs | 1 + > drivers/gpu/drm/tyr/vm.rs | 1 - > 6 files changed, 341 insertions(+), 9 deletions(-) > > diff --git a/drivers/gpu/drm/tyr/Kconfig b/drivers/gpu/drm/tyr/Kconfig > index 79ea4bb214de..8f13e49f11f9 100644 > --- a/drivers/gpu/drm/tyr/Kconfig > +++ b/drivers/gpu/drm/tyr/Kconfig > @@ -13,6 +13,7 @@ config DRM_TYR > select IOMMU_IO_PGTABLE_LPAE > select RUST_DRM_GEM_SHMEM_HELPER > select RUST_DRM_GPUVM > + select RUST_FW_LOADER_ABSTRACTIONS > help > Rust DRM driver for ARM Mali CSF-based GPUs. > > diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs > index b6528d8cd3ce..8f87fd5b772a 100644 > --- a/drivers/gpu/drm/tyr/driver.rs > +++ b/drivers/gpu/drm/tyr/driver.rs > @@ -37,6 +37,7 @@ > > use crate::{ > file::TyrDrmFileData, > + fw::Firmware, > gem::Bo, > gpu, > gpu::GpuInfo, > @@ -67,6 +68,9 @@ pub(crate) struct TyrDrmRegistrationData<'bound> { > /// Parent platform device. > pub(crate) pdev: &'bound platform::Device<Bound>, > > + /// Firmware sections. > + pub(crate) fw: Firmware<'bound>, > + > #[pin] > clks: Mutex<Clocks>, > > @@ -144,10 +148,21 @@ fn probe<'bound>( > > let unreg_dev = drm::UnregisteredDevice::<TyrDrmDriver>::new(pdev, Ok(()))?; > > - let _mmu = Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_info)?; > + let mmu = Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_info)?; > + > + let firmware = Firmware::new( > + pdev.as_ref(), > + iomem.clone(), > + &unreg_dev, > + mmu.as_arc_borrow(), > + &gpu_info, > + )?; > + > + firmware.boot()?; > > - let reg_data = try_pin_init!(TyrDrmRegistrationData { > + let reg_data = pin_init!(TyrDrmRegistrationData { > pdev, > + fw: firmware, > clks <- new_mutex!(Clocks { > core: core_clk, > stacks: stacks_clk, > @@ -167,9 +182,7 @@ fn probe<'bound>( > > let driver = TyrPlatformDriverData { _reg: reg }; > > - // We need this to be dev_info!() because dev_dbg!() does not work at > - // all in Rust for now, and we need to see whether probe succeeded. > - dev_info!(pdev, "Tyr initialized correctly.\n"); > + dev_dbg!(pdev, "Tyr initialized correctly."); > Ok(driver) > } > } > diff --git a/drivers/gpu/drm/tyr/fw.rs b/drivers/gpu/drm/tyr/fw.rs > new file mode 100644 > index 000000000000..df11647023c2 > --- /dev/null > +++ b/drivers/gpu/drm/tyr/fw.rs > @@ -0,0 +1,321 @@ > +// SPDX-License-Identifier: GPL-2.0 or MIT > + > +//! Firmware loading and management for Mali CSF GPU. > +//! > +//! This module handles loading the Mali GPU firmware binary, parsing it into sections, > +//! and mapping those sections into the MCU's virtual address space. Each firmware section > +//! has specific properties (read/write/execute permissions, cache modes) and must be loaded > +//! at specific virtual addresses expected by the MCU. > +//! > +//! See [`Firmware`] for the main firmware management interface and [`Section`] for > +//! individual firmware sections. > +//! > +//! [`Firmware`]: crate::fw::Firmware > +//! [`Section`]: crate::fw::Section > + > +use kernel::{ > + device::{ > + Bound, > + Device, // > + }, > + drm::{ > + gem::BaseObject, // > + }, > + io::{ > + poll, > + Io, // > + }, > + num::Bounded, > + prelude::*, > + register, > + str::CString, > + sync::{ > + Arc, > + ArcBorrow, // > + }, > + time, // > +}; > + > +use crate::{ > + driver::{ > + IoMem, > + TyrDrmDevice, // > + }, > + fw::parser::{ > + FwParser, > + ParsedSection, // > + }, > + gem, > + gem::{ > + KernelBo, > + KernelBoVaAlloc, // > + }, > + gpu::GpuInfo, > + > + mmu::Mmu, > + regs::{ > + gpu_control::{ > + McuControlMode, > + McuStatus, > + GPU_ID, > + MCU_CONTROL, > + MCU_STATUS, // > + }, // > + job_control::{ > + JOB_IRQ_CLEAR, > + JOB_IRQ_RAWSTAT, // > + }, // > + }, > + vm::Vm, // > +}; > + > +mod parser; > + > +pub(super) const CSF_MCU_SHARED_REGION_START: u32 = 0x04000000; > + > +#[derive(Copy, Clone, Debug, PartialEq, Eq)] > +#[repr(u8)] > +pub(super) enum CacheMode { > + None = 0, > + Cached = 1, > + UncachedCoherent = 2, > + CachedCoherent = 3, > +} > + > +impl From<Bounded<u32, 2>> for CacheMode { > + fn from(value: Bounded<u32, 2>) -> Self { > + match value.get() { > + 0 => Self::None, > + 1 => Self::Cached, > + 2 => Self::UncachedCoherent, > + 3 => Self::CachedCoherent, > + _ => unreachable!(), > + } > + } > +} > + > +impl From<CacheMode> for Bounded<u32, 2> { > + fn from(value: CacheMode) -> Self { > + Bounded::try_new(value as u32).unwrap() > + } > +} > + > +register! { > + #[allow(non_upper_case_globals)] > + pub(super) SectionFlags(u32) @ 0x0 { > + 0:0 read => bool; > + 1:1 write => bool; > + 2:2 exec => bool; > + 4:3 cache_mode => CacheMode; > + 5:5 prot => bool; > + 30:30 shared => bool; > + 31:31 zero => bool; > + } > +} > + > +impl SectionFlags { > + const VALID_MASK: u32 = Self::READ_MASK > + | Self::WRITE_MASK > + | Self::EXEC_MASK > + | Self::CACHE_MODE_MASK > + | Self::PROT_MASK > + | Self::SHARED_MASK > + | Self::ZERO_MASK; > + > + fn try_from_fw(value: u32) -> Result<Self> { > + if value & !Self::VALID_MASK != 0 { > + Err(EINVAL) > + } else { > + Ok(Self::from_raw(value)) > + } > + } > +} > + > +/// A parsed section of the firmware binary. > +struct Section<'bound> { > + // Raw firmware section data for reset purposes > + #[expect(dead_code)] > + data: KVec<u8>, > + > + // Keep the BO backing this firmware section so that both the > + // GPU mapping and CPU mapping remain valid until the Section is dropped. > + #[expect(dead_code)] > + mem: gem::KernelBo<'bound>, > +} > + > +/// Loaded firmware with sections mapped into MCU VM. > +pub(crate) struct Firmware<'bound> { > + /// Iomem need to access registers. > + iomem: Arc<IoMem<'bound>>, > + > + /// MCU VM. > + vm: Arc<Vm<'bound>>, > + > + /// List of firmware sections. > + #[expect(dead_code)] > + sections: KVec<Section<'bound>>, > +} > + > +impl<'bound> Drop for Firmware<'bound> { > + fn drop(&mut self) { > + // Stop the MCU before releasing its firmware mappings and memory. > + let _ = self.stop(); > + > + // AS slots retain a VM ref, we need to kill the circular ref manually. > + self.vm.kill(); > + } > +} > + > +impl<'bound> Firmware<'bound> { > + fn init_section_mem(dev: &Device, mem: &mut KernelBo<'bound>, data: &KVec<u8>) -> Result { > + if data.is_empty() { > + return Ok(()); > + } > + > + let vmap = mem.bo().vmap::<0>()?; > + let size = mem.bo().size(); > + > + if data.len() > size { > + dev_err!(dev, "fw section {} bigger than BO {}", data.len(), size); > + return Err(EINVAL); > + } > + > + for (i, &byte) in data.iter().enumerate() { > + vmap.try_write8(byte, i)?; Can this use vmap.copy_from_slice() instead? > + } > + > + Ok(()) > + } > + > + fn request(ddev: &TyrDrmDevice, gpu_info: &GpuInfo) -> Result<kernel::firmware::Firmware> { > + let gpu_id = GPU_ID::from_raw(gpu_info.gpu_id); > + > + let path = CString::try_from_fmt(fmt!( > + "arm/mali/arch{}.{}/mali_csffw.bin", > + gpu_id.arch_major().get(), > + gpu_id.arch_minor().get() > + ))?; > + > + kernel::firmware::Firmware::request(&path, ddev.as_ref().as_ref()) > + } > + > + fn load( > + dev: &Device, > + ddev: &TyrDrmDevice, > + gpu_info: &GpuInfo, > + ) -> Result<(kernel::firmware::Firmware, KVec<ParsedSection>)> { > + let fw = Self::request(ddev, gpu_info)?; > + let mut parser = FwParser::new(dev, fw.data()); > + > + let parsed_sections = parser.parse()?; > + > + Ok((fw, parsed_sections)) > + } > + > + /// Load firmware and map sections into MCU VM. > + pub(crate) fn new( > + dev: &'bound Device<Bound>, > + iomem: Arc<IoMem<'bound>>, > + ddev: &TyrDrmDevice, > + mmu: ArcBorrow<'_, Mmu<'bound>>, > + gpu_info: &GpuInfo, > + ) -> Result<Firmware<'bound>> { > + let vm = Vm::new(dev, ddev, mmu, gpu_info)?; > + vm.activate()?; > + > + let result = (|| { > + let (fw, parsed_sections) = Self::load(dev, ddev, gpu_info)?; > + let mut sections = KVec::new(); > + for parsed in parsed_sections { > + let size = u64::from(parsed.va.end.checked_sub(parsed.va.start).ok_or(EINVAL)?); > + > + let va = u64::from(parsed.va.start); > + > + let mut mem = KernelBo::new( > + ddev, > + vm.clone(), > + size, > + KernelBoVaAlloc::Explicit(va), > + parsed.vm_map_flags, > + )?; > + > + let section_start = parsed.data_range.start as usize; > + let section_end = parsed.data_range.end as usize; > + let mut data = KVec::new(); > + > + // Ensure that the firmware slice is not out of bounds. > + let fw_data = fw.data(); > + let bytes = fw_data.get(section_start..section_end).ok_or(EINVAL)?; > + data.extend_from_slice(bytes, GFP_KERNEL)?; > + > + Self::init_section_mem(dev, &mut mem, &data)?; > + > + sections.push(Section { data, mem }, GFP_KERNEL)?; > + } > + > + Ok(Firmware { > + iomem, > + vm: vm.clone(), > + sections, > + }) > + })(); > + > + if result.is_err() { > + vm.kill(); > + } > + > + result > + } > + > + pub(crate) fn boot(&self) -> Result { > + let io = &self.iomem; > + > + // Discard any stale global interrupt. > + io.write_reg(JOB_IRQ_CLEAR::zeroed().with_glb(true)); > + > + io.write_reg(MCU_CONTROL::zeroed().with_req(McuControlMode::Auto)); > + > + if let Err(e) = poll::read_poll_timeout( > + || Ok((io.read(MCU_STATUS), io.read(JOB_IRQ_RAWSTAT))), > + |(mcu_status, irq_rawstat)| { > + mcu_status.value() == McuStatus::Enabled && irq_rawstat.glb() > + }, > + time::Delta::from_millis(1), > + time::Delta::from_millis(100), > + ) { > + let status = io.read(MCU_STATUS); > + dev_err!( > + self.vm.dev(), > + "MCU failed to boot, status: {:?}", > + status.value() > + ); > + return Err(e); > + } > + > + io.write_reg(JOB_IRQ_CLEAR::zeroed().with_glb(true)); > + > + Ok(()) > + } > + > + fn stop(&self) -> Result { > + let io = &self.iomem; > + io.write_reg(MCU_CONTROL::zeroed().with_req(McuControlMode::Disable)); > + > + if let Err(e) = poll::read_poll_timeout( > + || Ok(io.read(MCU_STATUS)), > + |status| status.value() == McuStatus::Disabled, > + time::Delta::from_micros(10), > + time::Delta::from_millis(100), > + ) { > + let status = io.read(MCU_STATUS); > + dev_err!( > + self.vm.dev(), > + "MCU failed to stop, status: {:?}", > + status.value() > + ); > + return Err(e); > + } > + > + Ok(()) > + } > +} > diff --git a/drivers/gpu/drm/tyr/gem.rs b/drivers/gpu/drm/tyr/gem.rs > index b371299028d6..342d3303a0c0 100644 > --- a/drivers/gpu/drm/tyr/gem.rs > +++ b/drivers/gpu/drm/tyr/gem.rs > @@ -72,7 +72,6 @@ pub(crate) fn new_dummy_object(ddev: &TyrDrmDevice) -> Result<ARef<Bo>> { > /// An automatic VA allocation strategy will be added in the future. > pub(crate) enum KernelBoVaAlloc { > /// Explicit VA address specified by the caller. > - #[expect(dead_code)] > Explicit(u64), > } > > @@ -98,7 +97,6 @@ impl<'bound> KernelBo<'bound> { > /// This function allocates a new shmem-backed GEM object and immediately maps > /// it into the specified GPU virtual memory space. The mapping is automatically > /// cleaned up when the [`KernelBo`] is dropped. > - #[expect(dead_code)] > pub(crate) fn new( > ddev: &TyrDrmDevice, > vm: Arc<Vm<'bound>>, > @@ -135,7 +133,6 @@ pub(crate) fn new( > }) > } > > - #[expect(dead_code)] > pub(crate) fn bo(&self) -> &Bo { > &self.bo > } > diff --git a/drivers/gpu/drm/tyr/tyr.rs b/drivers/gpu/drm/tyr/tyr.rs > index 92f6885cdaae..e7ec450bdc9c 100644 > --- a/drivers/gpu/drm/tyr/tyr.rs > +++ b/drivers/gpu/drm/tyr/tyr.rs > @@ -9,6 +9,7 @@ > > mod driver; > mod file; > +mod fw; > mod gem; > mod gpu; > mod mmu; > diff --git a/drivers/gpu/drm/tyr/vm.rs b/drivers/gpu/drm/tyr/vm.rs > index c113820b5505..418f98ab07fb 100644 > --- a/drivers/gpu/drm/tyr/vm.rs > +++ b/drivers/gpu/drm/tyr/vm.rs > @@ -6,7 +6,6 @@ > //! the illusion of owning the entire virtual address (VA) range, similar to CPU virtual memory. > //! Each virtual memory (VM) area is backed by ARM64 LPAE Stage 1 page tables and can be > //! mapped into hardware address space (AS) slots for GPU execution. > -#![expect(dead_code)] > > use core::marker::PhantomData; > use core::ops::Range; > > -- > 2.55.0 >