Re: [PATCH v9 7/7] drm/tyr: add Microcontroller Unit (MCU) booting

Daniel Almeida <[email protected]>
Newsgroups org.kernel.vger.rust-for-linux,org.freedesktop.lists.dri-devel,org.kernel.vger.linux-kernel
Message-ID <[email protected]>

> On 22 Jul 2026, at 20:54, Deborah Brouwer <[email protected]> wrote:
> 
> Add a firmware module to load, parse, and map the MCU firmware sections
> into shared GEM memory at the required virtual addresses accessible by the
> GPU.
> 
> Create a firmware instance during probe and store it inside the
> TyrDrmRegistrationData to keep it alive after probe. Use the firmware
> instance to boot the MCU.
> 
> Remove the dead-code annotations from the MMU, VM, slot manager, and
> kernel BO code now that these paths are used by the firmware module.
> 
> Update Kconfig to add the RUST_FW_LOADER_ABSTRACTIONS dependency
> required by this module.
> 
> Co-developed-by: Boris Brezillon <[email protected]>
> Signed-off-by: Boris Brezillon <[email protected]>
> Signed-off-by: Deborah Brouwer <[email protected]>
> ---
> drivers/gpu/drm/tyr/Kconfig   |   1 +
> drivers/gpu/drm/tyr/driver.rs |  23 ++-
> drivers/gpu/drm/tyr/fw.rs     | 321 ++++++++++++++++++++++++++++++++++++++++++
> drivers/gpu/drm/tyr/gem.rs    |   3 -
> drivers/gpu/drm/tyr/tyr.rs    |   1 +
> drivers/gpu/drm/tyr/vm.rs     |   1 -
> 6 files changed, 341 insertions(+), 9 deletions(-)
> 
> diff --git a/drivers/gpu/drm/tyr/Kconfig b/drivers/gpu/drm/tyr/Kconfig
> index 79ea4bb214de..8f13e49f11f9 100644
> --- a/drivers/gpu/drm/tyr/Kconfig
> +++ b/drivers/gpu/drm/tyr/Kconfig
> @@ -13,6 +13,7 @@ config DRM_TYR
> select IOMMU_IO_PGTABLE_LPAE
> select RUST_DRM_GEM_SHMEM_HELPER
> select RUST_DRM_GPUVM
> + select RUST_FW_LOADER_ABSTRACTIONS
> help
>  Rust DRM driver for ARM Mali CSF-based GPUs.
> 
> diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs
> index b6528d8cd3ce..8f87fd5b772a 100644
> --- a/drivers/gpu/drm/tyr/driver.rs
> +++ b/drivers/gpu/drm/tyr/driver.rs
> @@ -37,6 +37,7 @@
> 
> use crate::{
>     file::TyrDrmFileData,
> +    fw::Firmware,
>     gem::Bo,
>     gpu,
>     gpu::GpuInfo,
> @@ -67,6 +68,9 @@ pub(crate) struct TyrDrmRegistrationData<'bound> {
>     /// Parent platform device.
>     pub(crate) pdev: &'bound platform::Device<Bound>,
> 
> +    /// Firmware sections.
> +    pub(crate) fw: Firmware<'bound>,
> +
>     #[pin]
>     clks: Mutex<Clocks>,
> 
> @@ -144,10 +148,21 @@ fn probe<'bound>(
> 
>         let unreg_dev = drm::UnregisteredDevice::<TyrDrmDriver>::new(pdev, Ok(()))?;
> 
> -        let _mmu = Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_info)?;
> +        let mmu = Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_info)?;
> +
> +        let firmware = Firmware::new(
> +            pdev.as_ref(),
> +            iomem.clone(),
> +            &unreg_dev,
> +            mmu.as_arc_borrow(),
> +            &gpu_info,
> +        )?;
> +
> +        firmware.boot()?;
> 
> -        let reg_data = try_pin_init!(TyrDrmRegistrationData {
> +        let reg_data = pin_init!(TyrDrmRegistrationData {
>                 pdev,
> +                fw: firmware,
>                 clks <- new_mutex!(Clocks {
>                     core: core_clk,
>                     stacks: stacks_clk,
> @@ -167,9 +182,7 @@ fn probe<'bound>(
> 
>         let driver = TyrPlatformDriverData { _reg: reg };
> 
> -        // We need this to be dev_info!() because dev_dbg!() does not work at
> -        // all in Rust for now, and we need to see whether probe succeeded.
> -        dev_info!(pdev, "Tyr initialized correctly.\n");
> +        dev_dbg!(pdev, "Tyr initialized correctly.");
>         Ok(driver)
>     }
> }
> diff --git a/drivers/gpu/drm/tyr/fw.rs b/drivers/gpu/drm/tyr/fw.rs
> new file mode 100644
> index 000000000000..df11647023c2
> --- /dev/null
> +++ b/drivers/gpu/drm/tyr/fw.rs
> @@ -0,0 +1,321 @@
> +// SPDX-License-Identifier: GPL-2.0 or MIT
> +
> +//! Firmware loading and management for Mali CSF GPU.
> +//!
> +//! This module handles loading the Mali GPU firmware binary, parsing it into sections,
> +//! and mapping those sections into the MCU's virtual address space. Each firmware section
> +//! has specific properties (read/write/execute permissions, cache modes) and must be loaded
> +//! at specific virtual addresses expected by the MCU.
> +//!
> +//! See [`Firmware`] for the main firmware management interface and [`Section`] for
> +//! individual firmware sections.
> +//!
> +//! [`Firmware`]: crate::fw::Firmware
> +//! [`Section`]: crate::fw::Section
> +
> +use kernel::{
> +    device::{
> +        Bound,
> +        Device, //
> +    },
> +    drm::{
> +        gem::BaseObject, //
> +    },
> +    io::{
> +        poll,
> +        Io, //
> +    },
> +    num::Bounded,
> +    prelude::*,
> +    register,
> +    str::CString,
> +    sync::{
> +        Arc,
> +        ArcBorrow, //
> +    },
> +    time, //
> +};
> +
> +use crate::{
> +    driver::{
> +        IoMem,
> +        TyrDrmDevice, //
> +    },
> +    fw::parser::{
> +        FwParser,
> +        ParsedSection, //
> +    },
> +    gem,
> +    gem::{
> +        KernelBo,
> +        KernelBoVaAlloc, //
> +    },
> +    gpu::GpuInfo,
> +
> +    mmu::Mmu,
> +    regs::{
> +        gpu_control::{
> +            McuControlMode,
> +            McuStatus,
> +            GPU_ID,
> +            MCU_CONTROL,
> +            MCU_STATUS, //
> +        }, //
> +        job_control::{
> +            JOB_IRQ_CLEAR,
> +            JOB_IRQ_RAWSTAT, //
> +        }, //
> +    },
> +    vm::Vm, //
> +};
> +
> +mod parser;
> +
> +pub(super) const CSF_MCU_SHARED_REGION_START: u32 = 0x04000000;
> +
> +#[derive(Copy, Clone, Debug, PartialEq, Eq)]
> +#[repr(u8)]
> +pub(super) enum CacheMode {
> +    None = 0,
> +    Cached = 1,
> +    UncachedCoherent = 2,
> +    CachedCoherent = 3,
> +}
> +
> +impl From<Bounded<u32, 2>> for CacheMode {
> +    fn from(value: Bounded<u32, 2>) -> Self {
> +        match value.get() {
> +            0 => Self::None,
> +            1 => Self::Cached,
> +            2 => Self::UncachedCoherent,
> +            3 => Self::CachedCoherent,
> +            _ => unreachable!(),
> +        }
> +    }
> +}
> +
> +impl From<CacheMode> for Bounded<u32, 2> {
> +    fn from(value: CacheMode) -> Self {
> +        Bounded::try_new(value as u32).unwrap()
> +    }
> +}
> +
> +register! {
> +     #[allow(non_upper_case_globals)]
> +    pub(super) SectionFlags(u32) @ 0x0 {
> +        0:0 read => bool;
> +        1:1 write => bool;
> +        2:2 exec => bool;
> +        4:3 cache_mode => CacheMode;
> +        5:5 prot => bool;
> +        30:30 shared => bool;
> +        31:31 zero => bool;
> +    }
> +}
> +
> +impl SectionFlags {
> +    const VALID_MASK: u32 = Self::READ_MASK
> +        | Self::WRITE_MASK
> +        | Self::EXEC_MASK
> +        | Self::CACHE_MODE_MASK
> +        | Self::PROT_MASK
> +        | Self::SHARED_MASK
> +        | Self::ZERO_MASK;
> +
> +    fn try_from_fw(value: u32) -> Result<Self> {
> +        if value & !Self::VALID_MASK != 0 {
> +            Err(EINVAL)
> +        } else {
> +            Ok(Self::from_raw(value))
> +        }
> +    }
> +}
> +
> +/// A parsed section of the firmware binary.
> +struct Section<'bound> {
> +    // Raw firmware section data for reset purposes
> +    #[expect(dead_code)]
> +    data: KVec<u8>,
> +
> +    // Keep the BO backing this firmware section so that both the
> +    // GPU mapping and CPU mapping remain valid until the Section is dropped.
> +    #[expect(dead_code)]
> +    mem: gem::KernelBo<'bound>,
> +}
> +
> +/// Loaded firmware with sections mapped into MCU VM.
> +pub(crate) struct Firmware<'bound> {
> +    /// Iomem need to access registers.
> +    iomem: Arc<IoMem<'bound>>,
> +
> +    /// MCU VM.
> +    vm: Arc<Vm<'bound>>,
> +
> +    /// List of firmware sections.
> +    #[expect(dead_code)]
> +    sections: KVec<Section<'bound>>,
> +}
> +
> +impl<'bound> Drop for Firmware<'bound> {
> +    fn drop(&mut self) {
> +        // Stop the MCU before releasing its firmware mappings and memory.
> +        let _ = self.stop();
> +
> +        // AS slots retain a VM ref, we need to kill the circular ref manually.
> +        self.vm.kill();
> +    }
> +}
> +
> +impl<'bound> Firmware<'bound> {
> +    fn init_section_mem(dev: &Device, mem: &mut KernelBo<'bound>, data: &KVec<u8>) -> Result {
> +        if data.is_empty() {
> +            return Ok(());
> +        }
> +
> +        let vmap = mem.bo().vmap::<0>()?;
> +        let size = mem.bo().size();
> +
> +        if data.len() > size {
> +            dev_err!(dev, "fw section {} bigger than BO {}", data.len(), size);
> +            return Err(EINVAL);
> +        }
> +
> +        for (i, &byte) in data.iter().enumerate() {
> +            vmap.try_write8(byte, i)?;

Can this use vmap.copy_from_slice() instead?

> +        }
> +
> +        Ok(())
> +    }
> +
> +    fn request(ddev: &TyrDrmDevice, gpu_info: &GpuInfo) -> Result<kernel::firmware::Firmware> {
> +        let gpu_id = GPU_ID::from_raw(gpu_info.gpu_id);
> +
> +        let path = CString::try_from_fmt(fmt!(
> +            "arm/mali/arch{}.{}/mali_csffw.bin",
> +            gpu_id.arch_major().get(),
> +            gpu_id.arch_minor().get()
> +        ))?;
> +
> +        kernel::firmware::Firmware::request(&path, ddev.as_ref().as_ref())
> +    }
> +
> +    fn load(
> +        dev: &Device,
> +        ddev: &TyrDrmDevice,
> +        gpu_info: &GpuInfo,
> +    ) -> Result<(kernel::firmware::Firmware, KVec<ParsedSection>)> {
> +        let fw = Self::request(ddev, gpu_info)?;
> +        let mut parser = FwParser::new(dev, fw.data());
> +
> +        let parsed_sections = parser.parse()?;
> +
> +        Ok((fw, parsed_sections))
> +    }
> +
> +    /// Load firmware and map sections into MCU VM.
> +    pub(crate) fn new(
> +        dev: &'bound Device<Bound>,
> +        iomem: Arc<IoMem<'bound>>,
> +        ddev: &TyrDrmDevice,
> +        mmu: ArcBorrow<'_, Mmu<'bound>>,
> +        gpu_info: &GpuInfo,
> +    ) -> Result<Firmware<'bound>> {
> +        let vm = Vm::new(dev, ddev, mmu, gpu_info)?;
> +        vm.activate()?;
> +
> +        let result = (|| {
> +            let (fw, parsed_sections) = Self::load(dev, ddev, gpu_info)?;
> +            let mut sections = KVec::new();
> +            for parsed in parsed_sections {
> +                let size = u64::from(parsed.va.end.checked_sub(parsed.va.start).ok_or(EINVAL)?);
> +
> +                let va = u64::from(parsed.va.start);
> +
> +                let mut mem = KernelBo::new(
> +                    ddev,
> +                    vm.clone(),
> +                    size,
> +                    KernelBoVaAlloc::Explicit(va),
> +                    parsed.vm_map_flags,
> +                )?;
> +
> +                let section_start = parsed.data_range.start as usize;
> +                let section_end = parsed.data_range.end as usize;
> +                let mut data = KVec::new();
> +
> +                // Ensure that the firmware slice is not out of bounds.
> +                let fw_data = fw.data();
> +                let bytes = fw_data.get(section_start..section_end).ok_or(EINVAL)?;
> +                data.extend_from_slice(bytes, GFP_KERNEL)?;
> +
> +                Self::init_section_mem(dev, &mut mem, &data)?;
> +
> +                sections.push(Section { data, mem }, GFP_KERNEL)?;
> +            }
> +
> +            Ok(Firmware {
> +                iomem,
> +                vm: vm.clone(),
> +                sections,
> +            })
> +        })();
> +
> +        if result.is_err() {
> +            vm.kill();
> +        }
> +
> +        result
> +    }
> +
> +    pub(crate) fn boot(&self) -> Result {
> +        let io = &self.iomem;
> +
> +        // Discard any stale global interrupt.
> +        io.write_reg(JOB_IRQ_CLEAR::zeroed().with_glb(true));
> +
> +        io.write_reg(MCU_CONTROL::zeroed().with_req(McuControlMode::Auto));
> +
> +        if let Err(e) = poll::read_poll_timeout(
> +            || Ok((io.read(MCU_STATUS), io.read(JOB_IRQ_RAWSTAT))),
> +            |(mcu_status, irq_rawstat)| {
> +                mcu_status.value() == McuStatus::Enabled && irq_rawstat.glb()
> +            },
> +            time::Delta::from_millis(1),
> +            time::Delta::from_millis(100),
> +        ) {
> +            let status = io.read(MCU_STATUS);
> +            dev_err!(
> +                self.vm.dev(),
> +                "MCU failed to boot, status: {:?}",
> +                status.value()
> +            );
> +            return Err(e);
> +        }
> +
> +        io.write_reg(JOB_IRQ_CLEAR::zeroed().with_glb(true));
> +
> +        Ok(())
> +    }
> +
> +    fn stop(&self) -> Result {
> +        let io = &self.iomem;
> +        io.write_reg(MCU_CONTROL::zeroed().with_req(McuControlMode::Disable));
> +
> +        if let Err(e) = poll::read_poll_timeout(
> +            || Ok(io.read(MCU_STATUS)),
> +            |status| status.value() == McuStatus::Disabled,
> +            time::Delta::from_micros(10),
> +            time::Delta::from_millis(100),
> +        ) {
> +            let status = io.read(MCU_STATUS);
> +            dev_err!(
> +                self.vm.dev(),
> +                "MCU failed to stop, status: {:?}",
> +                status.value()
> +            );
> +            return Err(e);
> +        }
> +
> +        Ok(())
> +    }
> +}
> diff --git a/drivers/gpu/drm/tyr/gem.rs b/drivers/gpu/drm/tyr/gem.rs
> index b371299028d6..342d3303a0c0 100644
> --- a/drivers/gpu/drm/tyr/gem.rs
> +++ b/drivers/gpu/drm/tyr/gem.rs
> @@ -72,7 +72,6 @@ pub(crate) fn new_dummy_object(ddev: &TyrDrmDevice) -> Result<ARef<Bo>> {
> /// An automatic VA allocation strategy will be added in the future.
> pub(crate) enum KernelBoVaAlloc {
>     /// Explicit VA address specified by the caller.
> -    #[expect(dead_code)]
>     Explicit(u64),
> }
> 
> @@ -98,7 +97,6 @@ impl<'bound> KernelBo<'bound> {
>     /// This function allocates a new shmem-backed GEM object and immediately maps
>     /// it into the specified GPU virtual memory space. The mapping is automatically
>     /// cleaned up when the [`KernelBo`] is dropped.
> -    #[expect(dead_code)]
>     pub(crate) fn new(
>         ddev: &TyrDrmDevice,
>         vm: Arc<Vm<'bound>>,
> @@ -135,7 +133,6 @@ pub(crate) fn new(
>         })
>     }
> 
> -    #[expect(dead_code)]
>     pub(crate) fn bo(&self) -> &Bo {
>         &self.bo
>     }
> diff --git a/drivers/gpu/drm/tyr/tyr.rs b/drivers/gpu/drm/tyr/tyr.rs
> index 92f6885cdaae..e7ec450bdc9c 100644
> --- a/drivers/gpu/drm/tyr/tyr.rs
> +++ b/drivers/gpu/drm/tyr/tyr.rs
> @@ -9,6 +9,7 @@
> 
> mod driver;
> mod file;
> +mod fw;
> mod gem;
> mod gpu;
> mod mmu;
> diff --git a/drivers/gpu/drm/tyr/vm.rs b/drivers/gpu/drm/tyr/vm.rs
> index c113820b5505..418f98ab07fb 100644
> --- a/drivers/gpu/drm/tyr/vm.rs
> +++ b/drivers/gpu/drm/tyr/vm.rs
> @@ -6,7 +6,6 @@
> //! the illusion of owning the entire virtual address (VA) range, similar to CPU virtual memory.
> //! Each virtual memory (VM) area is backed by ARM64 LPAE Stage 1 page tables and can be
> //! mapped into hardware address space (AS) slots for GPU execution.
> -#![expect(dead_code)]
> 
> use core::marker::PhantomData;
> use core::ops::Range;
> 
> -- 
> 2.55.0
>
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.