From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 3BE81449B2A; Wed, 22 Jul 2026 23:54:51 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764492; cv=pass; b=QxnAqKo5lTUm7Wzce2h4rPudg+3tbaVjzZbX92/swBdHCJUBqhQoXO33oYdLr2wlg2swIW5WivdkHKogyPW7op5RcI+IkgCocx8pGbly6N0rZGy742xVDHzvnD31D1W3oMYuCzM0NLjkIRf1OVm3RcCkWd66FbHIsm6s75qXXJM= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764492; c=relaxed/simple; bh=dKzL3X3uBJqRLYIHoO7ah22jAzO/LHpMDY7j+Dyg9ng=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=IjUe2/z6C0CN0jfv7v2HasmWsiREpEBlwXd4aL9k4VXulMS9KB3K1kZFmiTr9nRefZNsIhqEiuiTXJ2exGaodfkBO6QTEovLtSy8KDR/e7jG56LFiCBlgScXQ89qlWIG/t+kDwPVxrzoUjFFJMqfFS04JeE0UyLRgerP7K1novo= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=IahmPlZS; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="IahmPlZS" ARC-Seal: i=1; a=rsa-sha256; t=1784764454; cv=none; d=zohomail.com; s=zohoarc; b=S1MFF1UB7bWMg29OLm1pqKWuIr9lyF605u1DVu5sl7ORejAAqXoO06X3yiUY9rDC8ztgHb3RXG7saCYPzFN0FOavWZhX6MKEQvtwS97MDN1jbDDQokcTywGJtx8BYzoAmxMTcRiPk9Qo670FLNaldzDm04DdAbX9sPxC2uhkbHY= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764454; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=g3nbqIyxkNM73KC2oeZ+6NYURzs14m9OEzaGq9rVgWk=; b=Vf4N7DC6ExcBlrAOXfVVvJHnVBgs9LsEpmliwXk05L023m1IHac39+LgZqKI68lw5vrUn5cw+V7t7VO2urcihYM+IShqN8L3ZEGqAJklmMhG/cHuS2r1b1Z4OH8OzcLRPIo040bYeDKJoouyqL9QLGi9DRBZRPTseDNtRUYZpv0= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764454; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=g3nbqIyxkNM73KC2oeZ+6NYURzs14m9OEzaGq9rVgWk=; b=IahmPlZSAjZvh/4CS2JOs0leJkuloWTZBlXzsP0zT6Ou0PV4QsyzZ1PqOdX7UnGL RAD+tV4C8O+JDDKYZZgvy4Y7VtOHsofgFsi+4vTabYxs0T4TlVnao/8LThy8cmOomlG 8GQOWTj/ZwpiSaSUJ9N6t7zSGLNkhsoNsbQ/fwyo= Received: by mx.zohomail.com with SMTPS id 1784764453593797.4660081619885; Wed, 22 Jul 2026 16:54:13 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:07 -0700 Subject: [PATCH v9 1/7] drm/tyr: add resources to RegistrationData Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-1-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=5789; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=dKzL3X3uBJqRLYIHoO7ah22jAzO/LHpMDY7j+Dyg9ng=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcotHYm1EybO9bNRC/j4USpEvzz8XvGeX1efcwc/l ryxfH5IRykLgxgXg6yYIstZe6Me8ar3Rrrz/zfDzGFlAhnCwMUpABPZx8XI8GrOuTurGlZrvrsy 8ZXlhIdsNeUuMq/m5OWs9629FffMy5vhDw+/f2ng+t3F2k8V/2xWUzWuXWNpVlK2neEa1+7coBP zmQE= X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB Currently Tyr is not storing any resources in its drm::Driver RegistrationData. Move Tyr's device-private resources and gpu information from drm::Driver::Data to drm::Driver::RegistrationData. This allows Tyr to access this data safely within the lifetime of its binding to its parent platform device and while registered with userspace. Reviewed-by: Daniel Almeida Signed-off-by: Deborah Brouwer --- drivers/gpu/drm/tyr/driver.rs | 42 +++++++++++++++++++++------------------= --- drivers/gpu/drm/tyr/file.rs | 11 ++++++----- 2 files changed, 27 insertions(+), 26 deletions(-) diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs index 8348c6cd3929..46ce5c41e310 100644 --- a/drivers/gpu/drm/tyr/driver.rs +++ b/drivers/gpu/drm/tyr/driver.rs @@ -6,6 +6,7 @@ OptionalClk, // }, device::{ + Bound, Core, Device, DeviceContext, // @@ -27,10 +28,7 @@ regulator, regulator::Regulator, sizes::SZ_2M, - sync::{ - aref::ARef, - Mutex, // - }, + sync::Mutex, time, // }; =20 @@ -53,13 +51,17 @@ =20 #[pin_data(PinnedDrop)] pub(crate) struct TyrPlatformDriverData<'bound> { - _device: ARef, _reg: drm::Registration<'bound, TyrDrmDriver>, } =20 +/// Data owned by the DRM [`Registration`]. +/// +/// This data can have references tied to the parent platform device bindi= ng scope +/// and is accessible only while the DRM device is registered with userspa= ce. #[pin_data] -pub(crate) struct TyrDrmDeviceData { - pub(crate) pdev: ARef, +pub(crate) struct TyrDrmRegistrationData<'bound> { + /// Parent platform device. + pub(crate) pdev: &'bound platform::Device, =20 #[pin] clks: Mutex, @@ -67,9 +69,10 @@ pub(crate) struct TyrDrmDeviceData { #[pin] regulators: Mutex, =20 - /// Some information on the GPU. - /// - /// This is mainly queried by userspace, i.e.: Mesa. + /// GPU MMIO register mapping. + pub(crate) iomem: IoMem<'bound>, + + /// GPU information read from hardware during probe. pub(crate) gpu_info: GpuInfo, } =20 @@ -134,10 +137,10 @@ fn probe<'bound>( // other threads of execution. unsafe { pdev.dma_set_mask_and_coherent(DmaMask::try_new(pa_bits)?= )? }; =20 - let platform: ARef =3D pdev.into(); + let unreg_dev =3D drm::UnregisteredDevice::::new(pde= v, Ok(()))?; =20 - let data =3D try_pin_init!(TyrDrmDeviceData { - pdev: platform.clone(), + let reg_data =3D try_pin_init!(TyrDrmRegistrationData { + pdev, clks <- new_mutex!(Clocks { core: core_clk, stacks: stacks_clk, @@ -147,18 +150,15 @@ fn probe<'bound>( _mali: mali_regulator, _sram: sram_regulator, }), + iomem, gpu_info, }); =20 - let tdev =3D drm::UnregisteredDevice::::new(pdev, da= ta)?; // SAFETY: `reg` is stored in `TyrPlatformDriverData` and dropped = when the driver is // unbound; it is never forgotten. - let reg =3D unsafe { drm::Registration::new(pdev.as_ref(), tdev, (= ), 0)? }; + let reg =3D unsafe { drm::Registration::new(pdev.as_ref(), unreg_d= ev, reg_data, 0)? }; =20 - let driver =3D TyrPlatformDriverData { - _device: reg.device().into(), - _reg: reg, - }; + let driver =3D TyrPlatformDriverData { _reg: reg }; =20 // We need this to be dev_info!() because dev_dbg!() does not work= at // all in Rust for now, and we need to see whether probe succeeded. @@ -184,8 +184,8 @@ fn drop(self: Pin<&mut Self>) {} =20 #[vtable] impl drm::Driver for TyrDrmDriver { - type Data =3D TyrDrmDeviceData; - type RegistrationData<'a> =3D (); + type Data =3D (); + type RegistrationData<'bound> =3D TyrDrmRegistrationData<'bound>; type File =3D TyrDrmFileData; type Object =3D drm::gem::shmem::Object; type ParentDevice =3D platform::Device; diff --git a/drivers/gpu/drm/tyr/file.rs b/drivers/gpu/drm/tyr/file.rs index b686041d5d6b..9f60a90d4948 100644 --- a/drivers/gpu/drm/tyr/file.rs +++ b/drivers/gpu/drm/tyr/file.rs @@ -12,7 +12,8 @@ =20 use crate::driver::{ TyrDrmDevice, - TyrDrmDriver, // + TyrDrmDriver, + TyrDrmRegistrationData, // }; =20 #[pin_data] @@ -31,15 +32,15 @@ fn open(_dev: &drm::Device) -> Result>> { =20 impl TyrDrmFileData { pub(crate) fn dev_query( - ddev: &TyrDrmDevice, - _reg_data: &(), + _ddev: &TyrDrmDevice, + reg_data: &TyrDrmRegistrationData<'_>, devquery: &mut uapi::drm_panthor_dev_query, _file: &TyrDrmFile, ) -> Result { if devquery.pointer =3D=3D 0 { match devquery.type_ { uapi::drm_panthor_dev_query_type_DRM_PANTHOR_DEV_QUERY_GPU= _INFO =3D> { - devquery.size =3D core::mem::size_of_val(&ddev.gpu_inf= o) as u32; + devquery.size =3D core::mem::size_of_val(®_data.gpu= _info) as u32; Ok(0) } _ =3D> Err(EINVAL), @@ -53,7 +54,7 @@ pub(crate) fn dev_query( ) .writer(); =20 - writer.write(&ddev.gpu_info)?; + writer.write(®_data.gpu_info)?; =20 Ok(0) } --=20 2.55.0 From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 3BDD14457BA; Wed, 22 Jul 2026 23:54:51 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764493; cv=pass; b=Xb2Lm0CZtyRPfJhYnXWdHy9QraDv6FzLQWeWbek39L1/rREs6ImOny3jF5UYvWdScIOEwWZ5mt7LV/9uwE2JzwciSGVQMYhjrSPGTF4a2cspZLQgq3vLvVzDfd5iHIm5CINwBh+AH9SuoHsjTUbiYzmpqT+NWCv0n0pNkuusjSk= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764493; c=relaxed/simple; bh=terCo1ge57k5W45hRwNhjU57M46rTs/10zADTmy6rcU=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=ZUqc03Z6gd+M726o1uJgNSdv/y6nqFkdBtlXSGiLtqteJ2gBcpPekC2YWq5Yl6BsgvecJwue/xx+uGUpBffsDaLaa6JfFNeit1Cvciqh8GgEQ3dgLfPFUgv/R9vD5Mdx/sv3vyqeO/6bNvsPiJRNRlYxvxfXQx/d2nGgAWaZ3Oc= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=S31pxr+q; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="S31pxr+q" ARC-Seal: i=1; a=rsa-sha256; t=1784764456; cv=none; d=zohomail.com; s=zohoarc; b=DEBaOKLVJZO8xY4oMfIoaF909PXouYS50/tByXOh1jQJ6OzD7e3bINxBwf9Id2X53QrXMlY6RB5pWhOP78PZ5FgQqVDWXrzZ2quAgLXrQqhGiv5ml/AWftzKbcYZlffqOoX7JyQnaaU18D0GMJqXr8ueeujKij08oNVaRZIBz38= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764456; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=nzgvVj2YFg+yuz4duAIJqF9njTFSJ6sRESzHzi+IIuQ=; b=h8VR0NzgsdahsyV8upkzbW2L+M+4XdXn7p7AVNi7fMI2eVhqFvnTvTl+zMZg8D8ct33rVDosRsFnBn+DBSyPV9jOTV363teQrkbNXFgqbPL5munTLekITvZjWpaC373hkxSurLEOBJUTWJrIUrA/+Fums4OeqfF8OKysS76tr5g= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764456; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=nzgvVj2YFg+yuz4duAIJqF9njTFSJ6sRESzHzi+IIuQ=; b=S31pxr+q16Y9tUo+imDFk2TPB53mL4lgNrWIo4MFe8msmnRumekL8FSVofMexqnz FumSl2E25AXhZPTEWAGK+5pZ3QU8mWtQELrMw1qvZ0Em4wNZes94pat4BKGIFC8CNbJ og9PhNtqeIHgMX6vaww0I7GrYQWpnyuMVVpOv1q0= Received: by mx.zohomail.com with SMTPS id 1784764454777406.8023888276125; Wed, 22 Jul 2026 16:54:14 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:08 -0700 Subject: [PATCH v9 2/7] drm/tyr: add a generic slot manager Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-2-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=16923; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=4Y6uqS96E9XnIMI6yVGWCKWEa1lwaF587vv0NGT9XQg=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcp9t8LMbcwlny4xypm0dJX5XHtr3Qv3/6d1Pjxjw X7Picuoo5SFQYyLQVZMkeWsvVGPeNV7I935/5th5rAygQxh4OIUgInoXWP4K7yRS2LfbLvyxfm/ BFo+FTycOdM2qOee0Zo1L89Onrw/gI3hf9Gj/JotJk0Ot3bsW7qp76bV5hS9qlN1Lmqb7M+8Zsr 05AYA X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB From: Boris Brezillon Introduce a generic slot manager to dynamically allocate limited hardware slots to software "seats". It can be used for both address space (AS) and command stream group (CSG) slots. The slot manager initially assigns seats to its free slots. It will continue to reuse the same slot for a seat, as long as another seat does not start to use the slot in the interim. When contention arises because all of the slots are allocated, the slot manager will lazily evict and reuse slots that have become idle (if any). The seat state is protected using the LockedBy pattern with the same lock that guards the SlotManager. This ensures the seat state stays consistent across slot operations. Hardware specific behaviour is controlled through the SlotManager's specific manager type that implements the `SlotOperations` trait. Signed-off-by: Boris Brezillon Co-developed-by: Deborah Brouwer Signed-off-by: Deborah Brouwer Reviewed-by: Daniel Almeida --- drivers/gpu/drm/tyr/slot.rs | 405 ++++++++++++++++++++++++++++++++++++++++= ++++ drivers/gpu/drm/tyr/tyr.rs | 1 + 2 files changed, 406 insertions(+) diff --git a/drivers/gpu/drm/tyr/slot.rs b/drivers/gpu/drm/tyr/slot.rs new file mode 100644 index 000000000000..845846e07ee5 --- /dev/null +++ b/drivers/gpu/drm/tyr/slot.rs @@ -0,0 +1,405 @@ +// SPDX-License-Identifier: GPL-2.0 or MIT + +//! Slot management abstraction for limited hardware resources. +//! +//! This module provides a generic [`SlotManager`] that assigns limited ha= rdware +//! slots to logical "seats". A seat represents an entity (such as a virtu= al memory +//! (VM) address space) that needs access to a hardware slot. +//! +//! The [`SlotManager`] tracks slot allocation using sequence numbers (seq= no) to detect +//! when a seat's binding has been invalidated. When a seat requests activ= ation, +//! the manager will either reuse the seat's existing slot (if still valid= ), +//! allocate a free slot (if any are available), or evict the oldest idle = slot if any +//! slots are idle. +//! +//! Hardware-specific behavior is customized by implementing the [`SlotOpe= rations`] +//! trait, which allows callbacks when slots are activated or evicted. +//! +//! This is currently used for managing address space slots in the GPU, an= d it will +//! also be used to manage Command Stream Group (CSG) interface slots in t= he future. +//! +//! [SlotOperations]: crate::slot::SlotOperations +//! [SlotManager]: crate::slot::SlotManager +#![expect(dead_code)] + +use core::{ + mem, + ops::{ + Deref, + DerefMut, // + }, // +}; + +use kernel::{ + prelude::*, + sync::LockedBy, // +}; + +/// Seat information. +/// +/// This can't be accessed directly by the element embedding a `Seat`, +/// but is used by the generic slot manager logic to control residency +/// of a certain object on a hardware slot. +pub(crate) struct SeatInfo { + /// Slot used by this seat. + /// + /// This index is only valid if the slot pointed to by this index + /// has its `SlotInfo::seqno` match `SeatInfo::seqno`. Otherwise, + /// it means the object has been evicted from the hardware slot, + /// and a new slot needs to be acquired to make this object + /// resident again. + slot: u8, + + /// Sequence number encoding the last time this seat was active. + /// We also use it to check if a slot is still bound to a seat. + seqno: u64, +} + +/// Seat state. +/// +/// This is meant to be embedded in the object that wants to acquire +/// hardware slots. It also starts in the `Seat::NoSeat` state, and +/// the slot manager will change the object value when an active/evict +/// request is issued. +#[derive(Default)] +pub(crate) enum Seat { + #[expect(clippy::enum_variant_names)] + /// Resource is not resident. + /// + /// All objects start with a seat in the `Seat::NoSeat` state. The sea= t also + /// gets back to that state if the user requests eviction. It + /// can also end up in that state next time an operation is done + /// on a `Seat::Idle` seat and the slot manager finds out this + /// object has been evicted from the slot. + #[default] + NoSeat, + + /// Resource is actively used and resident. + /// + /// When a seat is in the `Seat::Active` state, it can't be evicted, a= nd the + /// slot pointed to by `SeatInfo::slot` is guaranteed to be reserved + /// for this object as long as the seat stays active. + Active(SeatInfo), + + /// Resource is idle and might or might not be resident. + /// + /// When a seat is in the`Seat::Idle` state, we can't know for sure if= the + /// object is resident or evicted until the next request we issue + /// to the slot manager. This tells the slot manager it can + /// reclaim the underlying slot if needed. + /// In order for the hardware to use this object again, the seat + /// needs to be turned into an `Seat::Active` state again + /// with a `SlotManager::activate()` call. + Idle(SeatInfo), +} + +impl Seat { + /// Get the slot index this seat is pointing to. + /// + /// If the seat is not `Seat::Active` we can't trust the + /// `SeatInfo`. In that case `None` is returned, otherwise + /// `Some(SeatInfo::slot)` is returned. + pub(crate) fn slot(&self) -> Option { + match self { + Self::Active(info) =3D> Some(info.slot), + _ =3D> None, + } + } +} + +/// Information related to a slot. +struct SlotInfo { + /// Type specific data attached to a slot. + slot_data: D, + + /// Sequence number from when this slot was last activated. + seqno: u64, +} + +/// Slot state. +#[derive(Default)] +enum Slot { + /// Slot is free. + #[default] + Free, + + /// Slot is active. + Active(SlotInfo), + + /// Slot is idle. + Idle(SlotInfo), +} + +pub(crate) type LockedSeat =3D LockedBy>; + +/// Trait describing the slot-related operations. +pub(crate) trait SlotOperations: Sized { + /// Implementation-specific data associated with each slot. + type SlotData; + + /// Returns the seat belonging to this slot data. + fn seat(slot_data: &Self::SlotData) -> &LockedSeat; + + /// Called when a slot is being activated for a seat. + fn activate(&mut self, _slot_idx: usize, _slot_data: &Self::SlotData) = -> Result { + Ok(()) + } + + /// Called when a slot is being evicted and freed. + fn evict(&mut self, _slot_idx: usize, _slot_data: &Self::SlotData) -> = Result { + Ok(()) + } +} + +/// A generic slot manager that provides access to a limited number of har= dware slots. +pub(crate) struct SlotManager, const MAX_SLOT= S: usize> { + /// A specific implementation of the generic slot manager. + manager: T, + + /// Number of slots actually available. + slot_count: usize, + + /// Slot array used to track the state of each slot. + slots: [Slot; MAX_SLOTS], + + /// Sequence number incremented each time a Seat is successfully activ= ated + use_seqno: u64, +} + +impl, const MAX_SLOTS: usize> SlotManager { + /// Creates a specific instance of a slot manager. + pub(crate) fn new(manager: T, slot_count: usize) -> Result { + if slot_count =3D=3D 0 { + return Err(EINVAL); + } + if slot_count > MAX_SLOTS { + return Err(EINVAL); + } + // Since the slot index is stored in SeatInfo as a u8, the maximum= number of slots is 256. + if slot_count > u8::MAX as usize + 1 { + return Err(EINVAL); + } + + Ok(Self { + manager, + slot_count, + slots: [const { Slot::Free }; MAX_SLOTS], + use_seqno: 1, + }) + } + + /// Records a newly activated slot for the given seat. + /// The slot manager takes ownership of the hardware-specific slot dat= a. + fn record_active_slot(&mut self, slot_idx: usize, slot_data: T::SlotDa= ta) { + let cur_seqno =3D self.use_seqno; + + *T::seat(&slot_data).access_mut(self) =3D Seat::Active(SeatInfo { + slot: slot_idx as u8, + seqno: cur_seqno, + }); + + self.slots[slot_idx] =3D Slot::Active(SlotInfo { + slot_data, + seqno: cur_seqno, + }); + + self.use_seqno +=3D 1; + } + + /// Reactivates an active/idle slot for a given seat without reprogram= ming the hardware. + /// The SlotManager reuses the existing slot_data. This ensures that t= he hardware-specific + /// information is not changed between subsequent uses. It also ensure= s that resources + /// owned by the existing slot_data remain alive while the hardware is= configured to use them. + fn reactivate_slot(&mut self, slot_idx: usize, slot_data: &T::SlotData= ) -> Result { + let cur_seqno =3D self.use_seqno; + + let mut slot_info =3D match mem::take(&mut self.slots[slot_idx]) { + Slot::Active(slot_info) | Slot::Idle(slot_info) =3D> slot_info, + Slot::Free =3D> { + *T::seat(slot_data).access_mut(self) =3D Seat::NoSeat; + return Err(EINVAL); + } + }; + + *T::seat(slot_data).access_mut(self) =3D Seat::Active(SeatInfo { + slot: slot_idx as u8, + seqno: cur_seqno, + }); + + slot_info.seqno =3D cur_seqno; + self.slots[slot_idx] =3D Slot::Active(slot_info); + + self.use_seqno +=3D 1; + + Ok(()) + } + + /// Activates a slot for the given seat. + fn activate_slot(&mut self, slot_idx: usize, slot_data: T::SlotData) -= > Result { + self.manager.activate(slot_idx, &slot_data)?; + self.record_active_slot(slot_idx, slot_data); + Ok(()) + } + + /// Finds a slot for the given seat. A free slot is preferred, but if = none + /// are available, the oldest idle slot is evicted and reused. Otherwi= se, if + /// there are no free or idle slots, return [`EBUSY`]. + fn allocate_slot(&mut self, slot_data: T::SlotData) -> Result { + let slots =3D &self.slots[..self.slot_count]; + + let mut idle_slot_idx =3D None; + let mut idle_slot_seqno: u64 =3D 0; + + for (slot_idx, slot) in slots.iter().enumerate() { + match slot { + Slot::Free =3D> { + return self.activate_slot(slot_idx, slot_data); + } + Slot::Idle(slot_info) =3D> { + if idle_slot_idx.is_none() || slot_info.seqno < idle_s= lot_seqno { + idle_slot_idx =3D Some(slot_idx); + idle_slot_seqno =3D slot_info.seqno; + } + } + Slot::Active(_) =3D> (), + } + } + + match idle_slot_idx { + Some(slot_idx) =3D> { + // Lazily evict idle slot just before it is reused. + if let Slot::Idle(slot_info) =3D &self.slots[slot_idx] { + self.manager.evict(slot_idx, &slot_info.slot_data)?; + mem::take(&mut self.slots[slot_idx]); + } + self.activate_slot(slot_idx, slot_data) + } + None =3D> Err(EBUSY), + } + } + + /// Converts an active slot and its seat to idle state. + fn idle_slot(&mut self, slot_idx: usize, locked_seat: &LockedSeat) -> Result { + let slot =3D mem::take(&mut self.slots[slot_idx]); + + self.slots[slot_idx] =3D match slot { + // If the slot was active, make it idle. + Slot::Active(slot_info) =3D> Slot::Idle(slot_info), + + // Preserve an already-idle slot. + Slot::Idle(slot_info) =3D> Slot::Idle(slot_info), + + // A free slot remains free. + Slot::Free =3D> Slot::Free, + }; + + // If the seat was active, make it idle, or keep it idle if it was= already idle. + *locked_seat.access_mut(self) =3D match locked_seat.access(self) { + Seat::Active(seat_info) | Seat::Idle(seat_info) =3D> Seat::Idl= e(SeatInfo { + slot: seat_info.slot, + seqno: seat_info.seqno, + }), + Seat::NoSeat =3D> Seat::NoSeat, + }; + Ok(()) + } + + /// Evicts an active or idle slot: calls the eviction callback and mar= ks the slot as free + /// and the seat as NoSeat. + fn evict_slot(&mut self, slot_idx: usize, locked_seat: &LockedSeat) -> Result { + match &self.slots[slot_idx] { + Slot::Active(slot_info) | Slot::Idle(slot_info) =3D> { + // If hardware eviction fails (e.g. times out), the slot r= etains + // its SlotData so that any resources still referenced by = the hardware + // will remain alive. This prevents use-after-free errors. + self.manager.evict(slot_idx, &slot_info.slot_data)?; + mem::take(&mut self.slots[slot_idx]); + } + _ =3D> (), + } + + *locked_seat.access_mut(self) =3D Seat::NoSeat; + Ok(()) + } + + /// Checks that the seat state matches the slot's state. + /// If they don't match, the seat is stale and is reset to `NoSeat`. + fn check_seat(&mut self, locked_seat: &LockedSeat) { + let (slot_idx, seat_seqno, is_active) =3D match locked_seat.access= (self) { + Seat::Active(seat_info) =3D> (seat_info.slot as usize, seat_in= fo.seqno, true), + Seat::Idle(seat_info) =3D> (seat_info.slot as usize, seat_info= .seqno, false), + _ =3D> return, + }; + + let valid =3D if is_active { + !kernel::warn_on!(!matches!( + &self.slots[slot_idx], + Slot::Active(slot_info) if slot_info.seqno =3D=3D seat_seq= no + )) + } else { + matches!( + &self.slots[slot_idx], + Slot::Idle(slot_info) if slot_info.seqno =3D=3D seat_seqno + ) + }; + + if !valid { + *locked_seat.access_mut(self) =3D Seat::NoSeat; + } + } + + /// Activates a resource on any available/reclaimable slot. + pub(crate) fn activate(&mut self, slot_data: T::SlotData) -> Result { + self.check_seat(T::seat(&slot_data)); + + // Copy out only the slot index so the borrow of slot_data ends he= re. + let slot_idx =3D match T::seat(&slot_data).access(self) { + Seat::Active(seat_info) | Seat::Idle(seat_info) =3D> Some(seat= _info.slot as usize), + Seat::NoSeat =3D> None, + }; + + match slot_idx { + Some(slot_idx) =3D> self.reactivate_slot(slot_idx, &slot_data), + None =3D> self.allocate_slot(slot_data), + } + } + + /// Flag a resource as idle. This method will be used for user VM supp= ort. + #[expect(dead_code)] + pub(crate) fn idle(&mut self, locked_seat: &LockedSeat) = -> Result { + self.check_seat(locked_seat); + if let Seat::Active(seat_info) =3D locked_seat.access(self) { + self.idle_slot(seat_info.slot as usize, locked_seat)?; + } + Ok(()) + } + + /// Evict a resource from its slot. + pub(crate) fn evict(&mut self, locked_seat: &LockedSeat)= -> Result { + self.check_seat(locked_seat); + + match locked_seat.access(self) { + Seat::Active(seat_info) | Seat::Idle(seat_info) =3D> { + let slot_idx =3D seat_info.slot as usize; + self.evict_slot(slot_idx, locked_seat)?; + } + _ =3D> (), + } + + Ok(()) + } +} + +impl, const MAX_SLOTS: usize> Deref for SlotM= anager { + type Target =3D T; + + fn deref(&self) -> &Self::Target { + &self.manager + } +} + +impl, const MAX_SLOTS: usize> DerefMut for Sl= otManager { + fn deref_mut(&mut self) -> &mut Self::Target { + &mut self.manager + } +} diff --git a/drivers/gpu/drm/tyr/tyr.rs b/drivers/gpu/drm/tyr/tyr.rs index 95cda7b0962f..7c9a8063b3b9 100644 --- a/drivers/gpu/drm/tyr/tyr.rs +++ b/drivers/gpu/drm/tyr/tyr.rs @@ -12,6 +12,7 @@ mod gem; mod gpu; mod regs; +mod slot; =20 kernel::module_platform_driver! { type: TyrPlatformDriver, --=20 2.55.0 From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 4ADE443849A; Wed, 22 Jul 2026 23:54:47 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764489; cv=pass; b=LeLc7z3kP0sRkWJzAPYLdl3gfVn7Oqbv/5cGPzdpdFayLibMOqxERTMrhfgZeAcclcA2dJpBiDYmAKdl/9bBrkIHUW7H5v9iKTcB3EbC1ngfGLB97Ta2IkE3UCDn622vrojlXBn1vnXwRlYf2i6OtaS4WlJhg1IxURcckwjUtL4= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764489; c=relaxed/simple; bh=Z/1qfw1FSbwCzIQ1FjMEaAdmeFdzn/gLd4f79sT6VOE=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=AJovSUqpGoBrXVzogQDRqNvHwXOvQynJHTvNc/WBMp2N82m9a568n1faRWU0wNowBa3DEAiNdPwcnQBlSCGeh7jFkAPgsiGJhkJlDBjaOQx61EIBv8Vs/364yyarqioQqf7uMYX8aqN8eoD8LfWvm2kaGftxLJosK2rHS/wBRTc= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=L4QF4YEL; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="L4QF4YEL" ARC-Seal: i=1; a=rsa-sha256; t=1784764458; cv=none; d=zohomail.com; s=zohoarc; b=JSbrznnAh27kG7O62I5YxL31N7dQ8pOT8G6yEWXjrrXVl9dGTEXgwDSZ2kGoTVlkLmHzPCgxXZK2rmrtOPYjmPnOPYywd1fWce6vUHHgCqIyU2b4+T+7bxr/Nn+LbXYMatWL255MOm/dJbno/6KwaWkYkq/vmTfMLLf6t6FY14A= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764458; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=JXjy3AzliaWXLMqw1rM4mDZLonOxf3B54Tx1UjeOCRg=; b=WzDBIBGvNFkcU2IpathjW+h+gY/2NvPP6Dm2Wv7Llun2H7/TSSsh82S+DbZ0rDPZuYNCo+tMNFyU2ddfEIu0kYSI3XrKG4d/CceQsNDkyJVRjhKYT5AjLZldlHc6DgY2UCswTJd3MsVYMI7x2vvOH9SqWHrglPBr8VuaYszwTK8= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764458; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=JXjy3AzliaWXLMqw1rM4mDZLonOxf3B54Tx1UjeOCRg=; b=L4QF4YEL0PpWPA05jKYNdGXSc4laGPojG53iLniVFBsq+uvvxdMGZ0eo4ItzoaoP bDgyuSnJSjHjOB/TskfNx8i7DUtZfexJ1asNWt0NXxy3UzrO6ChVpY7l0EvO/JlykBQ bPguXeaQfq7SJxZ0j0ewUjIPpBYUpw1GlXocEOGI= Received: by mx.zohomail.com with SMTPS id 1784764455838121.78215754400992; Wed, 22 Jul 2026 16:54:15 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:09 -0700 Subject: [PATCH v9 3/7] drm/tyr: add Memory Management Unit (MMU) support Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-3-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=35543; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=bANWr4koNffgdqn8WW+6rcXLNyuah06kKk9Izc+SvJI=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcp1L3b0u9vt/Nmls+LIqQ1+Du1f3PRtpwr7Z90s2 X3q49oTHaUsDGJcDLJiiixn7Y16xKveG+nO/98MM4eVCWQIAxenAExEaR8jw+07+5ccuH32lHik 5Zwc080HVaNKz6w8xvrBw6En4cj/CwYM/yNqr0y66isspSGv8nzr0z1f74hZLKlS2Zgd/+EjZ7h fPB8A X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB From: Boris Brezillon Add Memory Management Unit (MMU) support in Tyr. The MMU module wraps a SlotManager instance to allocate MMU address-space slots for use by virtual memory (VM) address spaces. The MMU's SlotManager uses an AddressSpaceManager to handle the hardware-specific callbacks. For example, the AddressSpaceManager activates and evicts VMs from slots by writing commands to the MMU registers. Add an implementation block for the MMU's MEMATTR register to provide a method for translating the Memory Attribute Indirection Register (MAIR) format from the pagetable configuration to a format understood by the MMU. Create an mmu instance during probe, it will be used by subsequent patches in this series. Wrap the iomem stored in TyrDrmRegistrationData in an Arc. The iomem is stored in the mmu through its AddressSpaceManager. In anticipation of the iomem also being stored in the firmware object, set up shared ownership of the iomem now. Update Kconfig to add the new MMU and IOMMU dependencies required by this MMU module. Signed-off-by: Boris Brezillon Co-developed-by: Deborah Brouwer Signed-off-by: Deborah Brouwer Reviewed-by: Daniel Almeida --- drivers/gpu/drm/tyr/Kconfig | 3 + drivers/gpu/drm/tyr/driver.rs | 13 +- drivers/gpu/drm/tyr/mmu.rs | 121 ++++++++ drivers/gpu/drm/tyr/mmu/address_space.rs | 511 +++++++++++++++++++++++++++= ++++ drivers/gpu/drm/tyr/regs.rs | 135 +++++++- drivers/gpu/drm/tyr/slot.rs | 1 - drivers/gpu/drm/tyr/tyr.rs | 1 + 7 files changed, 780 insertions(+), 5 deletions(-) diff --git a/drivers/gpu/drm/tyr/Kconfig b/drivers/gpu/drm/tyr/Kconfig index 51a68ef8212c..61a2fd6f961a 100644 --- a/drivers/gpu/drm/tyr/Kconfig +++ b/drivers/gpu/drm/tyr/Kconfig @@ -5,9 +5,12 @@ config DRM_TYR depends on DRM=3Dy depends on RUST depends on ARM || ARM64 || COMPILE_TEST + depends on MMU depends on !GENERIC_ATOMIC64 # for IOMMU_IO_PGTABLE_LPAE depends on COMMON_CLK + depends on IOMMU_SUPPORT default n + select IOMMU_IO_PGTABLE_LPAE select RUST_DRM_GEM_SHMEM_HELPER help Rust DRM driver for ARM Mali CSF-based GPUs. diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs index 46ce5c41e310..dbdd1a3f2ddc 100644 --- a/drivers/gpu/drm/tyr/driver.rs +++ b/drivers/gpu/drm/tyr/driver.rs @@ -28,7 +28,10 @@ regulator, regulator::Regulator, sizes::SZ_2M, - sync::Mutex, + sync::{ + Arc, + Mutex, // + }, time, // }; =20 @@ -37,6 +40,7 @@ gem::BoData, gpu, gpu::GpuInfo, + mmu::Mmu, regs::gpu_control::*, // }; =20 @@ -70,7 +74,7 @@ pub(crate) struct TyrDrmRegistrationData<'bound> { regulators: Mutex, =20 /// GPU MMIO register mapping. - pub(crate) iomem: IoMem<'bound>, + pub(crate) iomem: Arc>, =20 /// GPU information read from hardware during probe. pub(crate) gpu_info: GpuInfo, @@ -121,7 +125,8 @@ fn probe<'bound>( let sram_regulator =3D Regulator::::get(pdev.a= s_ref(), c"sram")?; =20 let request =3D pdev.io_request_by_index(0).ok_or(ENODEV)?; - let iomem =3D request.iomap_sized::()?; + + let iomem =3D Arc::new(request.iomap_sized::()?, GFP_KERNEL= )?; =20 issue_soft_reset(pdev.as_ref(), &iomem)?; gpu::l2_power_on(pdev.as_ref(), &iomem)?; @@ -139,6 +144,8 @@ fn probe<'bound>( =20 let unreg_dev =3D drm::UnregisteredDevice::::new(pde= v, Ok(()))?; =20 + let _mmu =3D Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_i= nfo)?; + let reg_data =3D try_pin_init!(TyrDrmRegistrationData { pdev, clks <- new_mutex!(Clocks { diff --git a/drivers/gpu/drm/tyr/mmu.rs b/drivers/gpu/drm/tyr/mmu.rs new file mode 100644 index 000000000000..78fa3f30e3c6 --- /dev/null +++ b/drivers/gpu/drm/tyr/mmu.rs @@ -0,0 +1,121 @@ +// SPDX-License-Identifier: GPL-2.0 or MIT + +//! Memory Management Unit (MMU) module. +//! +//! The GPU MMU provides a limited number of memory address spaces for use= by command streams. +//! The MMU translates virtual addresses to physical addresses and manages= memory configuration +//! and access permissions. +//! +//! This MMU module is essentially a locked wrapper around a [`SlotManager= `] instance. +//! The [`SlotManager`] manages the assignment of virtual address spaces t= o hardware address-space +//! (AS) slots. MMU commands such as updates and flushes are carried out b= y the +//! [`AddressSpaceManager`] which actually writes to the MMU registers. +#![expect(dead_code)] + +use core::ops::Range; + +use kernel::{ + device::{ + Bound, + Device, // + }, + new_mutex, + prelude::*, + sync::{ + Arc, + ArcBorrow, + Mutex, // + }, // +}; + +use crate::{ + driver::IoMem, + gpu::GpuInfo, + mmu::address_space::{ + AddressSpaceManager, + VmAsData, // + }, + regs::{ + gpu_control::AS_PRESENT, + MAX_AS, // + }, + slot::SlotManager, // +}; + +pub(crate) mod address_space; + +pub(crate) type AsSlotManager<'bound> =3D SlotManager, MAX_AS>; + +/// Locked wrapper for carrying out virtual memory (VM) operations on the = MMU. +#[pin_data] +pub(crate) struct Mmu<'bound> { + /// Slot Manager instance used to allocate hardware slots and write to= MMU registers. + #[pin] + pub(crate) as_manager: Mutex>, +} + +impl<'bound> Mmu<'bound> { + /// Create an MMU component for this device. + pub(crate) fn new( + dev: &'bound Device, + iomem: ArcBorrow<'_, IoMem<'bound>>, + gpu_info: &GpuInfo, + ) -> Result>> { + let present =3D AS_PRESENT::from_raw(gpu_info.as_present).present(= ).get(); + let slot_count =3D present.count_ones().try_into()?; + + let address_space_manager =3D AddressSpaceManager::new(dev, iomem.= into(), present)?; + let as_slot_manager =3D + SlotManager::new(address_space_manager, slot_count).inspect_er= r(|e| { + dev_err!( + dev, + "Failed to initialize MMU slot manager with {} slots: = {:?}", + slot_count, + e + ); + })?; + let mmu_init =3D try_pin_init!(Self{ + as_manager <- new_mutex!(as_slot_manager), + }); + Arc::pin_init(mmu_init, GFP_KERNEL) + } + + /// Assign a VM to an AS slot, provide a translation table, + /// and update the MMU to make the VM resident. + pub(crate) fn activate_vm(&self, vm_as_data: ArcBorrow<'_, VmAsData<'b= ound>>) -> Result { + self.as_manager.lock().activate_vm(vm_as_data) + } + + /// Evict a VM from its AS slot and flush the MMU. + pub(crate) fn deactivate_vm(&self, vm_as_data: &VmAsData<'bound>) -> R= esult { + self.as_manager.lock().deactivate_vm(vm_as_data) + } + + /// Flush MMU translation caches after a VM update. + pub(crate) fn flush_vm(&self, vm_as_data: &VmAsData<'bound>) -> Result= { + self.as_manager.lock().flush_vm(vm_as_data) + } + + /// Flags the start of a VM update. + /// + /// If the VM is resident, any GPU access on the memory range being + /// updated will be blocked until `Mmu::end_vm_update()` is called. + /// This guarantees the atomicity of a VM update. + /// If the VM is not resident, this is a NOP. + pub(crate) fn start_vm_update( + &self, + vm_as_data: &VmAsData<'bound>, + region: &Range, + ) -> Result { + self.as_manager.lock().start_vm_update(vm_as_data, region) + } + + /// Flags the end of a VM update. + /// + /// If the VM is resident, this will let GPU accesses on the updated + /// range go through, in case any of them were blocked. + /// If the VM is not resident, this is a NOP. + pub(crate) fn end_vm_update(&self, vm_as_data: &VmAsData<'bound>) -> R= esult { + self.as_manager.lock().end_vm_update(vm_as_data) + } +} diff --git a/drivers/gpu/drm/tyr/mmu/address_space.rs b/drivers/gpu/drm/tyr= /mmu/address_space.rs new file mode 100644 index 000000000000..38ebe5a48722 --- /dev/null +++ b/drivers/gpu/drm/tyr/mmu/address_space.rs @@ -0,0 +1,511 @@ +// SPDX-License-Identifier: GPL-2.0 or MIT + +//! Address space module. +//! +//! This module handles the hardware interaction for MMU operations through +//! MMIO register access. +//! + +use core::ops::Range; + +use kernel::{ + device::{ + Bound, + Device, // + }, // + error::Result, + io::{ + poll, + register::Array, + Io, // + }, + iommu::pgtable::{ + Config, + IoPageTable, + ARM64LPAES1, // + }, + num::Bounded, + prelude::*, + sizes::{ + SZ_2M, + SZ_4K, // + }, + sync::{ + Arc, + ArcBorrow, + LockedBy, // + }, + time::Delta, // +}; + +use crate::{ + driver::IoMem, + mmu::{ + AsSlotManager, + Mmu, // + }, + regs::{ + mmu_control::mmu_as_control, + mmu_control::mmu_as_control::*, + MAX_AS, // + }, + slot::{ + LockedSeat, + Seat, + SlotOperations, // + }, // +}; + +/// Address space configuration values to be written to MMU registers. +#[derive(Clone, Copy)] +struct AddressSpaceConfig { + /// Translation configuration. Configures how the MMU walks the page t= able for this + /// address space. + transcfg: u64, + + /// Translation table base address. The address of the page table. + transtab: u64, + + /// Memory attributes such as cacheability. + memattr: u64, +} + +/// Virtual memory (VM) address space data for use in MMU operations. +#[pin_data] +pub(crate) struct VmAsData<'bound> { + /// This address-space seat tracks this VM's binding to a hardware add= ress space slot. + /// It can only be accessed when holding the `Mmu::as_manager` lock. + as_seat: LockedSeat, MAX_AS>, + + /// Virtual address bits for this address space. + va_bits: u8, + + /// The page table which maps GPU virtual addresses to physical addres= ses for this VM. + #[pin] + pub(crate) page_table: IoPageTable<'bound, ARM64LPAES1>, +} + +impl<'bound> VmAsData<'bound> { + /// Creates VM address space data by initializing all of its fields. + pub(crate) fn new<'a>( + mmu: &'a Mmu<'bound>, + dev: &'bound Device, + va_bits: u32, + pa_bits: u32, + ) -> impl pin_init::PinInit, Error> + 'a { + let pt_config =3D Config { + quirks: 0, + pgsize_bitmap: SZ_4K | SZ_2M, + ias: va_bits, + oas: pa_bits, + coherent_walk: false, + }; + + let page_table_init =3D IoPageTable::new(dev, pt_config); + + try_pin_init!(Self { + as_seat: LockedBy::new(&mmu.as_manager, Seat::NoSeat), + va_bits: va_bits as u8, + page_table <- page_table_init, + }? Error) + } + + /// Computes the hardware configuration for this address space. + fn as_config(&self) -> Result { + let pt =3D &self.page_table; + // The hardware computes the valid input address range as: + // INA_BITS_VALID =3D min(HW_INA_BITS, 55 - INA_BITS) + // To configure our desired va_bits, we solve for INA_BITS: + // INA_BITS =3D 55 - va_bits + // This assumes HW_INA_BITS (hardware capability) >=3D va_bits. + let field =3D 55u64.checked_sub(self.va_bits.into()).ok_or(EINVAL)= ?; + let ina_bits =3D + match mmu_as_control::InaBits::try_from(Bounded::try_new(field= ).ok_or(EINVAL)?)? { + mmu_as_control::InaBits::Reset =3D> return Err(EINVAL), + bits =3D> bits, + }; + + let transcfg =3D mmu_as_control::TRANSCFG::zeroed() + .with_ptw_memattr(mmu_as_control::PtwMemattr::WriteBack) + .with_r_allocate(true) + .with_mode(mmu_as_control::AddressSpaceMode::Aarch64_4K) + .with_ina_bits(ina_bits) + .into_raw(); + + Ok(AddressSpaceConfig { + transcfg, + // SAFETY: The SlotManager holds an `Arc` as SlotDat= a while this + // TTBR is programmed and stores that Arc in the active slot b= efore + // returning. Eviction flushes and disables the slot before re= leasing + // the Arc; if eviction fails, the slot retains it. Therefore = the page + // table cannot be dropped while the GPU is using it. + transtab: unsafe { pt.ttbr() }, + memattr: MEMATTR::from_mair(pt.mair()).into_raw(), + }) + } +} + +/// Coordinates all hardware-level address space operations through MMIO r= egister +/// operations including enabling, disabling, flushing, and updating addre= ss spaces. +pub(crate) struct AddressSpaceManager<'bound> { + /// Parent device used for logging. + dev: &'bound Device, + + /// Memory-mapped I/O region for GPU register access. + iomem: Arc>, + + /// Bitmask of present address space slots from GPU_AS_PRESENT registe= r. + as_present: u32, +} + +impl<'bound> AddressSpaceManager<'bound> { + /// Creates a new address space manager. + /// + /// Initializes the manager with references to the platform device and + /// I/O memory region, along with the bitmask of available AS slots. + pub(super) fn new( + dev: &'bound Device, + iomem: Arc>, + as_present: u32, + ) -> Result> { + if as_present.trailing_ones() !=3D as_present.count_ones() { + dev_err!( + dev, + "Sparse AS_PRESENT mask is unsupported: {:#x}", + as_present + ); + return Err(EINVAL); + } + Ok(Self { + dev, + iomem, + as_present, + }) + } + + /// Validates that an AS slot number is within range and present in ha= rdware. + /// + /// Checks that the slot index is less than [`MAX_AS`] and that + /// the corresponding bit is set in the `as_present` mask read from th= e GPU. + /// + /// Returns [`EINVAL`] if the slot is out of range or not present in h= ardware. + fn validate_as_slot(&self, as_nr: usize) -> Result { + if as_nr >=3D MAX_AS { + dev_err!( + self.dev, + "AS slot {} out of valid range (max {})", + as_nr, + MAX_AS + ); + return Err(EINVAL); + } + + if (self.as_present & (1 << as_nr)) =3D=3D 0 { + dev_err!( + self.dev, + "AS slot {} not present in hardware (AS_PRESENT=3D{:#x})", + as_nr, + self.as_present + ); + return Err(EINVAL); + } + Ok(()) + } + + /// Waits for an AS slot to become ready (not active). + /// + /// Returns an error if polling times out after 10ms or if register ac= cess fails. + fn as_wait_ready(&self, as_nr: usize) -> Result { + let io =3D &*self.iomem; + let op =3D || { + let status_reg =3D STATUS::try_at(as_nr).ok_or(EINVAL)?; + Ok(io.read(status_reg)) + }; + let cond =3D |status: &STATUS| -> bool { !status.active_ext() }; + poll::read_poll_timeout(op, cond, Delta::from_micros(50), Delta::f= rom_millis(10))?; + + Ok(()) + } + + /// Sends a command to an AS slot. + /// + /// Returns an error if waiting for ready times out or if register wri= te fails. + fn as_send_cmd(&mut self, as_nr: usize, cmd: MmuCommand) -> Result { + self.as_wait_ready(as_nr)?; + let io =3D &*self.iomem; + let command_reg =3D COMMAND::try_at(as_nr).ok_or(EINVAL)?; + io.write(command_reg, COMMAND::zeroed().with_command(cmd)); + Ok(()) + } + + /// Sends a command to an AS slot and waits for completion. + /// + /// Returns an error if sending the command fails or if waiting for co= mpletion times out. + fn as_send_cmd_and_wait(&mut self, as_nr: usize, cmd: MmuCommand) -> R= esult { + self.as_send_cmd(as_nr, cmd)?; + self.as_wait_ready(as_nr)?; + Ok(()) + } + + /// Enables an AS slot with the provided configuration. + /// + /// Returns an error if the slot is invalid or if register writes/comm= ands fail. + fn as_enable(&mut self, as_nr: usize, as_config: &AddressSpaceConfig) = -> Result { + self.validate_as_slot(as_nr)?; + + let io =3D &*self.iomem; + + let transtab =3D as_config.transtab; + io.write( + TRANSTAB_LO::try_at(as_nr).ok_or(EINVAL)?, + TRANSTAB_LO::from_raw(transtab as u32), + ); + io.write( + TRANSTAB_HI::try_at(as_nr).ok_or(EINVAL)?, + TRANSTAB_HI::from_raw((transtab >> 32) as u32), + ); + + let transcfg =3D as_config.transcfg; + io.write( + TRANSCFG_LO::try_at(as_nr).ok_or(EINVAL)?, + TRANSCFG_LO::from_raw(transcfg as u32), + ); + io.write( + TRANSCFG_HI::try_at(as_nr).ok_or(EINVAL)?, + TRANSCFG_HI::from_raw((transcfg >> 32) as u32), + ); + + let memattr =3D as_config.memattr; + io.write( + MEMATTR_LO::try_at(as_nr).ok_or(EINVAL)?, + MEMATTR_LO::from_raw(memattr as u32), + ); + io.write( + MEMATTR_HI::try_at(as_nr).ok_or(EINVAL)?, + MEMATTR_HI::from_raw((memattr >> 32) as u32), + ); + + self.as_send_cmd_and_wait(as_nr, MmuCommand::Update)?; + + Ok(()) + } + + /// Disables an AS slot and clears its configuration. + /// + /// Returns an error if the slot is invalid or if register writes/comm= ands fail. + fn as_disable(&mut self, as_nr: usize) -> Result { + self.validate_as_slot(as_nr)?; + + // Flush AS before disabling + self.as_send_cmd_and_wait(as_nr, MmuCommand::FlushMem)?; + + let io =3D &*self.iomem; + + io.write( + TRANSTAB_LO::try_at(as_nr).ok_or(EINVAL)?, + TRANSTAB_LO::from_raw(0), + ); + io.write( + TRANSTAB_HI::try_at(as_nr).ok_or(EINVAL)?, + TRANSTAB_HI::from_raw(0), + ); + + io.write( + MEMATTR_LO::try_at(as_nr).ok_or(EINVAL)?, + MEMATTR_LO::from_raw(0), + ); + io.write( + MEMATTR_HI::try_at(as_nr).ok_or(EINVAL)?, + MEMATTR_HI::from_raw(0), + ); + + let transcfg =3D TRANSCFG::zeroed() + .with_mode(AddressSpaceMode::Unmapped) + .into_raw(); + + io.write( + TRANSCFG_LO::try_at(as_nr).ok_or(EINVAL)?, + TRANSCFG_LO::from_raw(transcfg as u32), + ); + io.write( + TRANSCFG_HI::try_at(as_nr).ok_or(EINVAL)?, + TRANSCFG_HI::from_raw((transcfg >> 32) as u32), + ); + + self.as_send_cmd_and_wait(as_nr, MmuCommand::Update)?; + + Ok(()) + } + + /// Locks a region of the translation tables for an atomic update. + /// + /// Programs the MMU [`LOCKADDR`] register for the given address space= and issues + /// the lock command. The hardware rounds the requested range up to a + /// power-of-two region aligned to its size. + /// + /// Returns an error if the slot is invalid or if register writes/comm= ands fail. + fn as_start_update(&mut self, as_nr: usize, region: &Range) -> Re= sult { + self.validate_as_slot(as_nr)?; + + // Avoid both an empty range and an inverted range. + if region.start >=3D region.end { + return Err(EINVAL); + } + + // The lock operates on full 64-byte cache lines of translation ta= ble entries. + // Since each translation table entry (TTE) is 8 bytes, a cache li= ne has 8 TTEs. + // Since each TTE maps one page, the minimum locked region size wi= ll be 8 pages. + // + // With 4KiB pages (Aarch64_4K mode), the minimum locked region is= 32KiB. + let lock_region_min_size: u64 =3D 4096 * 8; + + // Count the number of trailing zero bits (zeros at the right/leas= t-significant + // end of the binary representation). For a power-of-two value, th= is equals the + // base-2 exponent (e.g., 32 KiB =3D 2^15 =E2=86=92 15). + let lock_region_min_size_log2 =3D lock_region_min_size.trailing_ze= ros() as u8; + + // XOR the first and last addresses to identify which bits differ = between them. + // The highest set bit in the result determines the exponent of th= e smallest + // power-of-two region that can contain both addresses. + // + // Example: + // addr_xor =3D 0x1000 ^ 0x2FFF =3D 0x3FFF + // highest set bit in 0x3FFF is bit 13 + // minimum region size =3D 2^(13 + 1) =3D 16 KiB + let addr_xor =3D region.start ^ (region.end - 1); + let region_size_log2 =3D 64 - addr_xor.leading_zeros() as u8; + + let lock_region_log2 =3D core::cmp::max(region_size_log2, lock_reg= ion_min_size_log2); + + let lock_region_size =3D 1u64.checked_shl(lock_region_log2.into())= .ok_or(EINVAL)?; + // Align the LOCKADDR base address down to the lock region size (1= << lock_region_log2). + // + // The MMU ignores the low lock_region_log2 bits of LOCKADDR base,= so ensure + // they are cleared in software to avoid ambiguity. + // + // Example: + // lock_region_log2 =3D 14 (16 KiB) + // region.start =3D 0x1000 + // lockaddr_base =3D 0x1000 & ~(0x3FFF) =3D 0x0000 + let lockaddr_base =3D region.start & !(lock_region_size - 1); + + // The LOCKADDR size field encodes the lock region size as log2(si= ze) - 1, + // per the hardware definition. For example, a 32 KiB region is en= coded as 14 + // because log2(32 KiB) =3D 15. + let lockaddr_size =3D lock_region_log2 - 1; + + let io =3D &*self.iomem; + + // The LOCKADDR base field stores address bits 63:12, so remove th= e low 12 bits + // before passing this value to the register macro helper. + // These bits are guaranteed to be zero anyway because of the mini= mum + // size of the locked region. + let lockaddr_base_field =3D lockaddr_base >> 12; + let lockaddr_val =3D LOCKADDR::zeroed() + .try_with_size(lockaddr_size)? + .try_with_base(lockaddr_base_field)? + .into_raw(); + + io.write( + LOCKADDR_LO::try_at(as_nr).ok_or(EINVAL)?, + LOCKADDR_LO::from_raw(lockaddr_val as u32), + ); + io.write( + LOCKADDR_HI::try_at(as_nr).ok_or(EINVAL)?, + LOCKADDR_HI::from_raw((lockaddr_val >> 32) as u32), + ); + + self.as_send_cmd_and_wait(as_nr, MmuCommand::Lock) + } + + /// Completes an atomic translation table update. + /// + /// Returns an error if the slot is invalid or if the flush command fa= ils. + fn as_end_update(&mut self, as_nr: usize) -> Result { + self.validate_as_slot(as_nr)?; + self.as_send_cmd_and_wait(as_nr, MmuCommand::FlushPt)?; + Ok(()) + } + + /// Flushes the translation table cache for an AS slot. + /// + /// Returns an error if the slot is invalid or if the flush command fa= ils. + fn as_flush(&mut self, as_nr: usize) -> Result { + self.validate_as_slot(as_nr)?; + self.as_send_cmd_and_wait(as_nr, MmuCommand::FlushPt) + } +} + +impl<'bound> SlotOperations for AddressSpaceManager<'bound> { + /// VM address space data associated with a hardware slot. + type SlotData =3D Arc>; + + fn seat(slot_data: &Self::SlotData) -> &LockedSeat { + &slot_data.as_seat + } + + /// Activates a VM in a hardware slot. + fn activate(&mut self, slot_idx: usize, slot_data: &Self::SlotData) ->= Result { + let as_config =3D slot_data.as_config()?; + self.as_enable(slot_idx, &as_config) + } + + /// Evicts a VM from a hardware slot. + fn evict(&mut self, slot_idx: usize, _slot_data: &Self::SlotData) -> R= esult { + self.as_flush(slot_idx)?; + self.as_disable(slot_idx)?; + Ok(()) + } +} + +impl<'bound> AsSlotManager<'bound> { + /// Locks a region for translation table updates if the VM has an acti= ve slot. + pub(super) fn start_vm_update( + &mut self, + vm_as_data: &VmAsData<'bound>, + region: &Range, + ) -> Result { + let seat =3D vm_as_data.as_seat.access(self); + match seat.slot() { + Some(slot) =3D> { + let as_nr =3D slot as usize; + self.as_start_update(as_nr, region) + } + _ =3D> Ok(()), + } + } + + /// Completes translation table updates and unlocks the region. + pub(super) fn end_vm_update(&mut self, vm_as_data: &VmAsData<'bound>) = -> Result { + let seat =3D vm_as_data.as_seat.access(self); + match seat.slot() { + Some(slot) =3D> { + let as_nr =3D slot as usize; + self.as_end_update(as_nr) + } + _ =3D> Ok(()), + } + } + + /// Flushes the translation table cache if the VM has an active slot. + pub(super) fn flush_vm(&mut self, vm_as_data: &VmAsData<'bound>) -> Re= sult { + let seat =3D vm_as_data.as_seat.access(self); + match seat.slot() { + Some(slot) =3D> { + let as_nr =3D slot as usize; + self.as_flush(as_nr) + } + _ =3D> Ok(()), + } + } + + /// Activates a VM by assigning it to a hardware slot. + pub(super) fn activate_vm(&mut self, vm_as_data: ArcBorrow<'_, VmAsDat= a<'bound>>) -> Result { + self.activate(vm_as_data.into()) + } + + /// Deactivates a VM by evicting it from its hardware slot. + pub(super) fn deactivate_vm(&mut self, vm_as_data: &VmAsData<'bound>) = -> Result { + self.evict(&vm_as_data.as_seat) + } +} diff --git a/drivers/gpu/drm/tyr/regs.rs b/drivers/gpu/drm/tyr/regs.rs index 831357a8ef87..a62724378ced 100644 --- a/drivers/gpu/drm/tyr/regs.rs +++ b/drivers/gpu/drm/tyr/regs.rs @@ -25,7 +25,7 @@ // // Nevertheless, it is useful to have most of them defined, like the C dri= ver // does. -#![allow(dead_code)] +#![expect(dead_code)] =20 /// Combine two 32-bit values into a single 64-bit value. pub(crate) fn join_u64(lo: u32, hi: u32) -> u64 { @@ -45,6 +45,8 @@ pub(crate) fn read_u64_no_tearing(lo_read: impl Fn() -> u= 32, hi_read: impl Fn() } } =20 +pub(crate) use mmu_control::mmu_as_control::MAX_AS; + /// These registers correspond to the GPU_CONTROL register page. /// They are involved in GPU configuration and control. pub(crate) mod gpu_control { @@ -965,6 +967,8 @@ pub(crate) mod mmu_as_control { register, // }; =20 + use pin_init::Zeroable; + /// Maximum number of hardware address space slots. /// The actual number of slots available is usually lower. pub(crate) const MAX_AS: usize =3D 16; @@ -1158,7 +1162,136 @@ fn from(val: MMU_MEMATTR_STAGE1) -> Self { pub(crate) MEMATTR_HI(u32)[MAX_AS, stride =3D STRIDE] @ 0x240c= { 31:0 value; } + } + + impl MEMATTR { + /// Outer cache-policy nibble indicating device memory. + const ARM_MAIR_DEVICE_MEMORY: u8 =3D 0x0; + + /// In the ARM Architecture Reference Manual, the MAIR encodin= g for Normal memory + /// uses the format `0bxxRW` where: + /// - `W` (bit 0) =3D Write-Allocate policy + /// - `R` (bit 1) =3D Read-Allocate policy + /// E.g., `0b0011` would allow both read and write allocatio= n on a cache miss. + /// + /// ARM MAIR Write-Allocate bit (bit 0 of a cache policy nibbl= e). + const ARM_MAIR_WRITE_ALLOCATE: u8 =3D 0x1; + /// ARM MAIR Read-Allocate bit (bit 1 of a cache policy nibble= ). + const ARM_MAIR_READ_ALLOCATE: u8 =3D 0x2; + + /// Write-back policy bit. For cacheable encodings, it is nece= ssary but not + /// sufficient to set bit 2 of the cache policy nibble. Bit 2 = does not + /// definitively determine write back because bit 2 is also se= t in `0b0100` + /// which encodes Normal non-cacheable memory. + const ARM_MAIR_WRITE_BACK_BIT: u8 =3D 0x4; + + /// Complete cache-policy nibble encoding for Normal Non-cache= able memory. + const ARM_MAIR_NON_CACHEABLE: u8 =3D 0x4; + + /// Mask for the inner cache policy nibble in MAIR attribute b= ytes. + const ARM_MAIR_INNER_MASK: u8 =3D 0x0f; + + /// Check if a MAIR attribute byte represents device memory. + /// + /// Device memory (memory-mapped I/O, registers) cannot be cac= hed because + /// reading and writing to this memory may have side effects. + fn is_device_memory(mair_attr: u8) -> bool { + // In AArch64 MAIR, outer nibble only is 0 for device memo= ry. + (mair_attr >> 4) =3D=3D Self::ARM_MAIR_DEVICE_MEMORY + } + + /// Check if normal memory is fully write-back cacheable. + /// + /// ARM MAIR has two cache policy levels (outer [7:4] and inne= r [3:0]). + /// For memory to be truly write-back, BOTH levels must have t= he write-back bit set. + /// If only one level is write-back, treat it as non-cacheable= for GPU purposes. + fn is_writeback_cacheable(mair_attr: u8) -> bool { + let outer =3D mair_attr >> 4; + let inner =3D mair_attr & Self::ARM_MAIR_INNER_MASK; + + outer !=3D Self::ARM_MAIR_NON_CACHEABLE + && inner !=3D Self::ARM_MAIR_NON_CACHEABLE + && (outer & Self::ARM_MAIR_WRITE_BACK_BIT) !=3D 0 + && (inner & Self::ARM_MAIR_WRITE_BACK_BIT) !=3D 0 + } + + // Helper to encode a MEMATTR attribute from its individual fi= elds. + fn encode_attribute( + alloc_w: bool, + alloc_r: bool, + alloc_sel: AllocPolicySelect, + coherency: Coherency, + memory_type: MemoryType, + ) -> MMU_MEMATTR_STAGE1 { + MMU_MEMATTR_STAGE1::zeroed() + .with_alloc_w(alloc_w) + .with_alloc_r(alloc_r) + .with_alloc_sel(alloc_sel) + .with_coherency(coherency) + .with_memory_type(memory_type) + } + + /// Convert one MAIR attribute byte into a MEMATTR attribute. + // TODO: Add a `coherent` parameter like panthor's mair_to_mem= attr(). + // For now, assume a non-coherent system and always encode wri= te-back + // memory with MidgardInnerDomain coherency. + fn attribute_from_mair(mair_attr: u8) -> MMU_MEMATTR_STAGE1 { + // Device memory or non-write-back normal memory + if Self::is_device_memory(mair_attr) || !Self::is_writebac= k_cacheable(mair_attr) { + return Self::encode_attribute( + false, + false, + AllocPolicySelect::Alloc, + Coherency::MidgardInnerDomain, + MemoryType::NonCacheable, + ); + } + + // Write-back cacheable normal memory + let inner: u8 =3D mair_attr & Self::ARM_MAIR_INNER_MASK; + Self::encode_attribute( + (inner & Self::ARM_MAIR_WRITE_ALLOCATE) !=3D 0, + (inner & Self::ARM_MAIR_READ_ALLOCATE) !=3D 0, + AllocPolicySelect::Alloc, + Coherency::MidgardInnerDomain, + MemoryType::WriteBack, + ) + } + + /// Write one converted MAIR attribute into a corresponding ME= MATTR slot. + fn with_encoded_attribute(self, index: usize, attr: MMU_MEMATT= R_STAGE1) -> Self { + debug_assert!(index < 8); + + let shift =3D index * 8; + let mask =3D !(0xffu64 << shift); + let raw =3D (self.into_raw() & mask) | ((u64::from(attr.in= to_raw())) << shift); + + Self::from_raw(raw) + } + + /// Convert an AArch64 MAIR value into the GPU MEMATTR registe= r encoding. + /// + /// Both MAIR and MEMATTR are 64-bit values with eight 8-bit m= emory + /// attribute entries, but the bits do not map directly. The G= PU MEMATTR encoding + /// is less detailed than the MAIR encoding, so MAIR is conve= rted to MEMATTR + /// conservatively as follows: + /// + /// 1. Device memory, or Normal Memory that is not write-back = cacheable, is encoded + /// as GPU `NonCacheable` + /// + /// 2. Normal memory that is write-back cacheable is encoded a= s GPU `WriteBack`, + /// and the inner allocation hints are preserved. + pub(crate) fn from_mair(mair: u64) -> Self { + mair.to_le_bytes() + .into_iter() + .enumerate() + .fold(Self::zeroed(), |acc, (i, attr)| { + acc.with_encoded_attribute(i, Self::attribute_from= _mair(attr)) + }) + } + } =20 + register! { /// Lock region address for each address space. pub(crate) LOCKADDR(u64)[MAX_AS, stride =3D STRIDE] @ 0x2410 { /// Lock region size. diff --git a/drivers/gpu/drm/tyr/slot.rs b/drivers/gpu/drm/tyr/slot.rs index 845846e07ee5..d194ead53f71 100644 --- a/drivers/gpu/drm/tyr/slot.rs +++ b/drivers/gpu/drm/tyr/slot.rs @@ -20,7 +20,6 @@ //! //! [SlotOperations]: crate::slot::SlotOperations //! [SlotManager]: crate::slot::SlotManager -#![expect(dead_code)] =20 use core::{ mem, diff --git a/drivers/gpu/drm/tyr/tyr.rs b/drivers/gpu/drm/tyr/tyr.rs index 7c9a8063b3b9..79045d0135a8 100644 --- a/drivers/gpu/drm/tyr/tyr.rs +++ b/drivers/gpu/drm/tyr/tyr.rs @@ -11,6 +11,7 @@ mod file; mod gem; mod gpu; +mod mmu; mod regs; mod slot; =20 --=20 2.55.0 From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id AD52344AB67; Wed, 22 Jul 2026 23:54:48 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764491; cv=pass; b=DsC2Xl9KpXB7Zez9gAargFS6BZxymvTwInCWWw357gMeS43vbUqZ7EhesfVFB5jKDmO9Fj1bc+rgVf7HRHFUrrmNhsjAePghy3Y3gRRcu20LCjarE1wAbQISetkplE+yW3/T9bbtr51ibwJZSm2pRQbDPA/bRdfQJIHtMTOSd24= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764491; c=relaxed/simple; bh=fbdSO1J12+xpAVjGBQqqwiWJv+j+6X40NnxF1Lxk75c=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=OIIXa+T+h5UZPM4rPVfSXc0dW/352KupqjMOqLYYZ/mcr57r84xQgwAjV2DyYjUN/3WrT9m2DvgdUe04iLmXEr2hp7WsJ0OyqgdKIDfgPKdxBLxcw1JxVA4Ih2iRwJdqR8fzKt5XMtCYUsS6LP/sHEUoVMyQIL9zg6O6H8Y3/LE= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=KqVTJNku; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="KqVTJNku" ARC-Seal: i=1; a=rsa-sha256; t=1784764458; cv=none; d=zohomail.com; s=zohoarc; b=iAyCRgH01OLuQf/YOfnHDaWDHoSAALBOsTg6T2MxnLiLwVDPQbzrj1eQP67asUfH5YCDXBskcPYNUhcahXNuc8d9ryFLbC8oaC9BR3NOjxZcmQ/Llq3Eh8cIZ2YAJbpieDT00ncwgf3Pmb/HiLIIPFGvnB22n37Ctuz4NXqJmPo= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764458; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=k0lPD1RQOl0hTPFH27hggXpFM5p3KZht4n1iOyVLNZE=; b=AqSJY/rULq21iGlPr2vgun2fOoL+LD8GlU9SbQmr/lc3KXpGsz/IixiGx50xklESEXcmwMbgigEchKdZj8Uf4bdapWyjO+AbWefhPyp7kp6TrPuSAV0yGMsvzZdPMLEeCNmRIvfwUMNFj4aaig1ZhEMr+1jLtAEQIO8uPPXwkw8= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764458; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=k0lPD1RQOl0hTPFH27hggXpFM5p3KZht4n1iOyVLNZE=; b=KqVTJNku9OyuLcQVZ/qgreTIwRWkFhQ66RVq3cAOigYDlSQXYxWx5xl9pO1/+YCX wE2wlYMoKmqdDuM1lZgIkb9rRIAozsJPWpnIv3PWS5T4T1Tkj5Q1B2zCewEeiQ4hf7H SaHXvo098lMpn8tooTFJMRkb2k9BUU8knWsKlw84= Received: by mx.zohomail.com with SMTPS id 1784764457057949.1624237758839; Wed, 22 Jul 2026 16:54:17 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:10 -0700 Subject: [PATCH v9 4/7] drm/tyr: add GPU virtual memory (VM) support Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-4-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=37020; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=zCx6XDRDt12wqkGtir1oaBpqRBVBm/up8njE93axono=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcrfL/S97n4Z1SVYFxSRm8UVU+jnym55nCkw7FHYh g97pxV1lLIwiHExyIopspy1N+oRr3pvpDv/fzPMHFYmkCEMXJwCMBEJeUaGRTL3/llfW6EwWc0z /f70K6k5a2I1WVcLvytao59zVcv5HyPDu9o5aV1ZYut4vzU92BvPUVQzpyJm81/Gt70XTxnpnD7 LBgA= X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB From: Boris Brezillon Add GPU virtual address space management using the DRM GPUVM framework. Each virtual memory (VM) space is backed by ARM64 LPAE Stage 1 page tables and can be mapped into hardware address space (AS) slots for GPU execution. The implementation provides memory isolation and virtual address allocation. VMs support mapping GEM buffer objects with configurable protection flags (readonly, noexec, uncached) and handle both 4KB and 2MB page sizes. A new_dummy_object() helper is provided to create a dummy GEM object for use as a GPUVM root. The vm module integrates with the MMU for address space activation and provides map/unmap/remap operations with page table synchronization. Signed-off-by: Boris Brezillon Co-developed-by: Daniel Almeida Signed-off-by: Daniel Almeida Co-developed-by: Deborah Brouwer Signed-off-by: Deborah Brouwer Reviewed-by: Daniel Almeida --- drivers/gpu/drm/tyr/Kconfig | 1 + drivers/gpu/drm/tyr/driver.rs | 4 +- drivers/gpu/drm/tyr/gem.rs | 26 +- drivers/gpu/drm/tyr/mmu.rs | 1 - drivers/gpu/drm/tyr/tyr.rs | 1 + drivers/gpu/drm/tyr/vm.rs | 938 ++++++++++++++++++++++++++++++++++++++= ++++ 6 files changed, 966 insertions(+), 5 deletions(-) diff --git a/drivers/gpu/drm/tyr/Kconfig b/drivers/gpu/drm/tyr/Kconfig index 61a2fd6f961a..79ea4bb214de 100644 --- a/drivers/gpu/drm/tyr/Kconfig +++ b/drivers/gpu/drm/tyr/Kconfig @@ -12,6 +12,7 @@ config DRM_TYR default n select IOMMU_IO_PGTABLE_LPAE select RUST_DRM_GEM_SHMEM_HELPER + select RUST_DRM_GPUVM help Rust DRM driver for ARM Mali CSF-based GPUs. =20 diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs index dbdd1a3f2ddc..b6528d8cd3ce 100644 --- a/drivers/gpu/drm/tyr/driver.rs +++ b/drivers/gpu/drm/tyr/driver.rs @@ -37,7 +37,7 @@ =20 use crate::{ file::TyrDrmFileData, - gem::BoData, + gem::Bo, gpu, gpu::GpuInfo, mmu::Mmu, @@ -194,7 +194,7 @@ impl drm::Driver for TyrDrmDriver { type Data =3D (); type RegistrationData<'bound> =3D TyrDrmRegistrationData<'bound>; type File =3D TyrDrmFileData; - type Object =3D drm::gem::shmem::Object; + type Object =3D Bo; type ParentDevice =3D platform::Device; =20 const INFO: drm::DriverInfo =3D INFO; diff --git a/drivers/gpu/drm/tyr/gem.rs b/drivers/gpu/drm/tyr/gem.rs index 1640a161754b..c28be61a01bb 100644 --- a/drivers/gpu/drm/tyr/gem.rs +++ b/drivers/gpu/drm/tyr/gem.rs @@ -5,8 +5,12 @@ //! DRM's GEM subsystem with shmem backing. =20 use kernel::{ - drm::gem, - prelude::*, // + drm::gem::{ + self, + shmem, // + }, + prelude::*, + sync::aref::ARef, // }; =20 use crate::driver::{ @@ -34,3 +38,21 @@ fn new(_dev: &TyrDrmDevice, _size: usize, args: BoCreate= Args) -> impl PinInit; + +/// Creates a dummy GEM object to serve as the root of a GPUVM. +pub(crate) fn new_dummy_object(ddev: &TyrDrmDevice) -> Result> { + let bo =3D Bo::new( + ddev, + 4096, + shmem::ObjectConfig { + map_wc: true, + parent_resv_obj: None, + }, + BoCreateArgs { flags: 0 }, + )?; + + Ok(bo) +} diff --git a/drivers/gpu/drm/tyr/mmu.rs b/drivers/gpu/drm/tyr/mmu.rs index 78fa3f30e3c6..8f597af6436d 100644 --- a/drivers/gpu/drm/tyr/mmu.rs +++ b/drivers/gpu/drm/tyr/mmu.rs @@ -10,7 +10,6 @@ //! The [`SlotManager`] manages the assignment of virtual address spaces t= o hardware address-space //! (AS) slots. MMU commands such as updates and flushes are carried out b= y the //! [`AddressSpaceManager`] which actually writes to the MMU registers. -#![expect(dead_code)] =20 use core::ops::Range; =20 diff --git a/drivers/gpu/drm/tyr/tyr.rs b/drivers/gpu/drm/tyr/tyr.rs index 79045d0135a8..92f6885cdaae 100644 --- a/drivers/gpu/drm/tyr/tyr.rs +++ b/drivers/gpu/drm/tyr/tyr.rs @@ -14,6 +14,7 @@ mod mmu; mod regs; mod slot; +mod vm; =20 kernel::module_platform_driver! { type: TyrPlatformDriver, diff --git a/drivers/gpu/drm/tyr/vm.rs b/drivers/gpu/drm/tyr/vm.rs new file mode 100644 index 000000000000..c113820b5505 --- /dev/null +++ b/drivers/gpu/drm/tyr/vm.rs @@ -0,0 +1,938 @@ +// SPDX-License-Identifier: GPL-2.0 or MIT + +//! GPU virtual memory management using the DRM GPUVM framework. +//! +//! This module manages GPU virtual address spaces, providing memory isola= tion and +//! the illusion of owning the entire virtual address (VA) range, similar = to CPU virtual memory. +//! Each virtual memory (VM) area is backed by ARM64 LPAE Stage 1 page tab= les and can be +//! mapped into hardware address space (AS) slots for GPU execution. +#![expect(dead_code)] + +use core::marker::PhantomData; +use core::ops::Range; + +use kernel::{ + device::{ + Bound, + Device, // + }, + drm::{ + gem::BaseObject, + gpuvm::{ + DriverGpuVm, + GpuVaAlloc, + GpuVm, + GpuVmBo, + OpMap, + OpMapRequest, + OpMapped, + OpRemap, + OpRemapped, + OpUnmap, + OpUnmapped, + UniqueRefGpuVm, // + }, // + }, + fmt, + impl_flags, + iommu::pgtable::{ + prot, + IoPageTable, + ARM64LPAES1, // + }, + new_mutex, + prelude::*, + sizes::{ + SZ_1G, + SZ_2M, + SZ_4K, // + }, + sync::{ + aref::ARef, + Arc, + ArcBorrow, + Mutex, // + }, + uapi, // +}; + +use crate::{ + driver::{ + TyrDrmDevice, + TyrDrmDriver, // + }, + gem, + gem::Bo, + gpu::GpuInfo, + mmu::{ + address_space::VmAsData, + Mmu, // + }, + regs::gpu_control::MMU_FEATURES, +}; + +impl_flags!( + /// Flags controlling virtual memory mapping behavior. + /// + /// These flags control access permissions and caching behavior for GP= U virtual + /// memory mappings. + #[derive(Debug, Clone, Default, Copy, PartialEq, Eq)] + pub(crate) struct VmMapFlags(u32); + + /// Individual flags that can be combined in [`VmMapFlags`]. + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + pub(crate) enum VmFlag { + /// Map as read-only. + Readonly =3D uapi::drm_panthor_vm_bind_op_flags_DRM_PANTHOR_VM_BIN= D_OP_MAP_READONLY as u32, + /// Map as non-executable. + Noexec =3D uapi::drm_panthor_vm_bind_op_flags_DRM_PANTHOR_VM_BIND_= OP_MAP_NOEXEC as u32, + /// Map as uncached. + Uncached =3D uapi::drm_panthor_vm_bind_op_flags_DRM_PANTHOR_VM_BIN= D_OP_MAP_UNCACHED as u32, + } +); + +impl VmMapFlags { + /// Convert the flags to `pgtable::prot`. + fn to_prot(self) -> u32 { + let mut prot =3D 0; + + if self.contains(VmFlag::Readonly) { + prot |=3D prot::READ; + } else { + prot |=3D prot::READ | prot::WRITE; + } + + if self.contains(VmFlag::Noexec) { + prot |=3D prot::NOEXEC; + } + + if !self.contains(VmFlag::Uncached) { + prot |=3D prot::CACHE; + } + + prot + } +} + +impl fmt::Display for VmMapFlags { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let mut first =3D true; + + if self.contains(VmFlag::Readonly) { + write!(f, "READONLY")?; + first =3D false; + } + if self.contains(VmFlag::Noexec) { + if !first { + write!(f, " | ")?; + } + write!(f, "NOEXEC")?; + first =3D false; + } + + if self.contains(VmFlag::Uncached) { + if !first { + write!(f, " | ")?; + } + write!(f, "UNCACHED")?; + } + + Ok(()) + } +} + +impl TryFrom for VmMapFlags { + type Error =3D Error; + + fn try_from(value: u32) -> Result { + let valid =3D VmFlag::Readonly as u32 | VmFlag::Noexec as u32 | Vm= Flag::Uncached as u32; + + if value & !valid !=3D 0 { + return Err(EINVAL); + } + Ok(Self(value)) + } +} + +/// Arguments for a virtual memory map operation. +struct VmMapArgs<'bound> { + /// Access permissions and caching behavior for the mapping. + flags: VmMapFlags, + /// GEM buffer object registered with the GPUVM framework. + vm_bo: ARef>>, + /// Offset in bytes from the start of the buffer object. + bo_offset: u64, +} + +/// Type of virtual memory operation. +enum VmOpType<'bound> { + /// Map a GEM buffer object into the virtual address space. + Map(VmMapArgs<'bound>), + /// Unmap a region from the virtual address space. + Unmap, +} + +/// Preallocated resources needed to execute a VM operation. +/// +/// VM operations may require allocating new GPUVA objects to track mappin= gs. +/// To avoid allocation failures during the operation, preallocate the +/// maximum number of GPUVAs that might be needed. +struct VmOpResources<'bound> { + /// Preallocated GPUVA objects for remap operations. + /// + /// Partial unmap requests or map requests overlapping existing mappin= gs + /// will trigger a remap call, which needs to register up to three VA + /// objects (one for the new mapping, and two for the previous and next + /// mappings). + preallocated_gpuvas: [Option>>; 3], +} + +/// Request to execute a virtual memory operation. +struct VmOpRequest<'bound> { + /// Request type. + op_type: VmOpType<'bound>, + + /// Region of the virtual address space covered by this request. + region: Range, +} + +/// Arguments for a page table map operation. +struct PtMapArgs { + /// Memory protection flags describing allowed accesses for this mappi= ng. + /// + /// This is directly derived from [`VmMapFlags`] via [`VmMapFlags::to_= prot`]. + prot: u32, +} + +/// Type of page table operation. +enum PtOpType { + /// Map pages into the page table. + Map(PtMapArgs), + /// Unmap pages from the page table. + Unmap, +} + +/// Context for updating the GPU page table. +/// +/// This context is created when beginning a page table update operation a= nd +/// automatically flushes changes when dropped. It ensures that the +/// Memory Management Unit (MMU) state is properly managed and Translation +/// Lookaside Buffer (TLB) entries are flushed. +pub(crate) struct PtUpdateContext<'ctx, 'bound> { + /// Device used for DMA-mapping GEM shmem SG tables. + dev: &'ctx Device, + + /// Page table. + pt: &'ctx IoPageTable<'bound, ARM64LPAES1>, + + /// MMU manager. + mmu: &'ctx Mmu<'bound>, + + /// Reference to the address space data to pass to the MMU functions. + as_data: &'ctx VmAsData<'bound>, + + /// Region of the virtual address space covered by this request. + region: Range, + + /// Operation type. + op_type: PtOpType, + + /// Preallocated resources that can be used when executing the request. + resources: &'ctx mut VmOpResources<'bound>, +} + +impl<'ctx, 'bound> PtUpdateContext<'ctx, 'bound> { + /// Creates a new page table update context. + /// + /// This prepares the MMU for a page table update. + /// The context will automatically flush the TLB and + /// complete the update when dropped. + fn new( + dev: &'ctx Device, + pt: &'ctx IoPageTable<'bound, ARM64LPAES1>, + mmu: &'ctx Mmu<'bound>, + as_data: &'ctx VmAsData<'bound>, + region: Range, + op_type: PtOpType, + resources: &'ctx mut VmOpResources<'bound>, + ) -> Result> { + mmu.start_vm_update(as_data, ®ion)?; + + Ok(Self { + dev, + pt, + mmu, + as_data, + region, + op_type, + resources, + }) + } + + /// Finds one of our pre-allocated VAs. + fn preallocated_gpuva(&mut self) -> Result>> { + self.resources + .preallocated_gpuvas + .iter_mut() + .find_map(|f| f.take()) + .ok_or(EINVAL) + } + + /// Returns an unused GPUVA object to the preallocated pool. + /// If the pool is already full, the unused allocation is simply dropp= ed. + fn return_preallocated_gpuva(&mut self, gpuva: GpuVaAlloc>) { + if let Some(slot) =3D self + .resources + .preallocated_gpuvas + .iter_mut() + .find(|slot| slot.is_none()) + { + *slot =3D Some(gpuva); + } + } +} + +impl Drop for PtUpdateContext<'_, '_> { + fn drop(&mut self) { + if let Err(e) =3D self.mmu.end_vm_update(self.as_data) { + dev_err!(self.dev, "Failed to end VM update {:?}", e); + } + + if let Err(e) =3D self.mmu.flush_vm(self.as_data) { + dev_err!(self.dev, "Failed to flush VM {:?}", e); + } + } +} + +/// Driver implementation for the GPUVM framework. +/// +/// Implements [`DriverGpuVm`] to provide VM operation callbacks (map, unm= ap, remap) +/// and associated types for buffer objects, virtual addresses, and contex= ts. +pub(crate) struct GpuVmData<'bound> { + _phantom: PhantomData<&'bound ()>, +} + +/// GPU virtual address space. +/// +/// Each VM can be mapped into a hardware address space slot. +#[pin_data] +pub(crate) struct Vm<'bound> { + /// Data referenced by an AS when the VM is active + as_data: Arc>, + /// MMU manager. + mmu: Arc>, + /// Parent device used for DMA mapping and page-table operations. + dev: &'bound Device, + /// DRM GPUVM core for managing virtual address space. + #[pin] + gpuvm_unique: Mutex>>, + /// Non-core part of the GPUVM. Can be used for stuff that doesn't mod= ify the + /// internal mapping tree, like GpuVm::obtain() + gpuvm: ARef>>, + /// VA range for this VM. + va_range: Range, +} + +impl<'bound> Vm<'bound> { + /// Creates a new GPU virtual address space. + /// + /// The VM is initialized with a page table configured according to th= e GPU's + /// address translation capabilities and registered with the GPUVM fra= mework. + pub(crate) fn new( + dev: &'bound Device, + ddev: &TyrDrmDevice, + mmu: ArcBorrow<'_, Mmu<'bound>>, + gpu_info: &GpuInfo, + ) -> Result>> { + let mmu_features =3D MMU_FEATURES::from_raw(gpu_info.mmu_features); + let va_bits =3D mmu_features.va_bits().get(); + let pa_bits =3D mmu_features.pa_bits().get(); + + let range =3D 0..(1u64 << va_bits); + let reserve_range =3D 0..0u64; + + // dummy_obj is used to initialize the GPUVM tree. + let dummy_obj =3D gem::new_dummy_object(ddev).inspect_err(|e| { + dev_err!(dev, "Failed to create dummy GEM object: {:?}", e); + })?; + + let gpuvm_unique =3D GpuVm::new::( + c"Tyr::GpuVm", + ddev, + &*dummy_obj, + range.clone(), + reserve_range, + GpuVmData::<'bound> { + _phantom: PhantomData::<&()>, + }, + ) + .inspect_err(|e| { + dev_err!(dev, "Failed to create GpuVm: {:?}", e); + })?; + let gpuvm =3D ARef::from(&*gpuvm_unique); + + let as_data =3D Arc::pin_init(VmAsData::new(&mmu, dev, va_bits, pa= _bits), GFP_KERNEL)?; + + let vm =3D Arc::pin_init( + pin_init!(Self{ + as_data, + dev, + mmu: mmu.into(), + gpuvm, + gpuvm_unique <- new_mutex!(gpuvm_unique), + va_range: range, + }), + GFP_KERNEL, + )?; + + Ok(vm) + } + + /// Returns the parent device used by this VM for DMA mapping and page= -table operations. + pub(crate) fn dev(&self) -> &'bound Device { + self.dev + } + + /// Activate the VM in a hardware address space slot. + pub(crate) fn activate(&self) -> Result { + self.mmu + .activate_vm(self.as_data.as_arc_borrow()) + .inspect_err(|e| { + dev_err!(self.dev, "Failed to activate VM: {:?}", e); + }) + } + + /// Deactivate the VM by evicting it from its address space slot. + fn deactivate(&self) -> Result { + self.mmu.deactivate_vm(&self.as_data).inspect_err(|e| { + dev_err!(self.dev, "Failed to deactivate VM: {:?}", e); + }) + } + + /// Kills the VM by deactivating it and unmapping all regions. + pub(crate) fn kill(&self) { + // TODO: Turn the VM into a state where it can't be used. + let _ =3D self.deactivate(); + let _ =3D self + .unmap_range(self.va_range.start, self.va_range.end - self.va_= range.start) + .inspect_err(|e| { + dev_err!(self.dev, "Failed to unmap range during deactivat= e: {:?}", e); + }); + } + + /// Executes a virtual memory operation. + /// + /// This handles both map and unmap operations by coordinating between= the + /// GPUVM framework and the hardware page table. + fn exec_op<'a>( + &self, + gpuvm_unique: &mut UniqueRefGpuVm>, + req: VmOpRequest<'bound>, + resources: &'a mut VmOpResources<'bound>, + ) -> Result { + let pt =3D &self.as_data.page_table; + + match req.op_type { + VmOpType::Map(args) =3D> { + let mut pt_upd =3D PtUpdateContext::new( + self.dev, + pt, + &self.mmu, + &self.as_data, + req.region, + PtOpType::Map(PtMapArgs { + prot: args.flags.to_prot(), + }), + resources, + )?; + + gpuvm_unique.sm_map(OpMapRequest { + addr: pt_upd.region.start, + range: pt_upd.region.end - pt_upd.region.start, + gem_offset: args.bo_offset, + vm_bo: &args.vm_bo, + context: &mut pt_upd, + }) + //PtUpdateContext drops here flushing the page table + } + VmOpType::Unmap =3D> { + let mut pt_upd =3D PtUpdateContext::new( + self.dev, + pt, + &self.mmu, + &self.as_data, + req.region, + PtOpType::Unmap, + resources, + )?; + + gpuvm_unique.sm_unmap( + pt_upd.region.start, + pt_upd.region.end - pt_upd.region.start, + &mut pt_upd, + ) + //PtUpdateContext drops here flushing the page table + } + } + } + + /// Maps a GEM buffer object range into the VM at the specified virtua= l address. + /// + /// This creates a mapping from GPU virtual address `va` to the physic= al pages + /// backing the GEM object, starting at `bo_offset` bytes into the obj= ect and + /// spanning `map_size` bytes. The mapping respects the access permiss= ions and + /// caching behavior specified in `flags`. + pub(crate) fn map_bo_range( + &self, + bo: &Bo, + bo_offset: u64, + map_size: u64, + va: u64, + flags: VmMapFlags, + ) -> Result { + let bo_size =3D u64::try_from(bo.size()).map_err(|_| EOVERFLOW)?; + let bo_end =3D bo_offset.checked_add(map_size).ok_or(EINVAL)?; + + if bo_end > bo_size { + dev_err!( + self.dev, + "BO mapping range {:#x}..{:#x} exceeds BO size {:#x}", + bo_offset, + bo_end, + bo_size + ); + return Err(EINVAL); + } + + let va_end: u64 =3D va.checked_add(map_size).ok_or(EINVAL)?; + + let req =3D VmOpRequest { + op_type: VmOpType::Map(VmMapArgs { + vm_bo: self.gpuvm.obtain(bo, ())?, + flags, + bo_offset, + }), + region: va..va_end, + }; + let mut resources =3D VmOpResources { + preallocated_gpuvas: [ + Some(GpuVaAlloc::>::new(GFP_KERNEL)?), + Some(GpuVaAlloc::>::new(GFP_KERNEL)?), + Some(GpuVaAlloc::>::new(GFP_KERNEL)?), + ], + }; + let result =3D { + let mut gpuvm_unique =3D self.gpuvm_unique.lock(); + self.exec_op(gpuvm_unique.as_mut().get_mut(), req, &mut resour= ces) + }; + // We flush the defer cleanup list now. Things will be different in + // the asynchronous VM_BIND path, where we want the cleanup to + // happen outside the DMA signalling path. + self.gpuvm.deferred_cleanup(); + result + } + + /// Unmaps a virtual address range from the VM. + /// + /// This removes any existing mappings in the specified range, freeing= the + /// virtual address space for reuse. + pub(crate) fn unmap_range(&self, va: u64, size: u64) -> Result { + let end =3D va.checked_add(size).ok_or(EINVAL)?; + + if va < self.va_range.start || end > self.va_range.end { + dev_err!( + self.dev, + "Unmap range {:#x}..{:#x} exceeds VM range {:#x}..{:#x}", + va, + end, + self.va_range.start, + self.va_range.end + ); + return Err(EINVAL); + } + + let req =3D VmOpRequest { + op_type: VmOpType::Unmap, + region: va..end, + }; + + let full_vm =3D va =3D=3D self.va_range.start && end =3D=3D self.v= a_range.end; + + let mut resources =3D VmOpResources { + preallocated_gpuvas: if full_vm { + // Unmapping the entire VM cannot split an existing mappin= g, + // so no GPUVA objects are needed for remap operations. + [None, None, None] + } else { + [ + Some(GpuVaAlloc::>::new(GFP_KERNEL)?= ), + Some(GpuVaAlloc::>::new(GFP_KERNEL)?= ), + Some(GpuVaAlloc::>::new(GFP_KERNEL)?= ), + ] + }, + }; + let result =3D { + let mut gpuvm_unique =3D self.gpuvm_unique.lock(); + self.exec_op(gpuvm_unique.as_mut().get_mut(), req, &mut resour= ces) + }; + // We flush the defer cleanup list now. Things will be different in + // the asynchronous VM_BIND path, where we want the cleanup to + // happen outside the DMA signalling path. + self.gpuvm.deferred_cleanup(); + result + } +} + +impl<'bound> DriverGpuVm for GpuVmData<'bound> { + type Driver =3D TyrDrmDriver; + type Object =3D Bo; + type VmBoData =3D (); + type VaData =3D (); + type SmContext<'ctx> + =3D PtUpdateContext<'ctx, 'bound> + where + Self: 'ctx; + + /// Create a new mapping. + fn sm_step_map<'op>( + &mut self, + op: OpMap<'op, Self>, + context: &mut Self::SmContext<'_>, + ) -> Result, Error> { + let start_iova =3D op.addr(); + let mut iova =3D start_iova; + let mut bytes_left_to_map =3D op.length(); + let mut gem_offset =3D op.gem_offset(); + + // Make sure that the end of the requested GEM range doesn't run p= ast the + // end of the GEM buffer itself. + let gem_range_end =3D op.gem_offset().checked_add(op.length()).ok_= or(EINVAL)?; + + if gem_range_end > op.obj().size() as u64 { + dev_err!( + context.dev, + "Requested GEM range ends at {} which is beyond the GEM bu= ffer size {}", + gem_range_end, + op.obj().size() + ); + return Err(EINVAL); + } + + let sgt =3D op.obj().sg_table(context.dev).inspect_err(|e| { + dev_err!(context.dev, "Failed to get sg_table: {:?}", e); + })?; + let prot =3D match &context.op_type { + PtOpType::Map(args) =3D> args.prot, + _ =3D> { + return Err(EINVAL); + } + }; + + for sgt_entry in sgt.iter() { + // Expressly convert to u64 to work with arm 32-bit builds. + #[allow(clippy::useless_conversion)] + let mut paddr =3D u64::from(sgt_entry.dma_address()); + #[allow(clippy::useless_conversion)] + let mut sgt_entry_length =3D u64::from(sgt_entry.dma_len()); + + if bytes_left_to_map =3D=3D 0 { + break; + } + + if gem_offset > 0 { + // Skip the entire SGT entry if the gem_offset exceeds its= length. + let skip =3D u64::min(sgt_entry_length, gem_offset); + paddr +=3D skip; + sgt_entry_length -=3D skip; + gem_offset -=3D skip; + } + + if sgt_entry_length =3D=3D 0 { + continue; + } + + let len =3D u64::min(sgt_entry_length, bytes_left_to_map); + + let segment_mapped =3D match pt_map(context.dev, context.pt, i= ova, paddr, len, prot) { + Ok(segment_mapped) =3D> segment_mapped, + Err(e) =3D> { + // clean up any successful mappings from previous SGT = entries. + let total_mapped =3D iova - start_iova; + if total_mapped > 0 { + let _ =3D pt_unmap( + context.dev, + context.pt, + start_iova..(start_iova + total_mapped), + ); + } + return Err(e); + } + }; + + bytes_left_to_map -=3D segment_mapped; + iova +=3D segment_mapped; + } + + if bytes_left_to_map !=3D 0 { + let total_mapped =3D iova - start_iova; + + if total_mapped > 0 { + let _ =3D pt_unmap(context.dev, context.pt, start_iova..io= va); + } + + dev_err!( + context.dev, + "SG table is too small for requested mapping: {} bytes rem= ain", + bytes_left_to_map + ); + + return Err(EINVAL); + } + + let gpuva =3D context.preallocated_gpuva()?; + let op =3D op.insert(gpuva, pin_init::init_zeroed()); + + Ok(op) + } + + /// Indicates that an existing mapping should be removed. + fn sm_step_unmap<'op>( + &mut self, + op: OpUnmap<'op, Self>, + context: &mut Self::SmContext<'_>, + ) -> Result, Error> { + let start_iova =3D op.va().addr(); + let length =3D op.va().length(); + + let region =3D start_iova..(start_iova + length); + pt_unmap(context.dev, context.pt, region.clone()).inspect_err(|e| { + dev_err!( + context.dev, + "Failed to unmap region {:#x}..{:#x}: {:?}", + region.start, + region.end, + e + ); + })?; + + let (op_unmapped, _va_removed) =3D op.remove(); + + Ok(op_unmapped) + } + + /// Split up an existing mapping. + fn sm_step_remap<'op>( + &mut self, + op: OpRemap<'op, Self>, + context: &mut Self::SmContext<'_>, + ) -> Result, Error> { + let unmap_start =3D if let Some(prev) =3D op.prev() { + prev.addr() + prev.length() + } else { + op.va_to_unmap().addr() + }; + + let unmap_end =3D if let Some(next) =3D op.next() { + next.addr() + } else { + op.va_to_unmap().addr() + op.va_to_unmap().length() + }; + + let unmap_length =3D unmap_end - unmap_start; + + if unmap_length > 0 { + let region =3D unmap_start..(unmap_start + unmap_length); + pt_unmap(context.dev, context.pt, region.clone()).inspect_err(= |e| { + dev_err!( + context.dev, + "Failed to unmap remap region {:#x}..{:#x}: {:?}", + region.start, + region.end, + e + ); + })?; + } + + let prev_va =3D context.preallocated_gpuva()?; + let next_va =3D context.preallocated_gpuva()?; + + let (op_remapped, remap_ret) =3D op.remap( + [prev_va, next_va], + pin_init::init_zeroed(), + pin_init::init_zeroed(), + ); + + if let Some(unused_va) =3D remap_ret.unused_va { + context.return_preallocated_gpuva(unused_va); + } + + Ok(op_remapped) + } +} + +/// This function selects the largest supported block size (currently 4KB = or 2MB) +/// that can be used for a mapping at the given address and size, respecti= ng alignment constraints. +/// +/// We can map multiple pages at once but we can't exceed the size of the +/// table entry itself. So, if mapping 4KB pages, figure out how many pages +/// can be mapped before we hit the 2MB boundary. Or, if mapping 2MB pages, +/// figure out how many pages can be mapped before hitting the 1GB boundary +/// Returns the page size (4KB or 2MB) and the number of pages that can be= mapped at that size. +fn get_pgsize(addr: u64, size: u64) -> (u64, u64) { + // Get the distance to the next boundary of 2MB block + let blk_offset_2m =3D addr.wrapping_neg() % (SZ_2M as u64); + + // Use 4K blocks if the address is not 2MB aligned, or we have less th= an 2MB to map + if blk_offset_2m !=3D 0 || size < SZ_2M as u64 { + let pgcount =3D if blk_offset_2m =3D=3D 0 { + size / SZ_4K as u64 + } else { + u64::min(blk_offset_2m, size) / SZ_4K as u64 + }; + return (SZ_4K as u64, pgcount); + } + + let blk_offset_1g =3D addr.wrapping_neg() % (SZ_1G as u64); + let blk_offset =3D if blk_offset_1g =3D=3D 0 { + SZ_1G as u64 + } else { + blk_offset_1g + }; + let pgcount =3D u64::min(blk_offset, size) / SZ_2M as u64; + + (SZ_2M as u64, pgcount) +} + +/// Maps a physical address range into the page table at the specified vir= tual address. +/// +/// This function maps `len` bytes of physical memory starting at `paddr` = to the +/// virtual address `iova`, using the protection flags specified in `prot`= . It +/// automatically selects optimal page sizes to minimize page table overhe= ad. +/// +/// If the mapping fails partway through, all successfully mapped pages are +/// unmapped before returning an error. +/// +/// Returns the number of bytes successfully mapped. +fn pt_map( + dev: &Device, + pt: &IoPageTable<'_, ARM64LPAES1>, + iova: u64, + paddr: u64, + len: u64, + prot: u32, +) -> Result { + let mut segment_mapped =3D 0u64; + while segment_mapped < len { + let remaining =3D len - segment_mapped; + let curr_iova =3D iova + segment_mapped; + let curr_paddr =3D paddr + segment_mapped; + + let (pgsize, pgcount) =3D get_pgsize(curr_iova | curr_paddr, remai= ning); + + // On 32-bit systems, usize is only 32 bits, so check that + // the iova can be converted without truncation. + let curr_iova =3D match usize::try_from(curr_iova) { + Ok(curr_iova) =3D> curr_iova, + Err(_) =3D> { + dev_err!( + dev, + "curr_iova {:#x} cannot be represented as usize (max {= :#x})", + curr_iova, + usize::MAX + ); + + if segment_mapped > 0 { + let _ =3D pt_unmap(dev, pt, iova..(iova + segment_mapp= ed)); + } + + return Err(EOVERFLOW); + } + }; + + // SAFETY: + // No other io-pgtable operation can currently access this range b= ecause Tyr holds + // the gpuvm_unique mutex for the entire sm_map() operation. + // The addresses being mapped won't overlap any existing mappings = in this + // page table because drm_gpuvm_sm_map() checks each requested map= ping and either unmaps + // or remaps any overlap before creating the new mapping. + let (mapped, result) =3D unsafe { + pt.map_pages( + curr_iova, + curr_paddr, + pgsize as usize, + pgcount as usize, + prot, + GFP_KERNEL, + ) + }; + + if let Err(e) =3D result { + // If map_pages fails, mapped will be zero because the ARM LPA= E backend + // only updates the mapped value after the entire request succ= eeds. + dev_err!(dev, "pt.map_pages failed at iova {:#x}: {:?}", curr_= iova, e); + if segment_mapped > 0 { + let _ =3D pt_unmap(dev, pt, iova..(iova + segment_mapped)); + } + return Err(e); + } + + if mapped =3D=3D 0 { + dev_err!(dev, "Failed to map any pages at iova {:#x}", curr_io= va); + if segment_mapped > 0 { + let _ =3D pt_unmap(dev, pt, iova..(iova + segment_mapped)); + } + return Err(ENOMEM); + } + + segment_mapped +=3D mapped as u64; + } + + Ok(segment_mapped) +} + +/// Unmaps a virtual address range from the page table. +/// +/// This function removes all page table entries in the specified range, +/// automatically handling different page sizes that may be present. +fn pt_unmap(dev: &Device, pt: &IoPageTable<'_, ARM64LPAES1>, range: Range<= u64>) -> Result { + let mut iova =3D range.start; + let mut bytes_left_to_unmap =3D range.end - range.start; + + while bytes_left_to_unmap > 0 { + // It is fine to use just the iova to determine the page size + // because if the actual mapping was represented with smaller page= sizes, + // (e.g. because the physical address was not 2MiB aligned) + // the ARM LPAE backend will notice and handle the lower-level tab= le correctly. + let (pgsize, pgcount) =3D get_pgsize(iova, bytes_left_to_unmap); + + // On 32-bit systems, usize is only 32 bits, so check that + // the iova can be converted without truncation. + let iova_usize =3D usize::try_from(iova).map_err(|_| { + dev_err!( + dev, + "IOVA {:#x} cannot be represented as usize (max {:#x})", + iova, + usize::MAX + ); + EOVERFLOW + })?; + + // SAFETY: + // No other io-pgtable operation can currently access this range b= ecause Tyr holds + // the gpuvm_unique mutex for the entire sm_unmap() operation. + // We know that this page table has one or more consecutive mappin= gs + // starting at `iova` with the total size of `pgcount * pgsize` be= cause + // gpuvm callbacks provide exactly the range that was previously m= apped. + let unmapped =3D unsafe { pt.unmap_pages(iova_usize, pgsize as usi= ze, pgcount as usize) }; + + if unmapped =3D=3D 0 { + dev_err!(dev, "Failed to unmap any bytes at iova {:#x}", iova_= usize); + return Err(EINVAL); + } + + bytes_left_to_unmap -=3D unmapped as u64; + iova +=3D unmapped as u64; + } + + Ok(()) +} --=20 2.55.0 From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 88E7D45A2A1; Wed, 22 Jul 2026 23:54:54 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764495; cv=pass; b=L5BpdpCRM0QhdUOHXHdpcb8mM0Xb2K1iDQ/b9DTlx5m4bbDK21UngWjgwN4M4wTreP7ShiTxzf74KOqzYL4ipgpRmZB1KXWUwFqpcJIGUWY3ukbZvsc+6vv0/Ii2wESbNLaE4W6LN1ZQVxwixEXd5ZuthGUGCNkhBPPN6PA0pME= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764495; c=relaxed/simple; bh=zHbLWiZ8WMuIDU4aejBoOrp0PLyfIMFgjgGvjWHbUzM=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=EEbKUR83Iz8LK5fibIaE76kHSv52ilcN+E8EM17EF4YkXXI1zrcfLuWGrD4K8IBMhS1cQRTnimSE/qcWA7mEcHudL9CcKlVczOp1z/a4chK2p25c+c5oPqFGY1k1yW4+QBmnbP+ea0FH3ZKOFoB8nDh4PjgQoayAvjmpJurDgLo= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=JPPKgPRv; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="JPPKgPRv" ARC-Seal: i=1; a=rsa-sha256; t=1784764460; cv=none; d=zohomail.com; s=zohoarc; b=heaV99LMyhzEh3hIfmThoe4UCy3EVNS/zg8Yu/YUGFYlXNEfHQw++0LMSncTOzI4jzHmxhmzEghOkwbEImbWRCUm6bxUGecz4Q+nkGyiYTdKCS8y3z780RnFsGv3lepC8p0AmcCUeqxZC0z4t6+X5+f8nDKDfkcxpi6QQWGv1Ys= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764460; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=kPxZcOCFJQSkWCGVRwm4EoZkxurlyOvE1JwNp6QWYAw=; b=HTki6I9oS5S3Cy3QG5eY2nEGfEWtuLfR8JMmuxHWoy74q07D+9F+m+IRygllNV+bd0D7bwE4mqX0EkMjOuSdjfXn2Qw5k7JiYt1EvqAelzFqk2DEydiY7OJl5lwDS5qVnaU6h1pw78rgwRgJ0x83hjZk6im/mv4uHhdapaJYKX4= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764460; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=kPxZcOCFJQSkWCGVRwm4EoZkxurlyOvE1JwNp6QWYAw=; b=JPPKgPRv2/nFwfhn/E4jtQgLJnGC8Rgq8FTfU1ZB8v/rjLzf4rZZ+nurlJyT6q08 VAkgBTcr1B7gahc6uZLpPx3Q3RSRkIK+OGlogIKbazTJF9Edk3uHbRKuPZzsMAesjxP 1j9HRmTkvmePSetvWxbRdrih42JkcpQsqF7hhKOU= Received: by mx.zohomail.com with SMTPS id 1784764458290601.0817471702306; Wed, 22 Jul 2026 16:54:18 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:11 -0700 Subject: [PATCH v9 5/7] drm/tyr: add a kernel buffer object Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-5-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=4887; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=zHbLWiZ8WMuIDU4aejBoOrp0PLyfIMFgjgGvjWHbUzM=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcrS5+4kzto2b3LDwy+pJwvaXb86VB1ZF+vzPLJ7/ yxBZ9NXHaUsDGJcDLJiiixn7Y16xKveG+nO/98MM4eVCWQIAxenAEzERp+R4X2n2rTtEz/qfd1S +ljqt/uEmXPWnf2aFfnMUspSWdPmiQ4jw7KrBhItn6+LJyjrdnlMZom6IJr1/A3PvW6hba+Vmp4 e4AUA X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB Introduce a buffer object type (KernelBo) for internal driver allocations that are managed by the kernel rather than userspace. KernelBo wraps a GEM shmem object and automatically handles GPU virtual address space mapping during creation and unmapping on drop. This provides a safe and convenient way for the driver to both allocate and clean up internal buffers for kernel-managed resources. Co-developed-by: Boris Brezillon Signed-off-by: Boris Brezillon Signed-off-by: Deborah Brouwer Reviewed-by: Daniel Almeida --- drivers/gpu/drm/tyr/gem.rs | 113 +++++++++++++++++++++++++++++++++++++++++= ++-- 1 file changed, 109 insertions(+), 4 deletions(-) diff --git a/drivers/gpu/drm/tyr/gem.rs b/drivers/gpu/drm/tyr/gem.rs index c28be61a01bb..b371299028d6 100644 --- a/drivers/gpu/drm/tyr/gem.rs +++ b/drivers/gpu/drm/tyr/gem.rs @@ -4,18 +4,29 @@ //! This module provides buffer object (BO) management functionality using //! DRM's GEM subsystem with shmem backing. =20 +use core::ops::Range; + use kernel::{ drm::gem::{ self, shmem, // }, prelude::*, - sync::aref::ARef, // + sync::{ + aref::ARef, + Arc, // + }, // }; =20 -use crate::driver::{ - TyrDrmDevice, - TyrDrmDriver, // +use crate::{ + driver::{ + TyrDrmDevice, + TyrDrmDriver, // + }, + vm::{ + Vm, + VmMapFlags, // + }, }; =20 /// Tyr's DriverObject type for GEM objects. @@ -56,3 +67,97 @@ pub(crate) fn new_dummy_object(ddev: &TyrDrmDevice) -> R= esult> { =20 Ok(bo) } + +/// Specifies how to choose a GPU virtual address for a [`KernelBo`]. +/// An automatic VA allocation strategy will be added in the future. +pub(crate) enum KernelBoVaAlloc { + /// Explicit VA address specified by the caller. + #[expect(dead_code)] + Explicit(u64), +} + +/// A kernel-owned buffer object with automatic GPU virtual address mappin= g. +/// +/// This structure represents a buffer object that is created and managed = entirely +/// by the kernel driver, as opposed to userspace-created GEM objects. It = combines +/// a GEM object with automatic GPU virtual address (VA) space mapping and= cleanup. +/// +/// When dropped, the buffer is automatically unmapped from the GPU VA spa= ce. +pub(crate) struct KernelBo<'bound> { + /// The underlying GEM buffer object. + bo: ARef, + /// The GPU VM this buffer is mapped into. + vm: Arc>, + /// The GPU VA range occupied by this buffer. + va_range: Range, +} + +impl<'bound> KernelBo<'bound> { + /// Creates a new kernel-owned buffer object and maps it into GPU VA s= pace. + /// + /// This function allocates a new shmem-backed GEM object and immediat= ely maps + /// it into the specified GPU virtual memory space. The mapping is aut= omatically + /// cleaned up when the [`KernelBo`] is dropped. + #[expect(dead_code)] + pub(crate) fn new( + ddev: &TyrDrmDevice, + vm: Arc>, + size: u64, + va_alloc: KernelBoVaAlloc, + flags: VmMapFlags, + ) -> Result { + if size =3D=3D 0 { + dev_err!(vm.dev(), "Cannot create KernelBo with size 0"); + return Err(EINVAL); + } + + let KernelBoVaAlloc::Explicit(va) =3D va_alloc; + + let bo_size =3D usize::try_from(size).map_err(|_| EOVERFLOW)?; + let va_end =3D va.checked_add(size).ok_or(EINVAL)?; + + let bo =3D Bo::new( + ddev, + bo_size, + shmem::ObjectConfig { + map_wc: true, + parent_resv_obj: None, + }, + BoCreateArgs { flags: 0 }, + )?; + + vm.map_bo_range(&bo, 0, size, va, flags)?; + + Ok(KernelBo { + bo, + vm, + va_range: va..va_end, + }) + } + + #[expect(dead_code)] + pub(crate) fn bo(&self) -> &Bo { + &self.bo + } +} + +impl Drop for KernelBo<'_> { + fn drop(&mut self) { + let va =3D self.va_range.start; + let size =3D self.va_range.end - self.va_range.start; + + if let Err(e) =3D self.vm.unmap_range(va, size) { + // If unmap_range fails, it is still safe to drop the + // KernelBo and its ARef to the GEM buffer object because + // GPUVM also holds a reference to the GEM buffer object. + // The physical pages won't be freed or reallocated. + dev_err!( + self.vm.dev(), + "Failed to unmap KernelBo range {:#x}..{:#x}: {:?}", + self.va_range.start, + self.va_range.end, + e + ); + } + } +} --=20 2.55.0 From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 01305459AF6; Wed, 22 Jul 2026 23:54:53 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764496; cv=pass; b=G+NaHzhEg6y+8vnkmm3HeABU+mlBcP2Mfi1ExxaSnSmjN4YMM/Y55RgwVPZfJWjMADsNwBrBrL0dYxqgG0YcxMagrgICHHhdS+mU//lFYxkLu0QwIQNHdu735MR/md9zABXy1vEIWm8No1hto4mCTYCitrvSRNqVB4cWEh7cnv0= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764496; c=relaxed/simple; bh=Kb3YIPW+GDuw/kMG0JZ42kyaNU2IyFnrcFVNa9dzdl0=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=mDs+3V6ylAVn8F4+kbXNn0CPHMxKWKNk8WETLmxtr4lh1KOPrKN4ZjacoHmNbwKUGUwpOhzYTVI0XwppjPptoGUv12KozrXONuF8t2eCXbqNjgTyAWof5Y+2Iem+vVgrDS0J+Q2E0zUZugLjS4hqX0s/EVxMmctiiFafDMNHG5A= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=BkyxHMX6; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="BkyxHMX6" ARC-Seal: i=1; a=rsa-sha256; t=1784764460; cv=none; d=zohomail.com; s=zohoarc; b=ObJrlTYrdAvYiAjQDNUG4eTwmdvljd+ewM93sYF3ywG1u/VnRHTeSvQUPceNuN1VrGM8XOXJJr2rFpExoJAs51PUFHLzC0zaS04G/0kjeQgQDGd/w+9/MecBxt3TTOqBY6rniT7Gzhfqszz6sJXlpfmECa/TEsbJgtzdmk437zM= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764460; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=TvXnYwGRr/5R68oyHjBcZr5UU+E2jdQ7GiYszvmDWyE=; b=eBt8eAXAYEAcq/qE7hhnyjHKCWgWkOdgHC9ADsgTjETslj4vr2vbOuRpC7R3kRSU491ptJB0X2BLkI0b/QriLXSIBG8v0PY1s0jbsNRiolExZAm0JjDGxVawHX+YgmlhQML5Xr2YIUfTOWw2Y33xQkrxmFOihxh5n86ICOZFs5g= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764460; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=TvXnYwGRr/5R68oyHjBcZr5UU+E2jdQ7GiYszvmDWyE=; b=BkyxHMX6p9vTNHeqzza13fC1kQ8DLF80YsvQ/HPqw/4lAuArxVklAz8kAT1T+cTE 9uMU1I0Q9J9xJ0AQLLGqV1H0y4mpfd8m7anThkNHIU0qABog1OFaiV+XDw9PuwFlmp0 RTxkIWJM/3+eEf8bFcdf52y276OxeXfPX8qE2N4o= Received: by mx.zohomail.com with SMTPS id 1784764459318369.9182970166802; Wed, 22 Jul 2026 16:54:19 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:12 -0700 Subject: [PATCH v9 6/7] drm/tyr: add parser for firmware binary Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-6-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=20040; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=4uauGNrmbFrjJl21ZnqBGuRTt3aBiHgNWqTuGmxqjWA=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcoFztc+zXjMqSw57cz6Y2se/vrdriJg0WGYFRTz6 8h3SYfjHaUsDGJcDLJiiixn7Y16xKveG+nO/98MM4eVCWQIAxenAEyEwZqR4YPwykBRse+nPLUd a+dzPHlb9VE9eRbHvpnpYba7l3uGv2ZkmOo7a9bq4to7d2dPOxTs1HOs2FNal/vA5yAl64MHRKY p8wIA X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB From: Daniel Almeida Add a parser for the Mali CSF GPU firmware binary format. The firmware consists of a header followed by entries describing how to load firmware sections into the MCU's memory. The parser extracts section metadata including virtual address ranges, data byte offsets within the binary, and section flags controlling permissions and cache modes. It validates the basic firmware structure and alignment and ignores protected-mode sections for now. Signed-off-by: Daniel Almeida Co-developed-by: Beata Michalska Signed-off-by: Beata Michalska Co-developed-by: Boris Brezillon Signed-off-by: Boris Brezillon Co-developed-by: Deborah Brouwer Signed-off-by: Deborah Brouwer Reviewed-by: Daniel Almeida --- drivers/gpu/drm/tyr/fw/parser.rs | 588 +++++++++++++++++++++++++++++++++++= ++++ 1 file changed, 588 insertions(+) diff --git a/drivers/gpu/drm/tyr/fw/parser.rs b/drivers/gpu/drm/tyr/fw/pars= er.rs new file mode 100644 index 000000000000..c4d0ad1d7899 --- /dev/null +++ b/drivers/gpu/drm/tyr/fw/parser.rs @@ -0,0 +1,588 @@ +// SPDX-License-Identifier: GPL-2.0 or MIT + +//! Firmware binary parser for Mali CSF (Command Stream Frontend) GPU. +//! +//! This module implements a parser for the Mali GPU firmware binary forma= t. The firmware +//! file contains a header followed by a sequence of entries, each describ= ing how to load +//! firmware sections into the MCU (Microcontroller Unit) memory. The pars= er extracts section +//! metadata including: +//! - Virtual address ranges where sections should be mapped +//! - Data ranges (byte offsets) within the firmware binary +//! - Section flags (permissions, cache modes) + +use core::{ + mem::size_of, + ops::Range, // +}; + +use kernel::{ + bits::bit_u32, + device::Device, + prelude::*, + sizes::SZ_4K, // +}; + +use crate::{ + fw::{ + CacheMode, + SectionFlags, + CSF_MCU_SHARED_REGION_START, // + }, + vm::{ + VmFlag, + VmMapFlags, // + }, // +}; + +/// A parsed firmware section ready for loading into MCU memory. +/// +/// Represents a single firmware section extracted from the firmware binar= y, containing +/// all information needed to map the section's data into the MCU's virtua= l address space. +pub(super) struct ParsedSection { + /// Byte offset range within the firmware binary where this section's = data resides. + pub(super) data_range: Range, + /// MCU virtual address range where this section should be mapped. + pub(super) va: Range, + /// Memory protection and caching flags for the mapping. + pub(super) vm_map_flags: VmMapFlags, +} + +/// A bare-bones `std::io::Cursor<[u8]>` clone to keep track of the curren= t position in the +/// firmware binary. +/// +/// Provides methods to sequentially read primitive types and byte arrays = from the firmware +/// binary while maintaining the current read position. +struct Cursor<'a> { + dev: &'a Device, + data: &'a [u8], + pos: usize, +} + +impl<'a> Cursor<'a> { + fn new(dev: &'a Device, data: &'a [u8]) -> Self { + Self { dev, data, pos: 0 } + } + + fn len(&self) -> usize { + self.data.len() + } + + fn pos(&self) -> usize { + self.pos + } + + /// Returns a view into the cursor's data. + /// + /// This spawns a new cursor, leaving the current cursor unchanged. + fn view(&self, range: Range) -> Result> { + if range.start < self.pos || range.end > self.data.len() { + dev_err!( + self.dev, + "Invalid cursor range {:?} for data of length {}", + range, + self.data.len() + ); + + Err(EINVAL) + } else { + Ok(Self { + dev: self.dev, + data: &self.data[range], + pos: 0, + }) + } + } + + /// Reads a slice of bytes from the current position and advances the = cursor. + /// + /// Returns an error if the read would exceed the data bounds. + fn read(&mut self, nbytes: usize) -> Result<&[u8]> { + let start =3D self.pos; + let end =3D start + nbytes; + + if end > self.data.len() { + dev_err!( + self.dev, + "Invalid firmware file: read of size {} at position {} is = out of bounds", + nbytes, + start, + ); + return Err(EINVAL); + } + + self.pos +=3D nbytes; + Ok(&self.data[start..end]) + } + + /// Reads a little-endian `u8` from the current position and advances = the cursor. + fn read_u8(&mut self) -> Result { + let bytes =3D self.read(size_of::())?; + Ok(bytes[0]) + } + + /// Reads a little-endian `u16` from the current position and advances= the cursor. + fn read_u16(&mut self) -> Result { + let bytes: [u8; 2] =3D self + .read(size_of::())? + .try_into() + .map_err(|_| EINVAL)?; + + Ok(u16::from_le_bytes(bytes)) + } + + /// Reads a little-endian `u32` from the current position and advances= the cursor. + fn read_u32(&mut self) -> Result { + let bytes: [u8; 4] =3D self + .read(size_of::())? + .try_into() + .map_err(|_| EINVAL)?; + + Ok(u32::from_le_bytes(bytes)) + } + + /// Advances the cursor position by the specified number of bytes. + /// + /// Returns an error if the advance would exceed the data bounds. + fn advance(&mut self, nbytes: usize) -> Result { + if self.pos + nbytes > self.data.len() { + dev_err!( + self.dev, + "Invalid firmware file: advance of size {} at position {} = is out of bounds", + nbytes, + self.pos, + ); + return Err(EINVAL); + } + self.pos +=3D nbytes; + Ok(()) + } +} + +/// Parser for Mali CSF GPU firmware binaries. +/// +/// Parses the firmware binary format, extracting section metadata includi= ng virtual +/// address ranges, data offsets, and memory protection flags needed to lo= ad firmware +/// into the MCU's memory. +pub(super) struct FwParser<'a> { + cursor: Cursor<'a>, +} + +impl<'a> FwParser<'a> { + /// Creates a new firmware parser for the given firmware binary data. + pub(super) fn new(dev: &'a Device, data: &'a [u8]) -> Self { + Self { + cursor: Cursor::new(dev, data), + } + } + + /// Parses the firmware binary and returns a collection of parsed sect= ions. + /// + /// This method validates the firmware header and iterates through all= entries + /// in the binary, extracting section information needed for loading. + pub(super) fn parse(&mut self) -> Result> { + let fw_header =3D self.parse_fw_header()?; + let header_end =3D fw_header.size as usize; + + let mut parsed_sections =3D KVec::new(); + while self.cursor.pos() < header_end { + let entry_section =3D self.parse_entry(header_end)?; + + if let Some(inner) =3D entry_section.inner { + parsed_sections.push(inner, GFP_KERNEL)?; + } + } + + if parsed_sections.is_empty() { + dev_err!(self.cursor.dev, "Firmware contains no loadable secti= ons"); + return Err(EINVAL); + } + + Ok(parsed_sections) + } + + fn parse_fw_header(&mut self) -> Result { + let fw_header: FirmwareHeader =3D match FirmwareHeader::new(&mut s= elf.cursor) { + Ok(fw_header) =3D> fw_header, + Err(e) =3D> { + dev_err!(self.cursor.dev, "Invalid firmware file: {}", e.t= o_errno()); + return Err(e); + } + }; + + if fw_header.size as usize > self.cursor.len() { + dev_err!(self.cursor.dev, "Firmware image is truncated"); + return Err(EINVAL); + } + Ok(fw_header) + } + + fn parse_entry(&mut self, header_end: usize) -> Result { + let entry_start =3D self.cursor.pos(); + + let entry_header_end =3D entry_start + .checked_add(size_of::()) + .ok_or(EINVAL)?; + + if entry_header_end > header_end { + dev_err!( + self.cursor.dev, + "Firmware entry header at {:#x} exceeds header region endi= ng at {:#x}", + entry_start, + header_end + ); + return Err(EINVAL); + } + + let entry_section =3D EntrySection { + entry_hdr: EntryHeader(self.cursor.read_u32()?), + inner: None, + }; + + let firmware_size =3D self.cursor.len(); + let entry_size =3D entry_section.entry_hdr.size() as usize; + + if self.cursor.pos() % size_of::() !=3D 0 + || entry_size % size_of::() !=3D 0 + || entry_size < size_of::() + { + dev_err!( + self.cursor.dev, + "Firmware entry isn't 32 bit aligned, offset=3D{:#x} size= =3D{:#x}", + self.cursor.pos() - size_of::(), + entry_size + ); + return Err(EINVAL); + } + + let entry_end =3D entry_start.checked_add(entry_size).ok_or(EINVAL= )?; + + if entry_end > header_end { + dev_err!( + self.cursor.dev, + "Firmware entry at {:#x} extends beyond header region endi= ng at {:#x}", + entry_start, + header_end + ); + return Err(EINVAL); + } + + let section_hdr_size =3D entry_size - size_of::(); + + let entry_section =3D { + let mut entry_cursor =3D self.cursor.view(self.cursor.pos()..e= ntry_end)?; + + match entry_section.entry_hdr.entry_type() { + Ok(EntryType::Iface) =3D> Ok(EntrySection { + entry_hdr: entry_section.entry_hdr, + inner: Self::parse_section_entry(&mut entry_cursor, fi= rmware_size)?, + }), + Ok( + EntryType::Config + | EntryType::FutfTest + | EntryType::TraceBuffer + | EntryType::TimelineMetadata + | EntryType::BuildInfoMetadata, + ) =3D> Ok(entry_section), + + Err(_) =3D> { + if entry_section.entry_hdr.optional() { + Ok(entry_section) + } else { + dev_err!( + self.cursor.dev, + "Failed to handle firmware entry type: {}", + entry_section.entry_hdr.entry_type_raw() + ); + Err(EINVAL) + } + } + } + }; + + if entry_section.is_ok() { + self.cursor.advance(section_hdr_size)?; + } + + entry_section + } + + fn parse_section_entry( + entry_cursor: &mut Cursor<'_>, + firmware_size: usize, + ) -> Result> { + let section_hdr: SectionHeader =3D SectionHeader::new(entry_cursor= )?; + + if section_hdr.data.end < section_hdr.data.start { + dev_err!( + entry_cursor.dev, + "Firmware corrupted, data.end < data.start (0x{:x} < 0x{:x= })", + section_hdr.data.end, + section_hdr.data.start + ); + return Err(EINVAL); + } + + if section_hdr.data.end as usize > firmware_size { + dev_err!( + entry_cursor.dev, + "Firmware data range {:#x}..{:#x} exceeds firmware size {:= #x}", + section_hdr.data.start, + section_hdr.data.end, + firmware_size, + ); + return Err(EINVAL); + } + + if section_hdr.va.start as usize % SZ_4K !=3D 0 || section_hdr.va.= end as usize % SZ_4K !=3D 0 { + dev_err!( + entry_cursor.dev, + "Firmware virtual address range {:#x}..{:#x} is not page a= ligned", + section_hdr.va.start, + section_hdr.va.end + ); + return Err(EINVAL); + } + + if section_hdr.section_flags.prot() { + dev_dbg!( + entry_cursor.dev, + "Firmware protected mode entry not supported, ignoring" + ); + return Ok(None); + } + + if section_hdr.va.start =3D=3D CSF_MCU_SHARED_REGION_START + && !section_hdr.section_flags.shared() + { + dev_err!( + entry_cursor.dev, + "Interface at 0x{:x} must be shared", + CSF_MCU_SHARED_REGION_START + ); + return Err(EINVAL); + } + + if section_hdr.va.is_empty() { + return Ok(None); + } + + let mut vm_map_flags =3D VmMapFlags::empty(); + + if !section_hdr.section_flags.write() { + vm_map_flags |=3D VmFlag::Readonly; + } + + if !section_hdr.section_flags.exec() { + vm_map_flags |=3D VmFlag::Noexec; + } + + // TODO: As in Panthor, map coherent firmware sections uncached un= til the VM + // supports a coherent mapping attribute. + if section_hdr.section_flags.cache_mode() !=3D CacheMode::Cached { + vm_map_flags |=3D VmFlag::Uncached; + } + + Ok(Some(ParsedSection { + data_range: section_hdr.data.clone(), + va: section_hdr.va, + vm_map_flags, + })) + } +} + +/// Firmware binary header containing version and size information. +/// +/// The header is located at the beginning of the firmware binary and cont= ains +/// a magic value for validation, version information, and the total size = of +/// all structured headers that follow. +#[expect(dead_code)] +struct FirmwareHeader { + /// Magic value to check binary validity. + magic: u32, + + /// Minor firmware version. + minor: u8, + + /// Major firmware version. + major: u8, + + /// Padding. Must be set to zero. + _padding1: u16, + + /// Firmware version hash. + version_hash: u32, + + /// Padding. Must be set to zero. + _padding2: u32, + + /// Total size of all the structured data headers at beginning of firm= ware binary. + size: u32, +} + +impl FirmwareHeader { + const FW_BINARY_MAGIC: u32 =3D 0xc3f13a6e; + const FW_BINARY_MAJOR_MAX: u8 =3D 0; + + /// Reads and validates a firmware header from the cursor. + /// + /// Verifies the magic value, version compatibility, and padding field= s. + fn new(cursor: &mut Cursor<'_>) -> Result { + let magic =3D cursor.read_u32()?; + if magic !=3D Self::FW_BINARY_MAGIC { + dev_err!(cursor.dev, "Invalid firmware magic"); + return Err(EINVAL); + } + + let minor =3D cursor.read_u8()?; + let major =3D cursor.read_u8()?; + + if major > Self::FW_BINARY_MAJOR_MAX { + dev_err!( + cursor.dev, + "Unsupported firmware binary header version {}.{} (expecte= d {}.x)", + major, + minor, + Self::FW_BINARY_MAJOR_MAX + ); + return Err(EINVAL); + } + + let padding1 =3D cursor.read_u16()?; + let version_hash =3D cursor.read_u32()?; + let padding2 =3D cursor.read_u32()?; + let size =3D cursor.read_u32()?; + + if padding1 !=3D 0 || padding2 !=3D 0 { + dev_err!( + cursor.dev, + "Invalid firmware file: header padding is not zero" + ); + return Err(EINVAL); + } + + let fw_header =3D Self { + magic, + minor, + major, + _padding1: padding1, + version_hash, + _padding2: padding2, + size, + }; + + Ok(fw_header) + } +} + +/// Firmware section header for loading binary sections into MCU memory. +#[derive(Debug)] +struct SectionHeader { + section_flags: SectionFlags, + /// MCU virtual range to map this binary section to. + va: Range, + /// References the data in the FW binary. + data: Range, +} + +impl SectionHeader { + /// Reads and validates a section header from the cursor. + /// + /// Parses section flags, virtual address range, and data range from t= he firmware binary. + fn new(cursor: &mut Cursor<'_>) -> Result { + let section_flags =3D SectionFlags::try_from_fw(cursor.read_u32()?= )?; + + let va_start =3D cursor.read_u32()?; + let va_end =3D cursor.read_u32()?; + + let va =3D va_start..va_end; + + if va.end < va.start { + dev_err!( + cursor.dev, + "Invalid firmware file: VA end precedes start at pos {}", + cursor.pos(), + ); + return Err(EINVAL); + } + + let data_start =3D cursor.read_u32()?; + let data_end =3D cursor.read_u32()?; + let data =3D data_start..data_end; + + Ok(Self { + section_flags, + va, + data, + }) + } +} + +/// A firmware entry containing a header and optional parsed section data. +/// +/// Represents a single entry in the firmware binary, which may contain lo= adable +/// section data or metadata that doesn't require loading. +struct EntrySection { + entry_hdr: EntryHeader, + inner: Option, +} + +/// Header for a firmware entry, packed into a single u32. +/// +/// The entry header encodes the entry type, size, and optional flag in a +/// 32-bit value with the following layout: +/// - Bits 0-7: Entry type +/// - Bits 8-15: Size in bytes +/// - Bit 31: Optional flag +struct EntryHeader(u32); + +impl EntryHeader { + fn entry_type_raw(&self) -> u8 { + (self.0 & 0xff) as u8 + } + + fn entry_type(&self) -> Result { + let v =3D self.entry_type_raw(); + EntryType::try_from(v) + } + + fn optional(&self) -> bool { + self.0 & bit_u32(31) !=3D 0 + } + + fn size(&self) -> u32 { + self.0 >> 8 & 0xff + } +} + +#[derive(Clone, Copy, Debug)] +#[repr(u8)] +enum EntryType { + /// Host <-> FW interface. + Iface =3D 0, + /// FW config. + Config =3D 1, + /// Unit tests. + FutfTest =3D 2, + /// Trace buffer interface. + TraceBuffer =3D 3, + /// Timeline metadata interface. + TimelineMetadata =3D 4, + /// Metadata about how the FW binary was built. + BuildInfoMetadata =3D 6, +} + +impl TryFrom for EntryType { + type Error =3D Error; + + fn try_from(value: u8) -> Result { + match value { + 0 =3D> Ok(EntryType::Iface), + 1 =3D> Ok(EntryType::Config), + 2 =3D> Ok(EntryType::FutfTest), + 3 =3D> Ok(EntryType::TraceBuffer), + 4 =3D> Ok(EntryType::TimelineMetadata), + 6 =3D> Ok(EntryType::BuildInfoMetadata), + _ =3D> Err(EINVAL), + } + } +} --=20 2.55.0 From nobody Fri Jul 24 22:17:47 2026 Received: from sender4-op-o11.zoho.com (sender4-op-o11.zoho.com [136.143.188.11]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 94BF54611E5; Wed, 22 Jul 2026 23:54:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=pass smtp.client-ip=136.143.188.11 ARC-Seal: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764500; cv=pass; b=rJLW7+aqfO4JIcZIrjr9VJpsYNW+ctmnugKkN3gTnR7NcBEaMI2wuVnonb+5X2AhVbxeEMYET1Fqq1J6qQDLhSCFgIJJgjzhcCmQIOgnPCqdaA/Tx+5bf4xArxlw/H7gflYwY1l1VMcnYZ1Ko1v3IxfnmOfD17CzptAm3clhL2Y= ARC-Message-Signature: i=2; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1784764500; c=relaxed/simple; bh=NWaAh7yaUfbdgLyDZ/0RX8WpE3m3erlF49nZ7vDDw3w=; h=From:Date:Subject:MIME-Version:Content-Type:Message-Id:References: In-Reply-To:To:Cc; b=ep8xrVeiRv4ulN0Wj34MFqIGk5MawgjiBncz3lnHUQkY/nwvZ3IjDTJGaNSvUP/Jgj7A0YIYBw4l3FyctLfQsKm4zKAQEZFvTOur87WJs6jUqbxj4HwTThCJN3G9ztxPovhaOmeKuishQUzQFq+w318RObZRsZp0g3Rg5DLyhXY= ARC-Authentication-Results: i=2; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com; spf=pass smtp.mailfrom=collabora.com; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b=faqV0jCF; arc=pass smtp.client-ip=136.143.188.11 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=collabora.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=collabora.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=collabora.com header.i=deborah.brouwer@collabora.com header.b="faqV0jCF" ARC-Seal: i=1; a=rsa-sha256; t=1784764462; cv=none; d=zohomail.com; s=zohoarc; b=XE6rq6zr5sSM8+JTvRSe0G5j6D/fTvGYu93Grp+F5vsws0l6G0WmTxKYpLmwP6qW+V9RMEDtSUpG/eF/W1UMgwB6JVdUZg34j0CV4XRb10huUvnKilnd8kqHlo/ov5t9HcY2DpfOa+slcb9VgtUtoCjY6yvFyS+FXjQSqBAGiGM= ARC-Message-Signature: i=1; a=rsa-sha256; c=relaxed/relaxed; d=zohomail.com; s=zohoarc; t=1784764462; h=Content-Type:Content-Transfer-Encoding:Cc:Cc:Date:Date:From:From:In-Reply-To:MIME-Version:Message-ID:Subject:Subject:To:To:Message-Id:Reply-To; bh=v/gD8OKp1x+gK9LcyeJ7sv7IGO6Bt++6bAQvdun5I4U=; b=oFUpPBEdBKfZxJ7lvVhv3Dze1tbAXUkYPqI6Eb8jd7gQ04Ts16q1rMtYuLNWpumEsmd1JR1HnNNc90I+oyQcgIIXsSDL5wRmKHUxQ+1RB7hsPV1m5smYqn/OifzOvbERptoREmSZYI9HWx3jwGW4bmJsu5ZkbeVWSl9ajb2y7tY= ARC-Authentication-Results: i=1; mx.zohomail.com; dkim=pass header.i=collabora.com; spf=pass smtp.mailfrom=deborah.brouwer@collabora.com; dmarc=pass header.from= DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; t=1784764462; s=zohomail; d=collabora.com; i=deborah.brouwer@collabora.com; h=From:From:Date:Date:Subject:Subject:MIME-Version:Content-Type:Content-Transfer-Encoding:Message-Id:Message-Id:In-Reply-To:To:To:Cc:Cc:Reply-To; bh=v/gD8OKp1x+gK9LcyeJ7sv7IGO6Bt++6bAQvdun5I4U=; b=faqV0jCFPEXZD5bl8eZcS5XMivY1FCxmEn1s9PPPFCBQ8B2O1MoqLpGbtwP4/Gdo s3jYYu38gnLiEYMjVQwgKQ6CqHgkKGYxpmd7AM6pvAOogQl0xPl06jAV6g+isOQoANp ueY+cNgGrrV8dOLKmNqUtY9dagepNqVrOpLTUs5g= Received: by mx.zohomail.com with SMTPS id 1784764460404955.6325363246353; Wed, 22 Jul 2026 16:54:20 -0700 (PDT) From: Deborah Brouwer Date: Wed, 22 Jul 2026 16:54:13 -0700 Subject: [PATCH v9 7/7] drm/tyr: add Microcontroller Unit (MCU) booting Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Message-Id: <20260722-fw-boot-b4-v9-7-8669d2a02590@collabora.com> References: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> In-Reply-To: <20260722-fw-boot-b4-v9-0-8669d2a02590@collabora.com> To: Daniel Almeida , Alice Ryhl , Danilo Krummrich , David Airlie , Simona Vetter , Benno Lossin , Gary Guo , Miguel Ojeda , Boqun Feng , =?utf-8?q?Bj=C3=B6rn_Roy_Baron?= , Andreas Hindborg , Trevor Gross , Tamir Duberstein , Alexandre Courbot , =?utf-8?q?Onur_=C3=96zkan?= Cc: dri-devel@lists.freedesktop.org, linux-kernel@vger.kernel.org, rust-for-linux@vger.kernel.org, Deborah Brouwer , samitolvanen@google.com, lyude@redhat.com, boris.brezillon@collabora.com, steven.price@arm.com, alvin.sun@linux.dev, laura.nao@collabora.com, beata.michalska@arm.com X-Mailer: b4 0.14.3 X-Developer-Signature: v=1; a=openpgp-sha256; l=14540; i=deborah.brouwer@collabora.com; h=from:subject:message-id; bh=NWaAh7yaUfbdgLyDZ/0RX8WpE3m3erlF49nZ7vDDw3w=; b=owGbwMvMwCVWuULzOU9c7WvG02pJDFmJEcpStyWU4w9kpn4UXFu5vM1Y491584vnroR2Hn+Un +IlXePfUcrCIMbFICumyHLW3qhHvOq9ke78/80wc1iZQIYwcHEKwESEzzH8d7/iM7tm/5ypTQ6s kez3Ze6X7HrBZKclua+9Si811rrGh+GfneZM65+Xtx2LyN1/j/vB3q0Rvs0KIrHmFr2hrYo3U3n ZAQ== X-Developer-Key: i=deborah.brouwer@collabora.com; a=openpgp; fpr=CD3F328C177AEF322D9FFF8379A829E70C5E7DEB Add a firmware module to load, parse, and map the MCU firmware sections into shared GEM memory at the required virtual addresses accessible by the GPU. Create a firmware instance during probe and store it inside the TyrDrmRegistrationData to keep it alive after probe. Use the firmware instance to boot the MCU. Remove the dead-code annotations from the MMU, VM, slot manager, and kernel BO code now that these paths are used by the firmware module. Update Kconfig to add the RUST_FW_LOADER_ABSTRACTIONS dependency required by this module. Co-developed-by: Boris Brezillon Signed-off-by: Boris Brezillon Signed-off-by: Deborah Brouwer Reviewed-by: Daniel Almeida --- drivers/gpu/drm/tyr/Kconfig | 1 + drivers/gpu/drm/tyr/driver.rs | 23 ++- drivers/gpu/drm/tyr/fw.rs | 321 ++++++++++++++++++++++++++++++++++++++= ++++ drivers/gpu/drm/tyr/gem.rs | 3 - drivers/gpu/drm/tyr/tyr.rs | 1 + drivers/gpu/drm/tyr/vm.rs | 1 - 6 files changed, 341 insertions(+), 9 deletions(-) diff --git a/drivers/gpu/drm/tyr/Kconfig b/drivers/gpu/drm/tyr/Kconfig index 79ea4bb214de..8f13e49f11f9 100644 --- a/drivers/gpu/drm/tyr/Kconfig +++ b/drivers/gpu/drm/tyr/Kconfig @@ -13,6 +13,7 @@ config DRM_TYR select IOMMU_IO_PGTABLE_LPAE select RUST_DRM_GEM_SHMEM_HELPER select RUST_DRM_GPUVM + select RUST_FW_LOADER_ABSTRACTIONS help Rust DRM driver for ARM Mali CSF-based GPUs. =20 diff --git a/drivers/gpu/drm/tyr/driver.rs b/drivers/gpu/drm/tyr/driver.rs index b6528d8cd3ce..8f87fd5b772a 100644 --- a/drivers/gpu/drm/tyr/driver.rs +++ b/drivers/gpu/drm/tyr/driver.rs @@ -37,6 +37,7 @@ =20 use crate::{ file::TyrDrmFileData, + fw::Firmware, gem::Bo, gpu, gpu::GpuInfo, @@ -67,6 +68,9 @@ pub(crate) struct TyrDrmRegistrationData<'bound> { /// Parent platform device. pub(crate) pdev: &'bound platform::Device, =20 + /// Firmware sections. + pub(crate) fw: Firmware<'bound>, + #[pin] clks: Mutex, =20 @@ -144,10 +148,21 @@ fn probe<'bound>( =20 let unreg_dev =3D drm::UnregisteredDevice::::new(pde= v, Ok(()))?; =20 - let _mmu =3D Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_i= nfo)?; + let mmu =3D Mmu::new(pdev.as_ref(), iomem.as_arc_borrow(), &gpu_in= fo)?; + + let firmware =3D Firmware::new( + pdev.as_ref(), + iomem.clone(), + &unreg_dev, + mmu.as_arc_borrow(), + &gpu_info, + )?; + + firmware.boot()?; =20 - let reg_data =3D try_pin_init!(TyrDrmRegistrationData { + let reg_data =3D pin_init!(TyrDrmRegistrationData { pdev, + fw: firmware, clks <- new_mutex!(Clocks { core: core_clk, stacks: stacks_clk, @@ -167,9 +182,7 @@ fn probe<'bound>( =20 let driver =3D TyrPlatformDriverData { _reg: reg }; =20 - // We need this to be dev_info!() because dev_dbg!() does not work= at - // all in Rust for now, and we need to see whether probe succeeded. - dev_info!(pdev, "Tyr initialized correctly.\n"); + dev_dbg!(pdev, "Tyr initialized correctly."); Ok(driver) } } diff --git a/drivers/gpu/drm/tyr/fw.rs b/drivers/gpu/drm/tyr/fw.rs new file mode 100644 index 000000000000..df11647023c2 --- /dev/null +++ b/drivers/gpu/drm/tyr/fw.rs @@ -0,0 +1,321 @@ +// SPDX-License-Identifier: GPL-2.0 or MIT + +//! Firmware loading and management for Mali CSF GPU. +//! +//! This module handles loading the Mali GPU firmware binary, parsing it i= nto sections, +//! and mapping those sections into the MCU's virtual address space. Each = firmware section +//! has specific properties (read/write/execute permissions, cache modes) = and must be loaded +//! at specific virtual addresses expected by the MCU. +//! +//! See [`Firmware`] for the main firmware management interface and [`Sect= ion`] for +//! individual firmware sections. +//! +//! [`Firmware`]: crate::fw::Firmware +//! [`Section`]: crate::fw::Section + +use kernel::{ + device::{ + Bound, + Device, // + }, + drm::{ + gem::BaseObject, // + }, + io::{ + poll, + Io, // + }, + num::Bounded, + prelude::*, + register, + str::CString, + sync::{ + Arc, + ArcBorrow, // + }, + time, // +}; + +use crate::{ + driver::{ + IoMem, + TyrDrmDevice, // + }, + fw::parser::{ + FwParser, + ParsedSection, // + }, + gem, + gem::{ + KernelBo, + KernelBoVaAlloc, // + }, + gpu::GpuInfo, + + mmu::Mmu, + regs::{ + gpu_control::{ + McuControlMode, + McuStatus, + GPU_ID, + MCU_CONTROL, + MCU_STATUS, // + }, // + job_control::{ + JOB_IRQ_CLEAR, + JOB_IRQ_RAWSTAT, // + }, // + }, + vm::Vm, // +}; + +mod parser; + +pub(super) const CSF_MCU_SHARED_REGION_START: u32 =3D 0x04000000; + +#[derive(Copy, Clone, Debug, PartialEq, Eq)] +#[repr(u8)] +pub(super) enum CacheMode { + None =3D 0, + Cached =3D 1, + UncachedCoherent =3D 2, + CachedCoherent =3D 3, +} + +impl From> for CacheMode { + fn from(value: Bounded) -> Self { + match value.get() { + 0 =3D> Self::None, + 1 =3D> Self::Cached, + 2 =3D> Self::UncachedCoherent, + 3 =3D> Self::CachedCoherent, + _ =3D> unreachable!(), + } + } +} + +impl From for Bounded { + fn from(value: CacheMode) -> Self { + Bounded::try_new(value as u32).unwrap() + } +} + +register! { + #[allow(non_upper_case_globals)] + pub(super) SectionFlags(u32) @ 0x0 { + 0:0 read =3D> bool; + 1:1 write =3D> bool; + 2:2 exec =3D> bool; + 4:3 cache_mode =3D> CacheMode; + 5:5 prot =3D> bool; + 30:30 shared =3D> bool; + 31:31 zero =3D> bool; + } +} + +impl SectionFlags { + const VALID_MASK: u32 =3D Self::READ_MASK + | Self::WRITE_MASK + | Self::EXEC_MASK + | Self::CACHE_MODE_MASK + | Self::PROT_MASK + | Self::SHARED_MASK + | Self::ZERO_MASK; + + fn try_from_fw(value: u32) -> Result { + if value & !Self::VALID_MASK !=3D 0 { + Err(EINVAL) + } else { + Ok(Self::from_raw(value)) + } + } +} + +/// A parsed section of the firmware binary. +struct Section<'bound> { + // Raw firmware section data for reset purposes + #[expect(dead_code)] + data: KVec, + + // Keep the BO backing this firmware section so that both the + // GPU mapping and CPU mapping remain valid until the Section is dropp= ed. + #[expect(dead_code)] + mem: gem::KernelBo<'bound>, +} + +/// Loaded firmware with sections mapped into MCU VM. +pub(crate) struct Firmware<'bound> { + /// Iomem need to access registers. + iomem: Arc>, + + /// MCU VM. + vm: Arc>, + + /// List of firmware sections. + #[expect(dead_code)] + sections: KVec>, +} + +impl<'bound> Drop for Firmware<'bound> { + fn drop(&mut self) { + // Stop the MCU before releasing its firmware mappings and memory. + let _ =3D self.stop(); + + // AS slots retain a VM ref, we need to kill the circular ref manu= ally. + self.vm.kill(); + } +} + +impl<'bound> Firmware<'bound> { + fn init_section_mem(dev: &Device, mem: &mut KernelBo<'bound>, data: &K= Vec) -> Result { + if data.is_empty() { + return Ok(()); + } + + let vmap =3D mem.bo().vmap::<0>()?; + let size =3D mem.bo().size(); + + if data.len() > size { + dev_err!(dev, "fw section {} bigger than BO {}", data.len(), s= ize); + return Err(EINVAL); + } + + for (i, &byte) in data.iter().enumerate() { + vmap.try_write8(byte, i)?; + } + + Ok(()) + } + + fn request(ddev: &TyrDrmDevice, gpu_info: &GpuInfo) -> Result { + let gpu_id =3D GPU_ID::from_raw(gpu_info.gpu_id); + + let path =3D CString::try_from_fmt(fmt!( + "arm/mali/arch{}.{}/mali_csffw.bin", + gpu_id.arch_major().get(), + gpu_id.arch_minor().get() + ))?; + + kernel::firmware::Firmware::request(&path, ddev.as_ref().as_ref()) + } + + fn load( + dev: &Device, + ddev: &TyrDrmDevice, + gpu_info: &GpuInfo, + ) -> Result<(kernel::firmware::Firmware, KVec)> { + let fw =3D Self::request(ddev, gpu_info)?; + let mut parser =3D FwParser::new(dev, fw.data()); + + let parsed_sections =3D parser.parse()?; + + Ok((fw, parsed_sections)) + } + + /// Load firmware and map sections into MCU VM. + pub(crate) fn new( + dev: &'bound Device, + iomem: Arc>, + ddev: &TyrDrmDevice, + mmu: ArcBorrow<'_, Mmu<'bound>>, + gpu_info: &GpuInfo, + ) -> Result> { + let vm =3D Vm::new(dev, ddev, mmu, gpu_info)?; + vm.activate()?; + + let result =3D (|| { + let (fw, parsed_sections) =3D Self::load(dev, ddev, gpu_info)?; + let mut sections =3D KVec::new(); + for parsed in parsed_sections { + let size =3D u64::from(parsed.va.end.checked_sub(parsed.va= .start).ok_or(EINVAL)?); + + let va =3D u64::from(parsed.va.start); + + let mut mem =3D KernelBo::new( + ddev, + vm.clone(), + size, + KernelBoVaAlloc::Explicit(va), + parsed.vm_map_flags, + )?; + + let section_start =3D parsed.data_range.start as usize; + let section_end =3D parsed.data_range.end as usize; + let mut data =3D KVec::new(); + + // Ensure that the firmware slice is not out of bounds. + let fw_data =3D fw.data(); + let bytes =3D fw_data.get(section_start..section_end).ok_o= r(EINVAL)?; + data.extend_from_slice(bytes, GFP_KERNEL)?; + + Self::init_section_mem(dev, &mut mem, &data)?; + + sections.push(Section { data, mem }, GFP_KERNEL)?; + } + + Ok(Firmware { + iomem, + vm: vm.clone(), + sections, + }) + })(); + + if result.is_err() { + vm.kill(); + } + + result + } + + pub(crate) fn boot(&self) -> Result { + let io =3D &self.iomem; + + // Discard any stale global interrupt. + io.write_reg(JOB_IRQ_CLEAR::zeroed().with_glb(true)); + + io.write_reg(MCU_CONTROL::zeroed().with_req(McuControlMode::Auto)); + + if let Err(e) =3D poll::read_poll_timeout( + || Ok((io.read(MCU_STATUS), io.read(JOB_IRQ_RAWSTAT))), + |(mcu_status, irq_rawstat)| { + mcu_status.value() =3D=3D McuStatus::Enabled && irq_rawsta= t.glb() + }, + time::Delta::from_millis(1), + time::Delta::from_millis(100), + ) { + let status =3D io.read(MCU_STATUS); + dev_err!( + self.vm.dev(), + "MCU failed to boot, status: {:?}", + status.value() + ); + return Err(e); + } + + io.write_reg(JOB_IRQ_CLEAR::zeroed().with_glb(true)); + + Ok(()) + } + + fn stop(&self) -> Result { + let io =3D &self.iomem; + io.write_reg(MCU_CONTROL::zeroed().with_req(McuControlMode::Disabl= e)); + + if let Err(e) =3D poll::read_poll_timeout( + || Ok(io.read(MCU_STATUS)), + |status| status.value() =3D=3D McuStatus::Disabled, + time::Delta::from_micros(10), + time::Delta::from_millis(100), + ) { + let status =3D io.read(MCU_STATUS); + dev_err!( + self.vm.dev(), + "MCU failed to stop, status: {:?}", + status.value() + ); + return Err(e); + } + + Ok(()) + } +} diff --git a/drivers/gpu/drm/tyr/gem.rs b/drivers/gpu/drm/tyr/gem.rs index b371299028d6..342d3303a0c0 100644 --- a/drivers/gpu/drm/tyr/gem.rs +++ b/drivers/gpu/drm/tyr/gem.rs @@ -72,7 +72,6 @@ pub(crate) fn new_dummy_object(ddev: &TyrDrmDevice) -> Re= sult> { /// An automatic VA allocation strategy will be added in the future. pub(crate) enum KernelBoVaAlloc { /// Explicit VA address specified by the caller. - #[expect(dead_code)] Explicit(u64), } =20 @@ -98,7 +97,6 @@ impl<'bound> KernelBo<'bound> { /// This function allocates a new shmem-backed GEM object and immediat= ely maps /// it into the specified GPU virtual memory space. The mapping is aut= omatically /// cleaned up when the [`KernelBo`] is dropped. - #[expect(dead_code)] pub(crate) fn new( ddev: &TyrDrmDevice, vm: Arc>, @@ -135,7 +133,6 @@ pub(crate) fn new( }) } =20 - #[expect(dead_code)] pub(crate) fn bo(&self) -> &Bo { &self.bo } diff --git a/drivers/gpu/drm/tyr/tyr.rs b/drivers/gpu/drm/tyr/tyr.rs index 92f6885cdaae..e7ec450bdc9c 100644 --- a/drivers/gpu/drm/tyr/tyr.rs +++ b/drivers/gpu/drm/tyr/tyr.rs @@ -9,6 +9,7 @@ =20 mod driver; mod file; +mod fw; mod gem; mod gpu; mod mmu; diff --git a/drivers/gpu/drm/tyr/vm.rs b/drivers/gpu/drm/tyr/vm.rs index c113820b5505..418f98ab07fb 100644 --- a/drivers/gpu/drm/tyr/vm.rs +++ b/drivers/gpu/drm/tyr/vm.rs @@ -6,7 +6,6 @@ //! the illusion of owning the entire virtual address (VA) range, similar = to CPU virtual memory. //! Each virtual memory (VM) area is backed by ARM64 LPAE Stage 1 page tab= les and can be //! mapped into hardware address space (AS) slots for GPU execution. -#![expect(dead_code)] =20 use core::marker::PhantomData; use core::ops::Range; --=20 2.55.0