From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id C893142AFB2; Wed, 5 Aug 2026 11:03:42 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927823; cv=none; b=epqAlLk4sAjAE4GRAGcC2d56BP8mVvpzI23a98qamNXZncKS0qJrl2dswLQqS54Ey4bLiYl7sbhvJY0PSEvnSfgGcgV+mEDr/RKMbzg6U6PcvDmcNJvVgbKDlMZaiS3bYsnLlQoRCckUJtw4miPIlSM3FeAeCA6c4LtUNyao6jM= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927823; c=relaxed/simple; bh=G2ucZGhu25RP2ebqVdqrXB2AEnPOxIw1b/0uzx4NuJg=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=BrFpPI1B7wrPBpM/IVMM5/q0o+h92ldQfdbxRMfxGJKgUJD0/wFpW0dtmbFIFkgfNHXH+u7OZ7cA9b98nwcBdqaVyH+6ENWETFQM6FeL/oIz7y6GzRfu4KLeREqX5qk08Fgw5hn6tideQ7wZ9JJ0uoMXJiWAjXANVLcXYEekzPQ= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=bM3kI5f3; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="bM3kI5f3" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 6092320B716B; Wed, 5 Aug 2026 04:03:21 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 6092320B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927801; bh=uoE2oQuYpg52deZOM9SASjntNtYyGWIXBhdOKD8iu3g=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=bM3kI5f3p1CK0Y2PpOX1FqfwO8P1dfLWg7uriIJeo/IGEVTKPh+n38kalb5B17XEU fAnJnqzwhvAS9j/nD32LPbjeUDTHe0HuEL6iPnycbmGU5TWzKwfLNoFkQEcImuwpfy JAoHoBnSd+I/S0kxKaa9LoDhBeRAM7TJKbNA2FEo= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 01/42] Fix merge issue - Remove duplicate definition for kvm_arch_has_irq_bypass Date: Wed, 5 Aug 2026 04:02:43 -0700 Message-ID: <20260805110324.25067-2-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" From: Sriram Nambakam --- arch/x86/include/asm/kvm_host.h | 5 ----- 1 file changed, 5 deletions(-) diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_hos= t.h index 979d1749f176..b7d478dcc1a5 100644 --- a/arch/x86/include/asm/kvm_host.h +++ b/arch/x86/include/asm/kvm_host.h @@ -2603,11 +2603,6 @@ static inline bool kvm_arch_has_irq_bypass(void) return enable_device_posted_irqs; } =20 -static inline bool kvm_arch_has_irq_bypass(void) -{ - return enable_device_posted_irqs; -} - int kvm_arch_nr_vcpu_planes(struct kvm *kvm); bool kvm_arch_planes_share_fpu(struct kvm *kvm); =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 93693421232; Wed, 5 Aug 2026 11:03:43 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927824; cv=none; b=SGjNNIHN8Ww15744EyVuR6YhxjtMhbSIFegNedXQ/cxscavH9trN1Cu8XQlOh7ALquy4hdEbfeJ+OzdwCrzxES80yahwK0FfxxIhvmKTP2jZpzyHLxhrTEzcg+5OCFbYlkpeHbOB0OOjzJpSzjEHgJz5Gle1JHzfhgcLUoUXCrc= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927824; c=relaxed/simple; bh=3Y/tRcdbU+t1tVcQ91xWggoBpDmuM74MboE1DFiASVc=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=UCvj/lD/xHcAyBTkXNeeSqBQhHlvUPODmXddKotxP65o2ZcIrrqxd1PlL7YVnQJKSaVBLVtcQ0TuZHiJuoq1U76M02sP8eRzel/MXITjAXQQ/GuH7/YeZ9rilvvA8g3lGuwAtmRFDLHhbawQIkHEA3YeSYaYNM7j4HcSplkLW0Y= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=YoWTgQAB; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="YoWTgQAB" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 66DE720B716C; Wed, 5 Aug 2026 04:03:22 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 66DE720B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927802; bh=w57Ramh9RwzEf5GMKjkCKRc6g8SfosvFKziATrRDyoo=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=YoWTgQAB/J3P7QaG3DoEJWiJ26nY5opBXD8/TBJLtQ07IMDh62lAbMM2C/2Onb3pt 9d06G0Hkx3kme91gjchp9KlHSNGWs/uzE4X4AcSsrSJTonUmhqbzv9zgIsUfs3FWqt 8jec1iZraQBn4FSSPbYFfolA3DfXjbWqY5AIDBDI= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 02/42] Fix compilation Date: Wed, 5 Aug 2026 04:02:44 -0700 Message-ID: <20260805110324.25067-3-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" From: Sriram Nambakam --- arch/x86/kvm/irq_comm.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/arch/x86/kvm/irq_comm.c b/arch/x86/kvm/irq_comm.c index 94f9db50384e..55f395f9d8af 100644 --- a/arch/x86/kvm/irq_comm.c +++ b/arch/x86/kvm/irq_comm.c @@ -122,7 +122,7 @@ void kvm_set_msi_irq(struct kvm *kvm, struct kvm_kernel= _irq_routing_entry *e, irq->shorthand =3D APIC_DEST_NOSHORT; irq->plane =3D e->msi.plane; } -EXPORT_SYMBOL_GPL(kvm_set_msi_irq); +EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_set_msi_irq); =20 static inline bool kvm_msi_route_invalid(struct kvm *kvm, struct kvm_kernel_irq_routing_entry *e) @@ -369,7 +369,7 @@ bool kvm_intr_is_single_vcpu(struct kvm *kvm, struct kv= m_lapic_irq *irq, =20 return r =3D=3D 1; } -EXPORT_SYMBOL_GPL(kvm_intr_is_single_vcpu); +EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_intr_is_single_vcpu); =20 #define IOAPIC_ROUTING_ENTRY(irq) \ { .gsi =3D irq, .type =3D KVM_IRQ_ROUTING_IRQCHIP, \ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id D1C4B42CB0D; Wed, 5 Aug 2026 11:03:44 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927827; cv=none; b=QcZ4rKpj6EB09np2oIhj90QxlN6V+ZVwQ3agnz0JbBWwiaZOmnuHn57656VtbwIEX4qRNXqccfV9FJ8LnTULpkzjAXf+6KymacgbK9f4Lcg6LdmNV3I1wrPVSVm1I46e/FWjy9+5Ku7SjrfV3F1aeEtvnZtS/UG7fhnJEU/IUF4= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927827; c=relaxed/simple; bh=HDJF478nIMab/5S+MkniuOJL93na06AXHZuDa3J6vus=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=jSLx/m44mEVtJkqgjVn/KE5MmcIeM7MVPlaLBPhjGrhtaC6quGHMcxWm/fpsA0GifwQKTevDuS9Us9X0VbeWY83ooTTf/0V6sFB6PudcSITqnffCEN0s6zjTJWCDo9ABfA5FFLCCrvZyPFdMQJ9CcMO0nQKxFpnUmKBIqYAnvfk= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=QbNx3Bbf; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="QbNx3Bbf" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 3B71C20B7169; Wed, 5 Aug 2026 04:03:23 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 3B71C20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927803; bh=27M/0RuBaS/3dfbGI7oEaAvgunuUxuFl5l+1w3bLmH4=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=QbNx3BbfgQ2uB0omiXEwGDP2Gx+INUeioNpiadyLyLcvMRcSDlRJtAuT0dO0M3Y5w sEQPjbxHMAF1LyKYDquyBAa14NjXfKWf7MFVuigdD/4BhTUj58Quh1wc6RPkk8v3jN oFIrRxpM/kbik0fF+AQsIm4CQh25j0W3I/3PMVB4= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 03/42] Fix compile error Date: Wed, 5 Aug 2026 04:02:45 -0700 Message-ID: <20260805110324.25067-4-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" From: Sriram Nambakam --- arch/x86/kvm/vmx/vmx.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/x86/kvm/vmx/vmx.c b/arch/x86/kvm/vmx/vmx.c index 03e1ce935799..ee1d606e3314 100644 --- a/arch/x86/kvm/vmx/vmx.c +++ b/arch/x86/kvm/vmx/vmx.c @@ -417,7 +417,7 @@ static noinstr void vmx_l1d_flush(struct kvm_vcpu *vcpu) kvm_clear_cpu_l1tf_flush_l1d(); } =20 - vcpu->stat.l1d_flush++; + vcpu->stat->l1d_flush++; =20 if (static_cpu_has(X86_FEATURE_FLUSH_L1D)) { native_wrmsrq(MSR_IA32_FLUSH_CMD, L1D_FLUSH); --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id CBA6B430CC0; Wed, 5 Aug 2026 11:03:45 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927829; cv=none; b=sRduvTX3RqQP9786uPpNJy+cNt0K+zyrEsu2Ce4lo1Tcz7UcqWd8ag+W7fnYuK29p+dNO9UDsvoOh7VbeloigjDpk+XuawWETM61tfJK6q3TJ8fNtVWq697UWOjxnZ0/sFYYZJnw9HHRuyQAh70zKTBYxPjyNuI+/AldsoR5Yko= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927829; c=relaxed/simple; bh=mWlBfRBicvr9Sp/4GK1gHvtyrXEItl+YRYkQ6JZlR/M=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=s1PPqnI4ltdDyhNpLK+xut4s5hWlEu5mkTZyWVhgq9hM1KNbTwC1BZ3kR32z6gnAp0/yKW0h8FF+HkWOWoagURkSXss3A5DSfBskkzCwo49s23mkgz2usSeVQGKhDbQoUReiX8xCvdEOzCTRfpuRinGjnRUKNzoanpW1DwAXwa4= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=PQjbxCLh; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="PQjbxCLh" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 7128A20B716A; Wed, 5 Aug 2026 04:03:24 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 7128A20B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927804; bh=GdugUXgaw/RRxSTb02hZRqyEZ8o4eEzaX+dPB+NEUjc=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=PQjbxCLha3GNCGWoMc6FrQRIw0ZqcYGqe3Yv9cp8t7mHodaLU6tqW9uEA+fII+bbw 1C/uaybt57CvxxOCMHUPCtIlR0SN0hz17F0UCE/Sbgxv6ULogiMwKAJzE4RCvOTUV3 CoDoCi4bsidQtc8CidTH3Q19u6fLMlmu4v8274dk= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 04/42] Fix compile errors Date: Wed, 5 Aug 2026 04:02:46 -0700 Message-ID: <20260805110324.25067-5-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" From: Sriram Nambakam --- arch/x86/kvm/svm/avic.c | 2 +- arch/x86/kvm/svm/sev.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/arch/x86/kvm/svm/avic.c b/arch/x86/kvm/svm/avic.c index 58e493a80cb0..251e36f5f0f7 100644 --- a/arch/x86/kvm/svm/avic.c +++ b/arch/x86/kvm/svm/avic.c @@ -404,7 +404,7 @@ static int avic_init_backing_page(struct kvm_vcpu *vcpu) * fully initialized AVIC. */ if (id > max_id) { - kvm_set_apicv_inhibit(vcpu->kvm, APICV_INHIBIT_REASON_PHYSICAL_ID_TOO_BI= G); + kvm_set_apicv_inhibit(vcpu->kvm->planes[0], APICV_INHIBIT_REASON_PHYSICA= L_ID_TOO_BIG); vcpu->arch.apic->apicv_active =3D false; return 0; } diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index fe5e05e2162d..79fee7ebc19b 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -569,7 +569,7 @@ static int __sev_guest_init(struct kvm *kvm, struct kvm= _sev_cmd *argp, INIT_LIST_HEAD(&sev->mirror_vms); sev->need_init =3D false; =20 - kvm_set_apicv_inhibit(kvm->planes[[0], APICV_INHIBIT_REASON_SEV); + kvm_set_apicv_inhibit(kvm->planes[0], APICV_INHIBIT_REASON_SEV); =20 return 0; =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 167BA429CC7; Wed, 5 Aug 2026 11:03:47 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927828; cv=none; b=Sr6IYKbJPyad8rMRBKoIzwD/NQ58CPoZv4Tx81b3rOu2jlmzFYvLR6SX6b5NTv5yrN9WdzSR75GlpvxmaAHj73h/7+4Lc7zng0uPvH7ttb0CGYtOfZ+6jg1PSic1yNb51OeTG0hV6mYSu5Yynf63Wf5/VufqVPuXetT/ryH+zss= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927828; c=relaxed/simple; bh=UVA6bsrvHu8PAPd6wcbwv+Ktpzqk0/zXguvpb1dB0QE=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=ZhUaln4IbnrZ9H00QjKbm0DRsFuKbB6MSmSSkQVPeNGjrPhaWW/pZOky2TN+65I1q/27m5M+9cni272di9/ljZXBaLFx1abLMG1MM4b0/k76UYTp7aPaEfr/bTG7/ujj/a0BjQqeb1Us0kCxjzK7av9sTMhTFqEQ9uIEL67qcvg= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=aiK/KsTS; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="aiK/KsTS" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 77FE920B716B; Wed, 5 Aug 2026 04:03:25 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 77FE920B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927805; bh=AJ1UHw5uzjpw7l7HTWL+cTQ09mxfzq4yQBen6QoG6XA=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=aiK/KsTSHjCk0YpDVsLehFqSa82TGSvx5cYUNCUv7+oe5dj7OkOhKDwXcm0/vmVCy F2m69HpewmZ0gh0aRKyBUonPTyG4rxqqiTipBtBJfSrrTFvqPjSVWq+whSibEqG3U7 1rxEouF7SdS40Igs+bG2ZC4zwWv3PI6EWSt3USns= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 05/42] Initial support for VM Planes - Add kernel config for CONFIG_VM_PLANES - Parse vm plane config from initrd for plane configuration - Make hypercalls to allocate memory for the vm planes. Date: Wed, 5 Aug 2026 04:02:47 -0700 Message-ID: <20260805110324.25067-6-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- arch/x86/include/asm/cpu.h | 8 + arch/x86/kernel/cpu/common.c | 38 ++++ include/linux/vm_planes.h | 22 ++ init/Kconfig | 16 ++ init/Makefile | 1 + init/main.c | 4 + init/vm_planes.c | 417 +++++++++++++++++++++++++++++++++++ 7 files changed, 506 insertions(+) create mode 100644 include/linux/vm_planes.h create mode 100644 init/vm_planes.c diff --git a/arch/x86/include/asm/cpu.h b/arch/x86/include/asm/cpu.h index 57a0786dfd75..8ab76adf14a9 100644 --- a/arch/x86/include/asm/cpu.h +++ b/arch/x86/include/asm/cpu.h @@ -4,11 +4,19 @@ =20 #include #include +#include #include #include #include #include =20 +#ifdef CONFIG_VM_PLANES +struct vm_plane_config; + +void __init alloc_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg); +#endif + #ifndef CONFIG_SMP #define cpu_physical_id(cpu) boot_cpu_physical_apicid #endif /* CONFIG_SMP */ diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c index a3df21d26460..166597204739 100644 --- a/arch/x86/kernel/cpu/common.c +++ b/arch/x86/kernel/cpu/common.c @@ -20,11 +20,13 @@ #include #include #include +#include #include #include #include #include #include +#include #include #include #include @@ -32,6 +34,7 @@ #include #include #include +#include #include #include #include @@ -77,6 +80,11 @@ =20 #include "cpu.h" =20 +#ifdef CONFIG_VM_PLANES +/* Private hypercall number for early VM plane configuration. */ +#define KVM_HC_VM_PLANES_CONFIG 0x1000 +#endif + DEFINE_PER_CPU_READ_MOSTLY(struct cpuinfo_x86, cpu_info); EXPORT_PER_CPU_SYMBOL(cpu_info); =20 @@ -2664,3 +2672,33 @@ void __init arch_cpu_finalize_init(void) */ mem_encrypt_init(); } + +#ifdef CONFIG_VM_PLANES +void __init alloc_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg) +{ + phys_addr_t phys; + long ret; + + if (!plane_count || !plane_cfg) + return; + + if (!kvm_para_available()) { + pr_warn("vm_planes: hypercall interface unavailable\n"); + return; + } + + phys =3D virt_to_phys((void *)plane_cfg); + + if (sizeof(unsigned long) < sizeof(phys_addr_t) && phys > ULONG_MAX) { + pr_warn("vm_planes: shared config address exceeds hypercall register wid= th\n"); + return; + } + + ret =3D kvm_hypercall2(KVM_HC_VM_PLANES_CONFIG, + (unsigned long)phys, + plane_count); + if (ret < 0) + pr_warn("vm_planes: hypercall failed: %ld\n", ret); +} +#endif diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h new file mode 100644 index 000000000000..4a91dc79b7ba --- /dev/null +++ b/include/linux/vm_planes.h @@ -0,0 +1,22 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef _LINUX_VM_PLANES_H +#define _LINUX_VM_PLANES_H + +#include +#include + +#ifdef CONFIG_VM_PLANES + +#define VM_PLANE_KERNEL_NAME_MAX 128 + +struct vm_plane_config { + phys_addr_t load_offset; + phys_addr_t memory_size; + char kernel[VM_PLANE_KERNEL_NAME_MAX]; +}; + +void __init arch_init_vm_planes(void); + +#endif /* CONFIG_VM_PLANES */ + +#endif /* _LINUX_VM_PLANES_H */ diff --git a/init/Kconfig b/init/Kconfig index 10f2013b5321..23d9cca334ba 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1713,6 +1713,22 @@ menuconfig EXPERT environments which can tolerate a "non-standard" kernel. Only use this if you really know what you are doing. =20 +config VM_PLANES + bool "Enable VM planes early boot support" if EXPERT + default n + help + Enable hypervisor enabled multi-kernel support. + + This allows processing the kernel command-line parameter + "enable-vm-planes" and, when requested, calling + arch_init_vm_planes() during start_kernel(). + + The initrd config-vm-planes file is expected to provide per-plane + entries for PLANE__KERNEL, PLANE__LOAD_OFFSET, and + PLANE__MEMORY_SIZE. + + If unsure, say N. + config UID16 bool "Enable 16-bit UID system calls" if EXPERT depends on HAVE_UID16 && MULTIUSER diff --git a/init/Makefile b/init/Makefile index d6f75d8907e0..113133c8cdd7 100644 --- a/init/Makefile +++ b/init/Makefile @@ -6,6 +6,7 @@ ccflags-y :=3D -fno-function-sections -fno-data-sections =20 obj-y :=3D main.o version.o mounts.o +obj-y +=3D vm_planes.o ifneq ($(CONFIG_BLK_DEV_INITRD),y) obj-y +=3D noinitramfs.o else diff --git a/init/main.c b/init/main.c index e363232b428b..3e35c2caca17 100644 --- a/init/main.c +++ b/init/main.c @@ -40,6 +40,7 @@ #include #include #include +#include #include #include #include @@ -993,6 +994,9 @@ void start_kernel(void) pr_notice("%s", linux_banner); setup_arch(&command_line); mm_core_init_early(); +#ifdef CONFIG_VM_PLANES + arch_init_vm_planes(); +#endif /* Static keys and static calls are needed by LSMs */ jump_label_init(); static_call_init(); diff --git a/init/vm_planes.c b/init/vm_planes.c new file mode 100644 index 000000000000..30f17da92332 --- /dev/null +++ b/init/vm_planes.c @@ -0,0 +1,417 @@ +// SPDX-License-Identifier: GPL-2.0-only + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#ifdef CONFIG_VM_PLANES +static bool __initdata enable_vm_planes_requested; + +#define VM_PLANES_CONFIG_FILE "config-vm-planes" +#define VM_PLANES_DEFAULT_COUNT 1 + +struct vm_plane_parse_state { + phys_addr_t have_load_offset; + phys_addr_t have_memory_size; + char have_kernel[VM_PLANE_KERNEL_NAME_MAX]; +}; + +#define VM_PLANES_UNSET_VALUE ((phys_addr_t)~0) + +struct cpio_newc_header { + char c_magic[6]; + char c_ino[8]; + char c_mode[8]; + char c_uid[8]; + char c_gid[8]; + char c_nlink[8]; + char c_mtime[8]; + char c_filesize[8]; + char c_devmajor[8]; + char c_devminor[8]; + char c_rdevmajor[8]; + char c_rdevminor[8]; + char c_namesize[8]; + char c_check[8]; +}; + +static int __init parse_hex_field(const char *field, size_t len, u32 *valu= e) +{ + u32 v =3D 0; + size_t i; + + for (i =3D 0; i < len; i++) { + u8 c =3D field[i]; + + v <<=3D 4; + if (c >=3D '0' && c <=3D '9') + v |=3D c - '0'; + else if (c >=3D 'a' && c <=3D 'f') + v |=3D c - 'a' + 10; + else if (c >=3D 'A' && c <=3D 'F') + v |=3D c - 'A' + 10; + else + return -EINVAL; + } + + *value =3D v; + return 0; +} + +static int __init parse_plane_count_line(const char *line, size_t len, + unsigned int *plane_count) +{ + const char *keys[] =3D { "PLANE_COUNT=3D", "CONFIG_PLANE_COUNT=3D" }; + unsigned int i; + + while (len && (*line =3D=3D ' ' || *line =3D=3D '\t')) { + line++; + len--; + } + + if (!len || *line =3D=3D '#') + return -ENOENT; + + for (i =3D 0; i < ARRAY_SIZE(keys); i++) { + size_t key_len =3D strlen(keys[i]); + size_t val_len =3D 0; + char tmp[32]; + + if (len <=3D key_len || strncmp(line, keys[i], key_len)) + continue; + + line +=3D key_len; + len -=3D key_len; + while (val_len < len && line[val_len] !=3D ' ' && + line[val_len] !=3D '\t' && line[val_len] !=3D '#') + val_len++; + + if (!val_len || val_len >=3D sizeof(tmp)) + return -EINVAL; + + memcpy(tmp, line, val_len); + tmp[val_len] =3D '\0'; + + if (kstrtouint(tmp, 0, plane_count)) + return -EINVAL; + if (!*plane_count) + return -EINVAL; + + return 0; + } + + return -ENOENT; +} + +static int __init parse_plane_count_kconfig(const char *buf, size_t len, + unsigned int *plane_count) +{ + const char *p =3D buf; + const char *end =3D buf + len; + + while (p < end) { + const char *eol =3D memchr(p, '\n', end - p); + size_t line_len =3D eol ? (size_t)(eol - p) : (size_t)(end - p); + int ret =3D parse_plane_count_line(p, line_len, plane_count); + + if (!ret) + return 0; + + p +=3D line_len; + if (p < end && *p =3D=3D '\n') + p++; + } + + return -ENOENT; +} + +static int __init parse_plane_cfg_line(const char *line, size_t len, + unsigned int plane_count, + struct vm_plane_config *plane_cfg, + struct vm_plane_parse_state *state) +{ + char tmp[192]; + char *p, *key, *val; + unsigned int plane_id; + u64 parsed_u64; + phys_addr_t parsed; + + if (len >=3D sizeof(tmp)) + return -E2BIG; + + memcpy(tmp, line, len); + tmp[len] =3D '\0'; + + p =3D strim(tmp); + if (!*p || *p =3D=3D '#') + return -ENOENT; + + val =3D strchr(p, '#'); + if (val) + *val =3D '\0'; + p =3D strim(p); + if (!*p) + return -ENOENT; + + if (!strncmp(p, "CONFIG_", 7)) + p +=3D 7; + + if (strncmp(p, "PLANE_", 6)) + return -ENOENT; + p +=3D 6; + + key =3D strchr(p, '_'); + if (!key) + return -EINVAL; + *key++ =3D '\0'; + + if (kstrtouint(p, 10, &plane_id) || plane_id >=3D plane_count) + return -EINVAL; + + val =3D strchr(key, '=3D'); + if (!val) + return -EINVAL; + *val++ =3D '\0'; + + key =3D strim(key); + val =3D strim(val); + if (!*val) + return -EINVAL; + + if (!strcmp(key, "KERNEL")) { + size_t val_len =3D strlen(val); + + if (val[0] =3D=3D '"') { + if (val_len < 2 || val[val_len - 1] !=3D '"') + return -EINVAL; + val[val_len - 1] =3D '\0'; + val++; + val =3D strim(val); + } + + if (!*val) + return -EINVAL; + + if (strscpy(plane_cfg[plane_id].kernel, val, + sizeof(plane_cfg[plane_id].kernel)) < 0) + return -EINVAL; + + strscpy(state[plane_id].have_kernel, val, + sizeof(state[plane_id].have_kernel)); + return 0; + } + + if (kstrtou64(val, 0, &parsed_u64)) + return -EINVAL; + + if (parsed_u64 > (u64)VM_PLANES_UNSET_VALUE) + return -ERANGE; + + parsed =3D (phys_addr_t)parsed_u64; + + if (!strcmp(key, "LOAD_OFFSET")) { + plane_cfg[plane_id].load_offset =3D parsed; + state[plane_id].have_load_offset =3D parsed; + return 0; + } + + if (!strcmp(key, "MEMORY_SIZE")) { + plane_cfg[plane_id].memory_size =3D parsed; + state[plane_id].have_memory_size =3D parsed; + return 0; + } + + return -ENOENT; +} + +static int __init parse_vm_planes_kconfig(const char *buf, size_t len, + unsigned int *plane_count, + struct vm_plane_config **plane_cfg) +{ + const char *p =3D buf; + const char *end =3D buf + len; + struct vm_plane_parse_state *state; + unsigned int i; + int ret; + + ret =3D parse_plane_count_kconfig(buf, len, plane_count); + if (ret) + return ret; + + if (*plane_count > UINT_MAX / sizeof(**plane_cfg)) + return -E2BIG; + + *plane_cfg =3D memblock_alloc(*plane_count * sizeof(**plane_cfg), + SMP_CACHE_BYTES); + if (!*plane_cfg) + return -ENOMEM; + + state =3D memblock_alloc(*plane_count * sizeof(*state), SMP_CACHE_BYTES); + if (!state) + return -ENOMEM; + + memset(*plane_cfg, 0, *plane_count * sizeof(**plane_cfg)); + for (i =3D 0; i < *plane_count; i++) { + state[i].have_load_offset =3D VM_PLANES_UNSET_VALUE; + state[i].have_memory_size =3D VM_PLANES_UNSET_VALUE; + state[i].have_kernel[0] =3D '\0'; + } + + while (p < end) { + const char *eol =3D memchr(p, '\n', end - p); + size_t line_len =3D eol ? (size_t)(eol - p) : (size_t)(end - p); + + ret =3D parse_plane_cfg_line(p, line_len, *plane_count, + *plane_cfg, state); + if (ret && ret !=3D -ENOENT) + return ret; + + p +=3D line_len; + if (p < end && *p =3D=3D '\n') + p++; + } + + for (i =3D 0; i < *plane_count; i++) { + if (state[i].have_load_offset =3D=3D VM_PLANES_UNSET_VALUE || + state[i].have_memory_size =3D=3D VM_PLANES_UNSET_VALUE || + !state[i].have_kernel[0]) + return -EINVAL; + } + + return 0; +} + +static bool __init cpio_name_match(const char *name, size_t namesize, + const char *target) +{ + while (namesize > 1 && (*name =3D=3D '/' || + (namesize > 2 && name[0] =3D=3D '.' && name[1] =3D=3D '/'))) { + if (*name =3D=3D '/') { + name++; + namesize--; + } else { + name +=3D 2; + namesize -=3D 2; + } + } + + return !strncmp(name, target, namesize - 1) && + strlen(target) =3D=3D namesize - 1; +} + +static int __init vm_planes_get_cfg_from_initrd(unsigned int *plane_count, + struct vm_plane_config **plane_cfg) +{ + const u8 *p =3D (const u8 *)(unsigned long)initrd_start; + const u8 *end =3D (const u8 *)(unsigned long)initrd_end; + + if (!initrd_start || !initrd_end || initrd_end <=3D initrd_start) + return -ENOENT; + + while (p + sizeof(struct cpio_newc_header) <=3D end) { + const struct cpio_newc_header *hdr; + const char *name; + const u8 *data; + u32 namesize, filesize; + u32 name_align, data_align; + int ret; + + hdr =3D (const struct cpio_newc_header *)p; + if (memcmp(hdr->c_magic, "070701", 6) && + memcmp(hdr->c_magic, "070702", 6)) + return -EINVAL; + + ret =3D parse_hex_field(hdr->c_namesize, sizeof(hdr->c_namesize), &names= ize); + if (ret) + return ret; + + ret =3D parse_hex_field(hdr->c_filesize, sizeof(hdr->c_filesize), &files= ize); + if (ret) + return ret; + + if (!namesize) + return -EINVAL; + + p +=3D sizeof(*hdr); + if (p + namesize > end) + return -EINVAL; + + name =3D (const char *)p; + name_align =3D ALIGN(namesize, 4); + if (p + name_align > end) + return -EINVAL; + + data =3D p + name_align; + if (data + filesize > end) + return -EINVAL; + + if (!strcmp(name, "TRAILER!!!")) + break; + + if (cpio_name_match(name, namesize, VM_PLANES_CONFIG_FILE)) + return parse_vm_planes_kconfig((const char *)data, + filesize, + plane_count, + plane_cfg); + + data_align =3D ALIGN(filesize, 4); + if (data + data_align < data || data + data_align > end) + return -EINVAL; + + p =3D data + data_align; + } + + return -ENOENT; +} + +static int __init parse_enable_vm_planes(char *str) +{ + bool enable; + + if (!str) { + enable_vm_planes_requested =3D true; + return 0; + } + + if (kstrtobool(str, &enable)) + return -EINVAL; + + enable_vm_planes_requested =3D enable; + return 0; +} + +early_param("enable-vm-planes", parse_enable_vm_planes); + +void __init __weak alloc_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg) { } + +void __init arch_init_vm_planes(void) +{ + unsigned int plane_count =3D VM_PLANES_DEFAULT_COUNT; + struct vm_plane_config *plane_cfg; + + if (!enable_vm_planes_requested) + return; + + if (!kvm_para_available()) + return; + + if (vm_planes_get_cfg_from_initrd(&plane_count, &plane_cfg)) { + pr_warn("vm_planes: failed to parse %s from initrd\n", + VM_PLANES_CONFIG_FILE); + return; + } + + pr_info("vm_planes: enabling %u planes (ids 0..%u)\n", + plane_count, plane_count - 1); + alloc_vm_planes(plane_count, plane_cfg); +} + +#endif /* CONFIG_VM_PLANES */ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id DA96742FCDE; Wed, 5 Aug 2026 11:03:47 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927829; cv=none; b=k8UcmdOZ3pMt2HK1ua8Shioy5+4y53WynK6lBtKuFrv8bgpIClMnGKFiBrzGuniMMKMdK7wv+AntyWx83ZpZXfeOmR3P7X6Jlhti5BGc/AMBbjn53pZ/nfE3XWRT8mqS7Mo2QboV6kvgnsMLff7lAxOxJ3c+mWvtQKvKVeeeq0U= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927829; c=relaxed/simple; bh=7JSY13KGScBy6F7PuLTG7DsSzKo6AwkqgIvHgiWmu9g=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=DEC368IN+kTw2fWU9G3ul/vsGPRjVQg2JfBCnJovR3NFR8Pwck2E2mvQyhcKZs/snzZEfjrMY4aEvrIir6EdXn06Khiryic0SDMBiFDq4uo/s3ovMzndxdJP6HNImlhVPVwXKWC748g3Yi6432mz31wAD5GDG8H8sePyJBOFIxc= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=quv4WmNt; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="quv4WmNt" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id C71B220B716E; Wed, 5 Aug 2026 04:03:26 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com C71B220B716E DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927806; bh=iR/umNSTyojyj4H/NIwYEl6RjfILL3Ggtuvntnwdf+Y=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=quv4WmNtrW+l6vStDs0L6FWJJi/BFQAYRlCvjT425YS3WtVgaYehVLFLrZv+yObJI XzueknOQMmQoWREqUbdFxbiZ+wgRkdo1zig6SAvjLkoL43bNaMDEButGF49Pnv5qXf J8xis3OotsxnyspfvbzOitQWF0PVWhErHWvMjpIc= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 06/42] Use vcpu count from the plane configuration Date: Wed, 5 Aug 2026 04:02:48 -0700 Message-ID: <20260805110324.25067-7-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- include/linux/vm_planes.h | 1 + init/vm_planes.c | 37 ++++++++++++++++++++++++------------- 2 files changed, 25 insertions(+), 13 deletions(-) diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index 4a91dc79b7ba..506de566cbd5 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -12,6 +12,7 @@ struct vm_plane_config { phys_addr_t load_offset; phys_addr_t memory_size; + unsigned int vcpu_count; char kernel[VM_PLANE_KERNEL_NAME_MAX]; }; =20 diff --git a/init/vm_planes.c b/init/vm_planes.c index 30f17da92332..8eafb1c5f8cf 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -18,9 +18,10 @@ static bool __initdata enable_vm_planes_requested; #define VM_PLANES_DEFAULT_COUNT 1 =20 struct vm_plane_parse_state { - phys_addr_t have_load_offset; - phys_addr_t have_memory_size; - char have_kernel[VM_PLANE_KERNEL_NAME_MAX]; + phys_addr_t load_offset; + phys_addr_t memory_size; + unsigned int vcpu_count; + char kernel[VM_PLANE_KERNEL_NAME_MAX]; }; =20 #define VM_PLANES_UNSET_VALUE ((phys_addr_t)~0) @@ -203,8 +204,8 @@ static int __init parse_plane_cfg_line(const char *line= , size_t len, sizeof(plane_cfg[plane_id].kernel)) < 0) return -EINVAL; =20 - strscpy(state[plane_id].have_kernel, val, - sizeof(state[plane_id].have_kernel)); + strscpy(state[plane_id].kernel, val, + sizeof(state[plane_id].kernel)); return 0; } =20 @@ -218,13 +219,21 @@ static int __init parse_plane_cfg_line(const char *li= ne, size_t len, =20 if (!strcmp(key, "LOAD_OFFSET")) { plane_cfg[plane_id].load_offset =3D parsed; - state[plane_id].have_load_offset =3D parsed; + state[plane_id].load_offset =3D parsed; return 0; } =20 if (!strcmp(key, "MEMORY_SIZE")) { plane_cfg[plane_id].memory_size =3D parsed; - state[plane_id].have_memory_size =3D parsed; + state[plane_id].memory_size =3D parsed; + return 0; + } + + if (!strcmp(key, "VCPU_COUNT")) { + if (parsed_u64 =3D=3D 0 || parsed_u64 > UINT_MAX) + return -EINVAL; + plane_cfg[plane_id].vcpu_count =3D (unsigned int)parsed_u64; + state[plane_id].vcpu_count =3D (unsigned int)parsed_u64; return 0; } =20 @@ -259,9 +268,10 @@ static int __init parse_vm_planes_kconfig(const char *= buf, size_t len, =20 memset(*plane_cfg, 0, *plane_count * sizeof(**plane_cfg)); for (i =3D 0; i < *plane_count; i++) { - state[i].have_load_offset =3D VM_PLANES_UNSET_VALUE; - state[i].have_memory_size =3D VM_PLANES_UNSET_VALUE; - state[i].have_kernel[0] =3D '\0'; + state[i].load_offset =3D VM_PLANES_UNSET_VALUE; + state[i].memory_size =3D VM_PLANES_UNSET_VALUE; + state[i].vcpu_count =3D 0; + state[i].kernel[0] =3D '\0'; } =20 while (p < end) { @@ -279,9 +289,10 @@ static int __init parse_vm_planes_kconfig(const char *= buf, size_t len, } =20 for (i =3D 0; i < *plane_count; i++) { - if (state[i].have_load_offset =3D=3D VM_PLANES_UNSET_VALUE || - state[i].have_memory_size =3D=3D VM_PLANES_UNSET_VALUE || - !state[i].have_kernel[0]) + if (state[i].load_offset =3D=3D VM_PLANES_UNSET_VALUE || + state[i].memory_size =3D=3D VM_PLANES_UNSET_VALUE || + !state[i].vcpu_count || + !state[i].kernel[0]) return -EINVAL; } =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id D62F34314AB; Wed, 5 Aug 2026 11:03:48 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927830; cv=none; b=pRuRcF2I50AEgeKxdlGkxViq1ZxLP0ylkQvouBpabF/3K/Qe3cQStLxrYoOgOgtNe0JxPRj24UGdmQO8rcSNONjHk6wGBhugYUuNGPTIlMS16EyLbPLTvKBh5Xc++yVddsZje/HiFO50DoY0gBZG6CVI/aMjqEkFC9mPOO/+AAE= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927830; c=relaxed/simple; bh=NCzTAqPe1CPtOwyiPAfxyC1Kikic7YuzixYDfQcexK8=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=MFZdl1yORDTQicj6Ix0GVm+gB3WkS9ttVO5037eGR1GTHosOQzfMHYYcmvjTMu/xUxirep1WlRTCnQ6zCx3yAssUevQ2IbTrpeQ4AukW3rwagSdu+F8GG32sGnCCK0cQ6D4W1HTVg9Kjrmag58u8+0sopYEoDyQG9F6uVk0ZlOs= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=CrI/cQwf; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="CrI/cQwf" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 9C3CC20B716C; Wed, 5 Aug 2026 04:03:27 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 9C3CC20B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927807; bh=nngqIvDsOttwEddUkU8UUGrWpj5Z4UPAk42CUedgjOc=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=CrI/cQwfScXAxgW60vbQMlzsRJ8KFnD6wuMdj3masEuzWY/f5N5N0vlePzXrVanmR jV7FDlb3fYwS6mhf2PRRjUrMhyPbIttBQgfOmXlR4uJSbMMbh/Lcj4wHb4XtMJZuvX ynF7/a0svMheQ25LmIMpCoXVUQAxDcOsam8+6yz8= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 07/42] skip processing plane configuration for plane 0 - plane 0 is the boot plane Date: Wed, 5 Aug 2026 04:02:49 -0700 Message-ID: <20260805110324.25067-8-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- init/vm_planes.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/init/vm_planes.c b/init/vm_planes.c index 8eafb1c5f8cf..ecfe8de409a8 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -288,7 +288,10 @@ static int __init parse_vm_planes_kconfig(const char *= buf, size_t len, p++; } =20 - for (i =3D 0; i < *plane_count; i++) { + /* Plane 0 is the already-running boot plane; only secondary planes + * must provide full allocation metadata. + */ + for (i =3D 1; i < *plane_count; i++) { if (state[i].load_offset =3D=3D VM_PLANES_UNSET_VALUE || state[i].memory_size =3D=3D VM_PLANES_UNSET_VALUE || !state[i].vcpu_count || --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id B49734322F0; Wed, 5 Aug 2026 11:03:49 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927830; cv=none; b=dUnMfNrQLK6iTmm2oV9fDS23VXvPXxPkojB3AMImDsXypG6Y23Y4eES8+zzJVwAqEE+T976EMfkXOaBG7XOY4IME2RUrXrKMcOIXVGxQBQIPLEkL3ZLrK4sau54kEa9w/SbNiZLoSN3eSy6gM8ouKybasw8vthYkN9VjX3MfwVA= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927830; c=relaxed/simple; bh=VYNZzZDfqdSOu7Tj0KRWvIM0pNL5XYyX5tJhmoeWU1o=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=YIFE1/qzyF/ZX/yzqWGO3ssWRdDZyry6gqh8OtATn8lY6NhD3NQWaUSbVDv8ceqKuxO7wH2B0UqJnRnBW6sCnZvUMvjoRIAMT9WrdMYGzMzkGfg2TW64xV7l9tJIpPzw7Jxk19GQKxXDg1vv+cvfND+dmdwOISmB0QJRwzKyLZw= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=GcRBX0jD; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="GcRBX0jD" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id B97AD20B7169; Wed, 5 Aug 2026 04:03:28 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com B97AD20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927808; bh=/YcaorVByUQCvbkLO88pJPdr2y/0TEfivReyfVIvkQw=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=GcRBX0jDljIsTTfRfQ7YEzdimdta21ku4VQ1/bkvZugLCM3GjLsFuVQ81aAiBCI0S KtwYxIEeVFLY6h3hEQP5MpM/bSDOsqtSjDxNhHKpg3r9VA7NtbStWRosErltvQfeSD p25Xepxfao+6C2S05POhmLNeHqSCjSJQuPpie+1g= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 08/42] Add plane config param to specify kernel image format Date: Wed, 5 Aug 2026 04:02:50 -0700 Message-ID: <20260805110324.25067-9-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- include/linux/vm_planes.h | 9 +++++++++ init/vm_planes.c | 20 ++++++++++++++++++++ 2 files changed, 29 insertions(+) diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index 506de566cbd5..6d0066f70349 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -9,14 +9,23 @@ =20 #define VM_PLANE_KERNEL_NAME_MAX 128 =20 +enum vm_plane_kernel_format { + VM_PLANE_KFMT_RAW =3D 0, + VM_PLANE_KFMT_BZIMAGE, + VM_PLANE_KFMT_ELF, +}; + struct vm_plane_config { phys_addr_t load_offset; phys_addr_t memory_size; unsigned int vcpu_count; + unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; }; =20 void __init arch_init_vm_planes(void); +void __init load_vm_plane_kernels(unsigned int plane_count, + struct vm_plane_config *plane_cfg); =20 #endif /* CONFIG_VM_PLANES */ =20 diff --git a/init/vm_planes.c b/init/vm_planes.c index ecfe8de409a8..da9c17de4a44 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -8,8 +8,10 @@ #include #include #include +#include #include #include +#include =20 #ifdef CONFIG_VM_PLANES static bool __initdata enable_vm_planes_requested; @@ -21,6 +23,7 @@ struct vm_plane_parse_state { phys_addr_t load_offset; phys_addr_t memory_size; unsigned int vcpu_count; + unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; }; =20 @@ -209,6 +212,23 @@ static int __init parse_plane_cfg_line(const char *lin= e, size_t len, return 0; } =20 + if (!strcmp(key, "KERNEL_FORMAT")) { + unsigned int fmt; + + if (!strcasecmp(val, "raw")) + fmt =3D VM_PLANE_KFMT_RAW; + else if (!strcasecmp(val, "bzimage")) + fmt =3D VM_PLANE_KFMT_BZIMAGE; + else if (!strcasecmp(val, "elf")) + fmt =3D VM_PLANE_KFMT_ELF; + else + return -EINVAL; + + plane_cfg[plane_id].kernel_format =3D fmt; + state[plane_id].kernel_format =3D fmt; + return 0; + } + if (kstrtou64(val, 0, &parsed_u64)) return -EINVAL; =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 854FD4334D3; Wed, 5 Aug 2026 11:03:50 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927832; cv=none; b=hO9sV3B2Jw07MjxS67O19Z1itx/McdRSNuCILrqohNTN9OSxgZFOt54EXFW+uPleA7ZcRB/lOJHM+HBe6cywnsc1wJMWpa+yTy704pMRBU6UrK1gKA8lYnynbeH07BQK04pXUKgaQA3l7kpwtw2jAL86sVUncLpsH23e55kN1Tk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927832; c=relaxed/simple; bh=MHcbeTX2dK1GEG7Bqjd2QyfuOrt9zKlxkq7qViV9rR8=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=W/9Wm1tRAc1ERAwjTPLVqDbCmLye19PBePLrn+jVPG8+nB3HXKjFWNAuJEBfo+GVuqhyt4/jJ9yE0Fhq+w6PKjlmXSAk6/J4hxOqRVk/jgArbnRNXsgr447hYIeFiTDZ6Go8vp4DxmPvk8EVgfCQ81oto1BulnXmNDvNTj1tjhU= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=eGdVrYuy; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="eGdVrYuy" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 81B2120B716A; Wed, 5 Aug 2026 04:03:29 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 81B2120B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927809; bh=K4mWSHsUdER4mUqbyNOHd9q3VHmmNBWDaskxeQLhVsA=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=eGdVrYuyo7R1JWp/rnoEi4g5c0CQaCeho36nQk/xQEa2IEprBQ7y7c6T0aT+s225e A1jiJTc9tqnW1elfLfb5yD+2S9HPh4eDiEZN+aaZ6n5YqcUzKOsVEcntmKQV3oODGK BDt9wzVzvuVJbGDtn5WGiT3J1OQiX08eQ1T7KQzA= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 09/42] Activate the VM Planes through the Hypervisor - Using KVM as the VMM Date: Wed, 5 Aug 2026 04:02:51 -0700 Message-ID: <20260805110324.25067-10-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- arch/x86/include/asm/cpu.h | 3 +- arch/x86/kernel/cpu/common.c | 39 ++++++-- include/linux/vm_planes.h | 4 +- init/vm_planes.c | 179 ++++++++++++++++++++++++++++++++++- 4 files changed, 212 insertions(+), 13 deletions(-) diff --git a/arch/x86/include/asm/cpu.h b/arch/x86/include/asm/cpu.h index 8ab76adf14a9..52e80c6ac8f0 100644 --- a/arch/x86/include/asm/cpu.h +++ b/arch/x86/include/asm/cpu.h @@ -13,8 +13,9 @@ #ifdef CONFIG_VM_PLANES struct vm_plane_config; =20 -void __init alloc_vm_planes(unsigned int plane_count, +int __init alloc_vm_planes(unsigned int plane_count, struct vm_plane_config *plane_cfg); +int __init activate_vm_planes(unsigned int plane_count); #endif =20 #ifndef CONFIG_SMP diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c index 166597204739..9912208d2010 100644 --- a/arch/x86/kernel/cpu/common.c +++ b/arch/x86/kernel/cpu/common.c @@ -82,7 +82,9 @@ =20 #ifdef CONFIG_VM_PLANES /* Private hypercall number for early VM plane configuration. */ -#define KVM_HC_VM_PLANES_CONFIG 0x1000 +#define KVM_HC_VM_PLANES_CONFIG 0x1000 +/* Private hypercall number to activate all configured planes. */ +#define KVM_HC_VM_PLANES_ACTIVATE 0x1001 #endif =20 DEFINE_PER_CPU_READ_MOSTLY(struct cpuinfo_x86, cpu_info); @@ -2674,31 +2676,56 @@ void __init arch_cpu_finalize_init(void) } =20 #ifdef CONFIG_VM_PLANES -void __init alloc_vm_planes(unsigned int plane_count, +int __init alloc_vm_planes(unsigned int plane_count, struct vm_plane_config *plane_cfg) { phys_addr_t phys; long ret; =20 if (!plane_count || !plane_cfg) - return; + return -EINVAL; =20 if (!kvm_para_available()) { pr_warn("vm_planes: hypercall interface unavailable\n"); - return; + return -ENODEV; } =20 phys =3D virt_to_phys((void *)plane_cfg); =20 if (sizeof(unsigned long) < sizeof(phys_addr_t) && phys > ULONG_MAX) { pr_warn("vm_planes: shared config address exceeds hypercall register wid= th\n"); - return; + return -EOVERFLOW; } =20 ret =3D kvm_hypercall2(KVM_HC_VM_PLANES_CONFIG, (unsigned long)phys, plane_count); - if (ret < 0) + if (ret < 0) { pr_warn("vm_planes: hypercall failed: %ld\n", ret); + return (int)ret; + } + + return 0; +} + +int __init activate_vm_planes(unsigned int plane_count) +{ + long ret; + + if (!plane_count) + return -EINVAL; + + if (!kvm_para_available()) { + pr_warn("vm_planes: hypercall interface unavailable\n"); + return -ENODEV; + } + + ret =3D kvm_hypercall1(KVM_HC_VM_PLANES_ACTIVATE, plane_count); + if (ret < 0) { + pr_warn("vm_planes: activate hypercall failed: %ld\n", ret); + return (int)ret; + } + + return 0; } #endif diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index 6d0066f70349..5850c9e0d097 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -24,8 +24,8 @@ struct vm_plane_config { }; =20 void __init arch_init_vm_planes(void); -void __init load_vm_plane_kernels(unsigned int plane_count, - struct vm_plane_config *plane_cfg); +int __init load_vm_plane_kernels(unsigned int plane_count, + struct vm_plane_config *plane_cfg); =20 #endif /* CONFIG_VM_PLANES */ =20 diff --git a/init/vm_planes.c b/init/vm_planes.c index da9c17de4a44..6fac52af4b77 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -11,7 +11,7 @@ #include #include #include -#include +#include =20 #ifdef CONFIG_VM_PLANES static bool __initdata enable_vm_planes_requested; @@ -405,6 +405,159 @@ static int __init vm_planes_get_cfg_from_initrd(unsig= ned int *plane_count, return -ENOENT; } =20 +static int __init find_initrd_file(const char *filename, + const u8 **out_data, u32 *out_size) +{ + const u8 *p =3D (const u8 *)(unsigned long)initrd_start; + const u8 *end =3D (const u8 *)(unsigned long)initrd_end; + + if (!initrd_start || !initrd_end || initrd_end <=3D initrd_start) + return -ENOENT; + + while (p + sizeof(struct cpio_newc_header) <=3D end) { + const struct cpio_newc_header *hdr; + const char *name; + const u8 *data; + u32 namesize, filesize; + u32 name_align, data_align; + int ret; + + hdr =3D (const struct cpio_newc_header *)p; + if (memcmp(hdr->c_magic, "070701", 6) && + memcmp(hdr->c_magic, "070702", 6)) + return -EINVAL; + + ret =3D parse_hex_field(hdr->c_namesize, + sizeof(hdr->c_namesize), &namesize); + if (ret) + return ret; + + ret =3D parse_hex_field(hdr->c_filesize, + sizeof(hdr->c_filesize), &filesize); + if (ret) + return ret; + + if (!namesize) + return -EINVAL; + + p +=3D sizeof(*hdr); + if (p + namesize > end) + return -EINVAL; + + name =3D (const char *)p; + name_align =3D ALIGN(namesize, 4); + if (p + name_align > end) + return -EINVAL; + + data =3D p + name_align; + if (data + filesize > end) + return -EINVAL; + + if (!strcmp(name, "TRAILER!!!")) + break; + + if (cpio_name_match(name, namesize, filename)) { + *out_data =3D data; + *out_size =3D filesize; + return 0; + } + + data_align =3D ALIGN(filesize, 4); + if (data + data_align < data || data + data_align > end) + return -EINVAL; + + p =3D data + data_align; + } + + return -ENOENT; +} + +static int __init copy_to_early_mem(phys_addr_t dest, const void *src, + unsigned long size) +{ + unsigned long slop, clen; + char *p; + + while (size) { + slop =3D offset_in_page(dest); + clen =3D size; + if (clen > PAGE_SIZE - slop) + clen =3D PAGE_SIZE - slop; + p =3D early_memremap(dest & PAGE_MASK, clen + slop); + if (!p) + return -ENOMEM; + memcpy(p + slop, src, clen); + early_memunmap(p, clen + slop); + dest +=3D clen; + src +=3D clen; + size -=3D clen; + } + return 0; +} + +static int __init load_plane_kernel_raw(const u8 *data, u32 size, + struct vm_plane_config *cfg) +{ + if (size > cfg->memory_size) { + pr_err("vm_planes: raw kernel image (%u bytes) exceeds plane memory (%ll= u bytes)\n", + size, (unsigned long long)cfg->memory_size); + return -ENOMEM; + } + + return copy_to_early_mem(cfg->load_offset, data, size); +} + +int __init load_vm_plane_kernels(unsigned int plane_count, + struct vm_plane_config *plane_cfg) +{ + unsigned int i; + int err =3D 0; + + for (i =3D 1; i < plane_count; i++) { + const u8 *data; + u32 size; + int ret; + + ret =3D find_initrd_file(plane_cfg[i].kernel, &data, &size); + if (ret) { + pr_err("vm_planes: plane %u: kernel image '%s' not found in initrd\n", + i, plane_cfg[i].kernel); + err =3D ret; + continue; + } + + switch (plane_cfg[i].kernel_format) { + case VM_PLANE_KFMT_RAW: + ret =3D load_plane_kernel_raw(data, size, + &plane_cfg[i]); + break; + case VM_PLANE_KFMT_BZIMAGE: + case VM_PLANE_KFMT_ELF: + pr_err("vm_planes: plane %u: kernel format not yet supported\n", + i); + err =3D -ENOSYS; + continue; + default: + pr_err("vm_planes: plane %u: unknown kernel format %u\n", + i, plane_cfg[i].kernel_format); + err =3D -EINVAL; + continue; + } + + if (ret) { + pr_err("vm_planes: plane %u: failed to load kernel image: %d\n", + i, ret); + err =3D ret; + } else { + pr_info("vm_planes: plane %u: loaded '%s' (%u bytes) at 0x%llx\n", + i, plane_cfg[i].kernel, + size, (unsigned long long)plane_cfg[i].load_offset); + } + } + + return err; +} + static int __init parse_enable_vm_planes(char *str) { bool enable; @@ -423,13 +576,16 @@ static int __init parse_enable_vm_planes(char *str) =20 early_param("enable-vm-planes", parse_enable_vm_planes); =20 -void __init __weak alloc_vm_planes(unsigned int plane_count, - struct vm_plane_config *plane_cfg) { } +int __init __weak alloc_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg) { return -ENOSYS; } + +int __init __weak activate_vm_planes(unsigned int plane_count) { return -E= NOSYS; } =20 void __init arch_init_vm_planes(void) { unsigned int plane_count =3D VM_PLANES_DEFAULT_COUNT; struct vm_plane_config *plane_cfg; + int ret; =20 if (!enable_vm_planes_requested) return; @@ -445,7 +601,22 @@ void __init arch_init_vm_planes(void) =20 pr_info("vm_planes: enabling %u planes (ids 0..%u)\n", plane_count, plane_count - 1); - alloc_vm_planes(plane_count, plane_cfg); + + ret =3D alloc_vm_planes(plane_count, plane_cfg); + if (ret) { + pr_err("vm_planes: failed to allocate planes: %d\n", ret); + return; + } + + ret =3D load_vm_plane_kernels(plane_count, plane_cfg); + if (ret) { + pr_err("vm_planes: failed to load plane kernels: %d\n", ret); + return; + } + + ret =3D activate_vm_planes(plane_count); + if (ret) + pr_err("vm_planes: failed to activate planes: %d\n", ret); } =20 #endif /* CONFIG_VM_PLANES */ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 009FC433E75; Wed, 5 Aug 2026 11:03:51 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927832; cv=none; b=juORHBQhlzMoUXsepsCsRt2LDFZLn99p6VoTM7uikv8GbIHR9iSCewooqQYc9bXfcY8YUiKJVNDo4vTR8J+fQk+nj+Rx0OyEtWfMPV7BRlkXhUhi4/yb3kc4Z64vQqaw2MFRLC3d/OHCCc+4xIhrPBFBHPrzRLac37kJHmzQgzk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927832; c=relaxed/simple; bh=cYgenNcI6DdbSIjRFFvZR8MMQhanxPuMNqdnDjZWzWI=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=p27l7kvLntreeEuIhM4mkH7LJsU72GyYaViGzKbtrv+bT8JJuBIrqSAay/z7jzWN5/07evQc5gXJyMdGciaDfcg9KtpRPjrRRlXXrnP6tVqquZt2tqtpyiGvgw6f+X2ZQd3PCzfGVjQjyqEi8LK4Y4R0n/xdkzVDc65QXqMbhiU= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=RXLhK71g; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="RXLhK71g" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 3691220B716B; Wed, 5 Aug 2026 04:03:30 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 3691220B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927810; bh=3beNJnH3+Za/7vunv7aa4Wf3uGZkjunoTVeLHc8NxAc=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=RXLhK71gbs3C9ukpuGxCrmPNCrdT9Drkv6I2VnKW28tZdy59zef3B8EZCQV54lf3K CJfK7XZniWjRmtSJi1kydly4+Sh0Bfnb7OrB3K9owjhaNZnHm7gzpqMTL4tjPuCV9u 1ZmX+cJ8ljsglCPJVGb9c3enrrscvWhBKp5ZTDGM= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 10/42] allow the command line to be specified for kernels in other planes Date: Wed, 5 Aug 2026 04:02:52 -0700 Message-ID: <20260805110324.25067-11-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- include/linux/vm_planes.h | 2 ++ init/vm_planes.c | 23 ++++++++++++++++++++++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index 5850c9e0d097..bb06dcbcf0cb 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -8,6 +8,7 @@ #ifdef CONFIG_VM_PLANES =20 #define VM_PLANE_KERNEL_NAME_MAX 128 +#define VM_PLANE_CMDLINE_MAX 512 =20 enum vm_plane_kernel_format { VM_PLANE_KFMT_RAW =3D 0, @@ -21,6 +22,7 @@ struct vm_plane_config { unsigned int vcpu_count; unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; + char cmdline[VM_PLANE_CMDLINE_MAX]; }; =20 void __init arch_init_vm_planes(void); diff --git a/init/vm_planes.c b/init/vm_planes.c index 6fac52af4b77..613daa161298 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -25,6 +25,7 @@ struct vm_plane_parse_state { unsigned int vcpu_count; unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; + char cmdline[VM_PLANE_CMDLINE_MAX]; }; =20 #define VM_PLANES_UNSET_VALUE ((phys_addr_t)~0) @@ -141,7 +142,7 @@ static int __init parse_plane_cfg_line(const char *line= , size_t len, struct vm_plane_config *plane_cfg, struct vm_plane_parse_state *state) { - char tmp[192]; + char tmp[VM_PLANE_CMDLINE_MAX + 64]; char *p, *key, *val; unsigned int plane_id; u64 parsed_u64; @@ -229,6 +230,25 @@ static int __init parse_plane_cfg_line(const char *lin= e, size_t len, return 0; } =20 + if (!strcmp(key, "CMDLINE")) { + size_t val_len =3D strlen(val); + + if (val_len >=3D 2 && val[0] =3D=3D '"') { + if (val[val_len - 1] !=3D '"') + return -EINVAL; + val[val_len - 1] =3D '\0'; + val++; + } + + if (strscpy(plane_cfg[plane_id].cmdline, val, + sizeof(plane_cfg[plane_id].cmdline)) < 0) + return -E2BIG; + + strscpy(state[plane_id].cmdline, val, + sizeof(state[plane_id].cmdline)); + return 0; + } + if (kstrtou64(val, 0, &parsed_u64)) return -EINVAL; =20 @@ -292,6 +312,7 @@ static int __init parse_vm_planes_kconfig(const char *b= uf, size_t len, state[i].memory_size =3D VM_PLANES_UNSET_VALUE; state[i].vcpu_count =3D 0; state[i].kernel[0] =3D '\0'; + state[i].cmdline[0] =3D '\0'; } =20 while (p < end) { --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 6F31B43802B; Wed, 5 Aug 2026 11:03:52 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927834; cv=none; b=X099STt5dKxJdheXuOXhSAVMUWMr/8S8HIy1CVhVH5cCiuTCDHbKAQWwsnexVksrdYmdp0qMYX1BiHucI9SxzYJkztlpBl09Tl2Trh/wOZxOBQqakTlgP+k+5elJrZ8yKDsBZmVP51cnGR9NDD7f1DFLFEt6gdSL3ZgvnyjOZOw= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927834; c=relaxed/simple; bh=D1PdRGy+W/GQGTigJBXozi74E4Z8ucqMCmtco5SiarI=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=fGFdEGO636Kp50hHh4gNXl9Nv0LYFbRLB2spCaGWNoXnoPF3jvEwi2B0TITkv1C7R5s1iY7vlx+uy/vy4HDw9Sum7hILyknSPVnMSyYkLTsWWT9hpVlVqugKEYsgkE86k+jaD501ZMbG8UZaLDtMHI+fAAf8jTvCU7+vW8Mckdo= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=lD4SuauX; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="lD4SuauX" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id C91AD20B7169; Wed, 5 Aug 2026 04:03:30 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com C91AD20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927811; bh=pr71ZzUrlyCcrAnUnK+RK3cbvl/A+yinC6vkxx+FmC0=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=lD4SuauXyYGIKHkk1tgK6kDc63dQQszgfO/YfAri2idrkveF7EeypcBMuwk0pnGjh SJl1HwpiCTwmqzOiBy8aA2ITxeDv5Ig5KGSC4t7SdCo83y4SJU17HNHXkyXlK54+lq EBCYFpf/LbKfAf50Q9tgxXsuORb/8OWFpf+XxQsw= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 11/42] Various changes to support VM Planes. Date: Wed, 5 Aug 2026 04:02:53 -0700 Message-ID: <20260805110324.25067-12-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Remove custom parsing of initrd. 1. **vm_planes.c** =E2=80=94 Core VM planes implementation (major rewrite) - Fixed config parser: `PLANE_COUNT=3D` line no longer causes fatal `-EI= NVAL` (returns `-ENOENT` to skip) - ELF loader: biases `p_paddr` by `load_offset` so kernel loads at corre= ct GPA - Entry point: computes physical entry from ELF vaddr=E2=86=92paddr mapp= ing, with fallback for physical `e_entry` - `activate_vm_planes()` now passes `plane_cfg` GPA so QEMU can read the= updated `entry_point` 2. **vm_planes.h** =E2=80=94 Added `entry_point` field declaration 3. **main.c** =E2=80=94 Minor adjustment to `arch_init_vm_planes()` call si= te 4. **common.c** =E2=80=94 KVM hypercall implementations - Removed hardcoded `0x1000`/`0x1001` hypercall numbers - `alloc_vm_planes()`: unchanged (uses correct HC numbers from UAPI head= er) - `activate_vm_planes()`: now passes `plane_cfg` GPA + `plane_count` (wa= s just `plane_count`) 5. **cpu.h** =E2=80=94 Updated `activate_vm_planes()` signature to include = `plane_cfg` 6. **x86.c** =E2=80=94 KVM host-side hypercall support - `KVM_EXIT_HYPERCALL_VALID_MASK`: added bits 13 and 14 for VM planes hy= percalls - Added `KVM_HC_VM_PLANES_CONFIG` and `KVM_HC_VM_PLANES_ACTIVATE` case h= andlers that exit to userspace (QEMU) 7. **kvm_para.h** =E2=80=94 Added hypercall numbers - `KVM_HC_VM_PLANES_CONFIG =3D 13` - `KVM_HC_VM_PLANES_ACTIVATE =3D 14` --- arch/x86/include/asm/cpu.h | 3 +- arch/x86/kernel/cpu/common.c | 19 +- arch/x86/kvm/x86.c | 25 +- include/linux/vm_planes.h | 1 + include/uapi/linux/kvm_para.h | 2 + init/main.c | 7 +- init/vm_planes.c | 422 ++++++++++++++++++---------------- 7 files changed, 266 insertions(+), 213 deletions(-) diff --git a/arch/x86/include/asm/cpu.h b/arch/x86/include/asm/cpu.h index 52e80c6ac8f0..f9cb541e6367 100644 --- a/arch/x86/include/asm/cpu.h +++ b/arch/x86/include/asm/cpu.h @@ -15,7 +15,8 @@ struct vm_plane_config; =20 int __init alloc_vm_planes(unsigned int plane_count, struct vm_plane_config *plane_cfg); -int __init activate_vm_planes(unsigned int plane_count); +int __init activate_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg); #endif =20 #ifndef CONFIG_SMP diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c index 9912208d2010..7edc2c4072cc 100644 --- a/arch/x86/kernel/cpu/common.c +++ b/arch/x86/kernel/cpu/common.c @@ -80,13 +80,6 @@ =20 #include "cpu.h" =20 -#ifdef CONFIG_VM_PLANES -/* Private hypercall number for early VM plane configuration. */ -#define KVM_HC_VM_PLANES_CONFIG 0x1000 -/* Private hypercall number to activate all configured planes. */ -#define KVM_HC_VM_PLANES_ACTIVATE 0x1001 -#endif - DEFINE_PER_CPU_READ_MOSTLY(struct cpuinfo_x86, cpu_info); EXPORT_PER_CPU_SYMBOL(cpu_info); =20 @@ -2708,11 +2701,13 @@ int __init alloc_vm_planes(unsigned int plane_count, return 0; } =20 -int __init activate_vm_planes(unsigned int plane_count) +int __init activate_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg) { + phys_addr_t phys; long ret; =20 - if (!plane_count) + if (!plane_count || !plane_cfg) return -EINVAL; =20 if (!kvm_para_available()) { @@ -2720,7 +2715,11 @@ int __init activate_vm_planes(unsigned int plane_cou= nt) return -ENODEV; } =20 - ret =3D kvm_hypercall1(KVM_HC_VM_PLANES_ACTIVATE, plane_count); + phys =3D virt_to_phys((void *)plane_cfg); + + ret =3D kvm_hypercall2(KVM_HC_VM_PLANES_ACTIVATE, + (unsigned long)phys, + plane_count); if (ret < 0) { pr_warn("vm_planes: activate hypercall failed: %ld\n", ret); return (int)ret; diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index a0a8818b3096..b7256f155bea 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -119,7 +119,9 @@ u64 __read_mostly efer_reserved_bits =3D ~((u64)(EFER_S= CE | EFER_LME | EFER_LMA)); static u64 __read_mostly efer_reserved_bits =3D ~((u64)EFER_SCE); #endif =20 -#define KVM_EXIT_HYPERCALL_VALID_MASK (1 << KVM_HC_MAP_GPA_RANGE) +#define KVM_EXIT_HYPERCALL_VALID_MASK (BIT(KVM_HC_MAP_GPA_RANGE) | \ + BIT(KVM_HC_VM_PLANES_CONFIG) | \ + BIT(KVM_HC_VM_PLANES_ACTIVATE)) =20 #define KVM_CAP_PMU_VALID_MASK KVM_PMU_CAP_DISABLE =20 @@ -10530,6 +10532,27 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vcp= u, int cpl, vcpu->arch.complete_userspace_io =3D complete_hypercall; return 0; } + case KVM_HC_VM_PLANES_CONFIG: + case KVM_HC_VM_PLANES_ACTIVATE: { + ret =3D -KVM_ENOSYS; + if (!user_exit_on_hypercall(vcpu->kvm, nr)) + break; + + vcpu->run->exit_reason =3D KVM_EXIT_HYPERCALL; + vcpu->run->hypercall.nr =3D nr; + vcpu->run->hypercall.ret =3D 0; + vcpu->run->hypercall.args[0] =3D a0; + vcpu->run->hypercall.args[1] =3D a1; + vcpu->run->hypercall.args[2] =3D a2; + vcpu->run->hypercall.args[3] =3D a3; + vcpu->run->hypercall.flags =3D 0; + if (op_64_bit) + vcpu->run->hypercall.flags |=3D KVM_EXIT_HYPERCALL_LONG_MODE; + + WARN_ON_ONCE(vcpu->run->hypercall.flags & KVM_EXIT_HYPERCALL_MBZ); + vcpu->arch.complete_userspace_io =3D complete_hypercall; + return 0; + } default: ret =3D -KVM_ENOSYS; break; diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index bb06dcbcf0cb..47f05fa80039 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -19,6 +19,7 @@ enum vm_plane_kernel_format { struct vm_plane_config { phys_addr_t load_offset; phys_addr_t memory_size; + phys_addr_t entry_point; unsigned int vcpu_count; unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; diff --git a/include/uapi/linux/kvm_para.h b/include/uapi/linux/kvm_para.h index 960c7e93d1a9..1b097f7ed937 100644 --- a/include/uapi/linux/kvm_para.h +++ b/include/uapi/linux/kvm_para.h @@ -30,6 +30,8 @@ #define KVM_HC_SEND_IPI 10 #define KVM_HC_SCHED_YIELD 11 #define KVM_HC_MAP_GPA_RANGE 12 +#define KVM_HC_VM_PLANES_CONFIG 13 +#define KVM_HC_VM_PLANES_ACTIVATE 14 =20 /* * hypercalls use architecture specific diff --git a/init/main.c b/init/main.c index 3e35c2caca17..1c779f6d60cc 100644 --- a/init/main.c +++ b/init/main.c @@ -994,9 +994,6 @@ void start_kernel(void) pr_notice("%s", linux_banner); setup_arch(&command_line); mm_core_init_early(); -#ifdef CONFIG_VM_PLANES - arch_init_vm_planes(); -#endif /* Static keys and static calls are needed by LSMs */ jump_label_init(); static_call_init(); @@ -1666,6 +1663,10 @@ static noinline void __init kernel_init_freeable(voi= d) wait_for_initramfs(); console_on_rootfs(); =20 +#ifdef CONFIG_VM_PLANES + arch_init_vm_planes(); +#endif + /* * check if there is an early userspace init. If yes, let it do all * the work diff --git a/init/vm_planes.c b/init/vm_planes.c index 613daa161298..274c0015fe76 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -3,12 +3,15 @@ #include #include #include -#include +#include #include #include +#include +#include #include #include #include +#include #include #include #include @@ -30,46 +33,49 @@ struct vm_plane_parse_state { =20 #define VM_PLANES_UNSET_VALUE ((phys_addr_t)~0) =20 -struct cpio_newc_header { - char c_magic[6]; - char c_ino[8]; - char c_mode[8]; - char c_uid[8]; - char c_gid[8]; - char c_nlink[8]; - char c_mtime[8]; - char c_filesize[8]; - char c_devmajor[8]; - char c_devminor[8]; - char c_rdevmajor[8]; - char c_rdevminor[8]; - char c_namesize[8]; - char c_check[8]; -}; - -static int __init parse_hex_field(const char *field, size_t len, u32 *valu= e) +/* + * Read a file from the rootfs into a newly allocated buffer. + * Caller must kfree(*out_data) when done. + */ +static int __init vm_planes_read_file(const char *path, + void **out_data, loff_t *out_size) { - u32 v =3D 0; - size_t i; - - for (i =3D 0; i < len; i++) { - u8 c =3D field[i]; - - v <<=3D 4; - if (c >=3D '0' && c <=3D '9') - v |=3D c - '0'; - else if (c >=3D 'a' && c <=3D 'f') - v |=3D c - 'a' + 10; - else if (c >=3D 'A' && c <=3D 'F') - v |=3D c - 'A' + 10; - else - return -EINVAL; + struct file *fp; + loff_t fsize; + void *buf; + ssize_t rd; + + fp =3D filp_open(path, O_RDONLY, 0); + if (IS_ERR(fp)) + return PTR_ERR(fp); + + fsize =3D i_size_read(file_inode(fp)); + if (fsize <=3D 0) { + fput(fp); + return -ENODATA; + } + + buf =3D kvmalloc(fsize, GFP_KERNEL); + if (!buf) { + fput(fp); + return -ENOMEM; + } + + rd =3D kernel_read(fp, buf, fsize, &(loff_t){0}); + fput(fp); + + if (rd !=3D fsize) { + kvfree(buf); + return (rd < 0) ? (int)rd : -EIO; } =20 - *value =3D v; + *out_data =3D buf; + *out_size =3D fsize; return 0; } =20 +/* ---- Config file parser (unchanged) ---- */ + static int __init parse_plane_count_line(const char *line, size_t len, unsigned int *plane_count) { @@ -174,7 +180,7 @@ static int __init parse_plane_cfg_line(const char *line= , size_t len, =20 key =3D strchr(p, '_'); if (!key) - return -EINVAL; + return -ENOENT; *key++ =3D '\0'; =20 if (kstrtouint(p, 10, &plane_id) || plane_id >=3D plane_count) @@ -297,16 +303,14 @@ static int __init parse_vm_planes_kconfig(const char = *buf, size_t len, if (*plane_count > UINT_MAX / sizeof(**plane_cfg)) return -E2BIG; =20 - *plane_cfg =3D memblock_alloc(*plane_count * sizeof(**plane_cfg), - SMP_CACHE_BYTES); + *plane_cfg =3D kzalloc(*plane_count * sizeof(**plane_cfg), GFP_KERNEL); if (!*plane_cfg) return -ENOMEM; =20 - state =3D memblock_alloc(*plane_count * sizeof(*state), SMP_CACHE_BYTES); + state =3D kzalloc(*plane_count * sizeof(*state), GFP_KERNEL); if (!state) return -ENOMEM; =20 - memset(*plane_cfg, 0, *plane_count * sizeof(**plane_cfg)); for (i =3D 0; i < *plane_count; i++) { state[i].load_offset =3D VM_PLANES_UNSET_VALUE; state[i].memory_size =3D VM_PLANES_UNSET_VALUE; @@ -329,9 +333,6 @@ static int __init parse_vm_planes_kconfig(const char *b= uf, size_t len, p++; } =20 - /* Plane 0 is the already-running boot plane; only secondary planes - * must provide full allocation metadata. - */ for (i =3D 1; i < *plane_count; i++) { if (state[i].load_offset =3D=3D VM_PLANES_UNSET_VALUE || state[i].memory_size =3D=3D VM_PLANES_UNSET_VALUE || @@ -340,179 +341,192 @@ static int __init parse_vm_planes_kconfig(const cha= r *buf, size_t len, return -EINVAL; } =20 + kfree(state); return 0; } =20 -static bool __init cpio_name_match(const char *name, size_t namesize, - const char *target) -{ - while (namesize > 1 && (*name =3D=3D '/' || - (namesize > 2 && name[0] =3D=3D '.' && name[1] =3D=3D '/'))) { - if (*name =3D=3D '/') { - name++; - namesize--; - } else { - name +=3D 2; - namesize -=3D 2; - } - } +/* ---- Config loading via VFS ---- */ =20 - return !strncmp(name, target, namesize - 1) && - strlen(target) =3D=3D namesize - 1; -} - -static int __init vm_planes_get_cfg_from_initrd(unsigned int *plane_count, - struct vm_plane_config **plane_cfg) +static int __init vm_planes_get_cfg(unsigned int *plane_count, + struct vm_plane_config **plane_cfg) { - const u8 *p =3D (const u8 *)(unsigned long)initrd_start; - const u8 *end =3D (const u8 *)(unsigned long)initrd_end; - - if (!initrd_start || !initrd_end || initrd_end <=3D initrd_start) - return -ENOENT; - - while (p + sizeof(struct cpio_newc_header) <=3D end) { - const struct cpio_newc_header *hdr; - const char *name; - const u8 *data; - u32 namesize, filesize; - u32 name_align, data_align; - int ret; - - hdr =3D (const struct cpio_newc_header *)p; - if (memcmp(hdr->c_magic, "070701", 6) && - memcmp(hdr->c_magic, "070702", 6)) - return -EINVAL; - - ret =3D parse_hex_field(hdr->c_namesize, sizeof(hdr->c_namesize), &names= ize); - if (ret) - return ret; - - ret =3D parse_hex_field(hdr->c_filesize, sizeof(hdr->c_filesize), &files= ize); - if (ret) - return ret; - - if (!namesize) - return -EINVAL; + void *buf; + loff_t size; + int ret; =20 - p +=3D sizeof(*hdr); - if (p + namesize > end) - return -EINVAL; + ret =3D vm_planes_read_file("/" VM_PLANES_CONFIG_FILE, &buf, &size); + if (ret) { + pr_err("vm_planes: cannot read /%s: %d\n", + VM_PLANES_CONFIG_FILE, ret); + return ret; + } =20 - name =3D (const char *)p; - name_align =3D ALIGN(namesize, 4); - if (p + name_align > end) - return -EINVAL; + ret =3D parse_vm_planes_kconfig(buf, (size_t)size, plane_count, plane_cfg= ); + kvfree(buf); + return ret; +} =20 - data =3D p + name_align; - if (data + filesize > end) - return -EINVAL; +/* ---- Kernel loading ---- */ =20 - if (!strcmp(name, "TRAILER!!!")) - break; +static int __init copy_to_early_mem(phys_addr_t dest, const void *src, + unsigned long size) +{ + unsigned long slop, clen; + char *p; =20 - if (cpio_name_match(name, namesize, VM_PLANES_CONFIG_FILE)) - return parse_vm_planes_kconfig((const char *)data, - filesize, - plane_count, - plane_cfg); + while (size) { + slop =3D offset_in_page(dest); + clen =3D size; + if (clen > PAGE_SIZE - slop) + clen =3D PAGE_SIZE - slop; + p =3D early_memremap(dest & PAGE_MASK, clen + slop); + if (!p) + return -ENOMEM; + memcpy(p + slop, src, clen); + early_memunmap(p, clen + slop); + dest +=3D clen; + src +=3D clen; + size -=3D clen; + } + return 0; +} =20 - data_align =3D ALIGN(filesize, 4); - if (data + data_align < data || data + data_align > end) - return -EINVAL; +static int __init zero_early_mem(phys_addr_t dest, unsigned long size) +{ + unsigned long slop, clen; + char *p; =20 - p =3D data + data_align; + while (size) { + slop =3D offset_in_page(dest); + clen =3D size; + if (clen > PAGE_SIZE - slop) + clen =3D PAGE_SIZE - slop; + p =3D early_memremap(dest & PAGE_MASK, clen + slop); + if (!p) + return -ENOMEM; + memset(p + slop, 0, clen); + early_memunmap(p, clen + slop); + dest +=3D clen; + size -=3D clen; } - - return -ENOENT; + return 0; } =20 -static int __init find_initrd_file(const char *filename, - const u8 **out_data, u32 *out_size) +static int __init load_plane_kernel_elf(const u8 *data, u32 size, + struct vm_plane_config *cfg) { - const u8 *p =3D (const u8 *)(unsigned long)initrd_start; - const u8 *end =3D (const u8 *)(unsigned long)initrd_end; + const Elf64_Ehdr *ehdr; + const Elf64_Phdr *phdr; + unsigned int i; + int ret; =20 - if (!initrd_start || !initrd_end || initrd_end <=3D initrd_start) - return -ENOENT; + if (size < sizeof(*ehdr)) { + pr_err("vm_planes: ELF image too small (%u bytes)\n", size); + return -EINVAL; + } =20 - while (p + sizeof(struct cpio_newc_header) <=3D end) { - const struct cpio_newc_header *hdr; - const char *name; - const u8 *data; - u32 namesize, filesize; - u32 name_align, data_align; - int ret; + ehdr =3D (const Elf64_Ehdr *)data; =20 - hdr =3D (const struct cpio_newc_header *)p; - if (memcmp(hdr->c_magic, "070701", 6) && - memcmp(hdr->c_magic, "070702", 6)) - return -EINVAL; + if (memcmp(ehdr->e_ident, ELFMAG, SELFMAG)) { + pr_err("vm_planes: not a valid ELF image\n"); + return -EINVAL; + } =20 - ret =3D parse_hex_field(hdr->c_namesize, - sizeof(hdr->c_namesize), &namesize); - if (ret) - return ret; + if (ehdr->e_ident[EI_CLASS] !=3D ELFCLASS64 || + ehdr->e_ident[EI_DATA] !=3D ELFDATA2LSB || + ehdr->e_type !=3D ET_EXEC || + ehdr->e_machine !=3D EM_X86_64) { + pr_err("vm_planes: unsupported ELF format (need x86_64 ET_EXEC LE)\n"); + return -EINVAL; + } =20 - ret =3D parse_hex_field(hdr->c_filesize, - sizeof(hdr->c_filesize), &filesize); - if (ret) - return ret; + if (!ehdr->e_phnum || ehdr->e_phentsize !=3D sizeof(Elf64_Phdr)) { + pr_err("vm_planes: invalid ELF program headers\n"); + return -EINVAL; + } =20 - if (!namesize) - return -EINVAL; + if (ehdr->e_phoff + (u64)ehdr->e_phnum * sizeof(Elf64_Phdr) > size) { + pr_err("vm_planes: ELF program headers extend beyond file\n"); + return -EINVAL; + } =20 - p +=3D sizeof(*hdr); - if (p + namesize > end) - return -EINVAL; + phdr =3D (const Elf64_Phdr *)(data + ehdr->e_phoff); =20 - name =3D (const char *)p; - name_align =3D ALIGN(namesize, 4); - if (p + name_align > end) - return -EINVAL; + for (i =3D 0; i < ehdr->e_phnum; i++, phdr++) { + phys_addr_t dest; + u64 bss_size; =20 - data =3D p + name_align; - if (data + filesize > end) - return -EINVAL; + if (phdr->p_type !=3D PT_LOAD) + continue; =20 - if (!strcmp(name, "TRAILER!!!")) - break; + if (!phdr->p_memsz) + continue; =20 - if (cpio_name_match(name, namesize, filename)) { - *out_data =3D data; - *out_size =3D filesize; - return 0; + /* + * Bias the ELF physical address by load_offset so that the + * kernel's link-time p_paddr values are treated as offsets + * within the plane's memory region. + */ + dest =3D cfg->load_offset + phdr->p_paddr; + + if (dest < cfg->load_offset || + dest + phdr->p_memsz > cfg->load_offset + cfg->memory_size) { + pr_err("vm_planes: ELF PT_LOAD at 0x%llx+0x%llx outside plane [0x%llx..= 0x%llx]\n", + (unsigned long long)dest, + (unsigned long long)phdr->p_memsz, + (unsigned long long)cfg->load_offset, + (unsigned long long)(cfg->load_offset + cfg->memory_size)); + return -EINVAL; } =20 - data_align =3D ALIGN(filesize, 4); - if (data + data_align < data || data + data_align > end) + if (phdr->p_offset + phdr->p_filesz > size) { + pr_err("vm_planes: ELF PT_LOAD file data beyond image\n"); return -EINVAL; + } =20 - p =3D data + data_align; - } + if (phdr->p_filesz) { + ret =3D copy_to_early_mem(dest, data + phdr->p_offset, + phdr->p_filesz); + if (ret) + return ret; + } =20 - return -ENOENT; -} + bss_size =3D phdr->p_memsz - phdr->p_filesz; + if (bss_size) { + ret =3D zero_early_mem(dest + phdr->p_filesz, bss_size); + if (ret) + return ret; + } =20 -static int __init copy_to_early_mem(phys_addr_t dest, const void *src, - unsigned long size) -{ - unsigned long slop, clen; - char *p; + /* + * Compute the physical entry point: if e_entry falls within + * this segment's virtual range, convert vaddr=E2=86=92paddr and bias. + * Also handle kernels where e_entry is already a physical + * address by checking the p_paddr range as a fallback. + */ + if (ehdr->e_entry >=3D phdr->p_vaddr && + ehdr->e_entry < phdr->p_vaddr + phdr->p_memsz) + cfg->entry_point =3D cfg->load_offset + + phdr->p_paddr + (ehdr->e_entry - phdr->p_vaddr); + else if (ehdr->e_entry >=3D phdr->p_paddr && + ehdr->e_entry < phdr->p_paddr + phdr->p_memsz) + cfg->entry_point =3D cfg->load_offset + ehdr->e_entry; + + pr_info("vm_planes: ELF PT_LOAD: paddr=3D0x%llx filesz=3D0x%llx memsz=3D= 0x%llx\n", + (unsigned long long)dest, + (unsigned long long)phdr->p_filesz, + (unsigned long long)phdr->p_memsz); + } =20 - while (size) { - slop =3D offset_in_page(dest); - clen =3D size; - if (clen > PAGE_SIZE - slop) - clen =3D PAGE_SIZE - slop; - p =3D early_memremap(dest & PAGE_MASK, clen + slop); - if (!p) - return -ENOMEM; - memcpy(p + slop, src, clen); - early_memunmap(p, clen + slop); - dest +=3D clen; - src +=3D clen; - size -=3D clen; + if (!cfg->entry_point) { + pr_err("vm_planes: ELF entry point 0x%llx not in any PT_LOAD segment\n", + (unsigned long long)ehdr->e_entry); + return -EINVAL; } + pr_info("vm_planes: ELF entry point: 0x%llx (virt 0x%llx)\n", + (unsigned long long)cfg->entry_point, + (unsigned long long)ehdr->e_entry); + return 0; } =20 @@ -525,6 +539,7 @@ static int __init load_plane_kernel_raw(const u8 *data,= u32 size, return -ENOMEM; } =20 + cfg->entry_point =3D cfg->load_offset; return copy_to_early_mem(cfg->load_offset, data, size); } =20 @@ -535,50 +550,59 @@ int __init load_vm_plane_kernels(unsigned int plane_c= ount, int err =3D 0; =20 for (i =3D 1; i < plane_count; i++) { - const u8 *data; - u32 size; + void *data; + loff_t fsize; int ret; =20 - ret =3D find_initrd_file(plane_cfg[i].kernel, &data, &size); + ret =3D vm_planes_read_file(plane_cfg[i].kernel, &data, &fsize); if (ret) { - pr_err("vm_planes: plane %u: kernel image '%s' not found in initrd\n", - i, plane_cfg[i].kernel); + pr_err("vm_planes: plane %u: kernel '%s' not found: %d\n", + i, plane_cfg[i].kernel, ret); err =3D ret; continue; } =20 switch (plane_cfg[i].kernel_format) { case VM_PLANE_KFMT_RAW: - ret =3D load_plane_kernel_raw(data, size, + ret =3D load_plane_kernel_raw(data, (u32)fsize, &plane_cfg[i]); break; - case VM_PLANE_KFMT_BZIMAGE: case VM_PLANE_KFMT_ELF: - pr_err("vm_planes: plane %u: kernel format not yet supported\n", + ret =3D load_plane_kernel_elf(data, (u32)fsize, + &plane_cfg[i]); + break; + case VM_PLANE_KFMT_BZIMAGE: + pr_err("vm_planes: plane %u: bzImage format not yet supported\n", i); err =3D -ENOSYS; + kvfree(data); continue; default: pr_err("vm_planes: plane %u: unknown kernel format %u\n", i, plane_cfg[i].kernel_format); err =3D -EINVAL; + kvfree(data); continue; } =20 if (ret) { - pr_err("vm_planes: plane %u: failed to load kernel image: %d\n", + pr_err("vm_planes: plane %u: failed to load kernel: %d\n", i, ret); err =3D ret; } else { - pr_info("vm_planes: plane %u: loaded '%s' (%u bytes) at 0x%llx\n", - i, plane_cfg[i].kernel, - size, (unsigned long long)plane_cfg[i].load_offset); + pr_info("vm_planes: plane %u: loaded '%s' (%lld bytes) at 0x%llx\n", + i, plane_cfg[i].kernel, fsize, + (unsigned long long)plane_cfg[i].load_offset); } + + kvfree(data); } =20 return err; } =20 +/* ---- Early param & activation ---- */ + static int __init parse_enable_vm_planes(char *str) { bool enable; @@ -600,7 +624,8 @@ early_param("enable-vm-planes", parse_enable_vm_planes); int __init __weak alloc_vm_planes(unsigned int plane_count, struct vm_plane_config *plane_cfg) { return -ENOSYS; } =20 -int __init __weak activate_vm_planes(unsigned int plane_count) { return -E= NOSYS; } +int __init __weak activate_vm_planes(unsigned int plane_count, + struct vm_plane_config *plane_cfg) { return -ENOSYS; } =20 void __init arch_init_vm_planes(void) { @@ -614,9 +639,10 @@ void __init arch_init_vm_planes(void) if (!kvm_para_available()) return; =20 - if (vm_planes_get_cfg_from_initrd(&plane_count, &plane_cfg)) { - pr_warn("vm_planes: failed to parse %s from initrd\n", - VM_PLANES_CONFIG_FILE); + ret =3D vm_planes_get_cfg(&plane_count, &plane_cfg); + if (ret) { + pr_warn("vm_planes: failed to parse %s: %d\n", + VM_PLANES_CONFIG_FILE, ret); return; } =20 @@ -635,7 +661,7 @@ void __init arch_init_vm_planes(void) return; } =20 - ret =3D activate_vm_planes(plane_count); + ret =3D activate_vm_planes(plane_count, plane_cfg); if (ret) pr_err("vm_planes: failed to activate planes: %d\n", ret); } --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 06CA843A802; Wed, 5 Aug 2026 11:03:54 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927839; cv=none; b=JW9uH3wSb3mvbd+xDhkeqNswsOK4uIfV9Ktg2PoX+5g0siN8P+PuCpoXra99AUdQuQgi5eMMiPJRoHjFArv9r249NEdKooMU30KtwO6L4mcFRkaiqQSv0U547usB95nRIQn+c/sEaZOaZD8HzMTv+fLSTfIeLSJhIKJICxpGNTo= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927839; c=relaxed/simple; bh=kxC46UAfwwIvNUmTMiiPTma8u95ibW8jf5EiHwcZIIw=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=RUycYYkYW70dN1nXICypa1tnC5qGsInWl46JVWFjRz7HhE54M6gy6tpO2tByL/I93eXy/+nHLKgOJHlgC9BkFxjUe2LUvS3e2jBZ39mKjMHM3KcX21fhdvmUjdKhhyyef23/xi+NZGNHZXjh0s4BLCovM+0OGyJj6zS1wxi27JM= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=oMpbLyel; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="oMpbLyel" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 2EDDE20B716A; Wed, 5 Aug 2026 04:03:32 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 2EDDE20B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927812; bh=OTbmf1wNLoEvJpk2flRCAOzPlEbX1zhK6e+giN5yucI=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=oMpbLyelgQwlMWsfKWRIFwO9cS0+IBC0ts2JQdzdPtIz8keV5cfzEhG6qRFvk/QxG 3yhHJjhr3J8JneMrBwXVcGnYmhHiTsXyZgsaqITsN7xPCeuvbyHppBEQgB8PbarIvT bWpWS1WbOjm1OT6uMcD2Uzavl2qQ7BIRga1RURzE= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 12/42] Add a Virtualization Based Security (VBS) framework. - Add backends for AMD SEV-SNP, Intel TDX, Arm CCA and KVM Planes. - Support VTL on Hyper-V in addition to Planes on KVM. Date: Wed, 5 Aug 2026 04:02:54 -0700 Message-ID: <20260805110324.25067-13-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable --- include/linux/vbs.h | 204 ++++++++++++++++++++++++++ security/Kconfig | 2 + security/Makefile | 3 + security/vbs/Kconfig | 69 +++++++++ security/vbs/Makefile | 9 ++ security/vbs/arm_cca.c | 300 ++++++++++++++++++++++++++++++++++++++ security/vbs/core.c | 166 +++++++++++++++++++++ security/vbs/hv_vsm.c | 257 ++++++++++++++++++++++++++++++++ security/vbs/internal.h | 41 ++++++ security/vbs/kvm_planes.c | 258 ++++++++++++++++++++++++++++++++ security/vbs/probe.c | 103 +++++++++++++ security/vbs/sev_snp.c | 224 ++++++++++++++++++++++++++++ security/vbs/tdx.c | 280 +++++++++++++++++++++++++++++++++++ 13 files changed, 1916 insertions(+) create mode 100644 include/linux/vbs.h create mode 100644 security/vbs/Kconfig create mode 100644 security/vbs/Makefile create mode 100644 security/vbs/arm_cca.c create mode 100644 security/vbs/core.c create mode 100644 security/vbs/hv_vsm.c create mode 100644 security/vbs/internal.h create mode 100644 security/vbs/kvm_planes.c create mode 100644 security/vbs/probe.c create mode 100644 security/vbs/sev_snp.c create mode 100644 security/vbs/tdx.c diff --git a/include/linux/vbs.h b/include/linux/vbs.h new file mode 100644 index 000000000000..c7dedb90d64c --- /dev/null +++ b/include/linux/vbs.h @@ -0,0 +1,204 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * VBS =E2=80=94 Virtualization-Based Security + * + * Transport-agnostic interface between the guest OS (plane-0 / VTL0 / VMP= L2+) + * and the secure kernel (plane-1 / VTL1 / VMPL0 / service TD). + * + * Backends: + * - KVM software planes (KVM_X86_DEFAULT_VM, QEMU-managed vCPU thread= s) + * - AMD SEV-SNP VMPL/SVSM (KVM_X86_SNP_VM, hardware VMPLs, SVSM protoco= l) + * - Intel TDX service TD (future =E2=80=94 separate TD with shared me= mory) + * - Hyper-V VSM (native VTL hypercalls) + * - Arm CCA (RSI host calls from Realm guest to RMM) + * + * The guest kernel calls vbs_*() functions. The active backend translates + * them into the appropriate transport (hypercall, VMGEXIT, shared-memory = IPC). + */ + +#ifndef _LINUX_VBS_H +#define _LINUX_VBS_H + +#include +#include + +struct module; + +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +/* Memory protection flags = */ +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +/* Permissions that the secure kernel can enforce on lower-plane memory. = */ +#define VBS_MEM_READ BIT(0) +#define VBS_MEM_WRITE BIT(1) +#define VBS_MEM_EXEC BIT(2) + +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +/* VTL-call request codes (plane-0 =E2=86=92 plane-1 direction) = */ +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +enum vbs_call_id { + /* Core lifecycle */ + VBS_CALL_INIT =3D 0x0001, /* plane-0 boot complete */ + VBS_CALL_SHUTDOWN =3D 0x0002, /* plane-0 shutting down */ + + /* Memory protection (HEKI) */ + VBS_CALL_PROTECT_MEMORY =3D 0x0100, /* set page permissions */ + VBS_CALL_SEAL_KERNEL =3D 0x0101, /* make kernel text immutable */ + + /* Module authentication */ + VBS_CALL_VALIDATE_MODULE =3D 0x0200, /* verify module signature */ + VBS_CALL_SET_MODULE_PERMS =3D 0x0201, /* set module section perms */ + VBS_CALL_UNLOAD_MODULE =3D 0x0202, /* module being freed */ + + /* Key / certificate management */ + VBS_CALL_ADD_KEY =3D 0x0300, /* add runtime key */ + VBS_CALL_REVOKE_KEY =3D 0x0301, /* revoke a key */ + VBS_CALL_SEND_CERTS =3D 0x0302, /* send system certificates */ + + /* Kexec validation */ + VBS_CALL_KEXEC_VALIDATE =3D 0x0400, /* validate kexec kernel */ + VBS_CALL_KEXEC_INVALIDATE =3D 0x0401, /* invalidate kexec state */ +}; + +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +/* Backend operations (one implementation per platform) = */ +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +/** + * struct vbs_ops - operations provided by an VBS backend + * + * All callbacks are optional; returning -ENOTSUP means the backend does + * not implement that feature. The core VBS layer will call these from + * process context with preemption enabled. + */ +struct vbs_ops { + const char *name; /* "kvm-planes", "svsm", "hv-vsm", =E2=80=A6 */ + + /* + * Lifecycle + */ + + /** @init: called once after plane-0 kernel boot is complete. */ + int (*init)(void); + + /** @shutdown: called before plane-0 halts/reboots. */ + void (*shutdown)(void); + + /* + * Raw VTL call =E2=80=94 send an arbitrary request to the secure kernel + * and wait for a response. @id is the call code, @arg / @arg_size + * point to request-specific data, @resp / @resp_size receive the + * reply. Returns 0 on success, negative errno on failure. + */ + int (*vtl_call)(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size); + + /* + * Memory protection (HEKI) + * + * Ask the secure kernel to enforce @perms (VBS_MEM_*) on the + * physical page range [pfn, pfn + nr_pages) from the perspective + * of the lower plane. + */ + int (*protect_memory)(unsigned long pfn, unsigned long nr_pages, + unsigned int perms); + + /** + * @seal_kernel: make the running kernel's text and rodata immutable. + * After this call, any attempt to write to kernel text from the + * lower plane traps to the secure kernel. + */ + int (*seal_kernel)(void); + + /* + * Module authentication + * + * @validate_module: send a module's ELF blob to the secure kernel + * for signature verification. Returns 0 if the signature is valid. + * + * @set_module_perms: after relocation, set per-section EPT permissions + * for the module (text=3DRX, rodata=3DR, data=3DRW). + * + * @unload_module: notify the secure kernel that a module is being freed + * so it can release EPT overrides. + */ + int (*validate_module)(const void *elf, size_t elf_size, + const void *sig, size_t sig_size); + int (*set_module_perms)(const struct module *mod); + int (*unload_module)(const struct module *mod); + + /* + * Key / certificate management + */ + int (*add_key)(const void *key, size_t key_size, unsigned int flags); + int (*revoke_key)(const void *key_id, size_t id_size); + int (*send_certs)(const void *certs, size_t certs_size); + + /* + * Kexec validation + */ + int (*kexec_validate)(const void *kernel, size_t kernel_size, + const void *sig, size_t sig_size); + int (*kexec_invalidate)(void); +}; + +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +/* Core VBS API (called by guest kernel subsystems) = */ +/* =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +#ifdef CONFIG_VBS + +/** + * vbs_register_backend() - register the platform-specific backend. + * + * Called once during early boot by the platform detection code. + * Only one backend can be active at a time. + */ +int vbs_register_backend(const struct vbs_ops *ops); + +/** + * vbs_available() - returns true if a backend is registered and ready. + */ +bool vbs_available(void); + +/* Convenience wrappers =E2=80=94 each calls through the active backend's = ops. */ +int vbs_protect_memory(unsigned long pfn, unsigned long nr_pages, + unsigned int perms); +int vbs_seal_kernel(void); +int vbs_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size); +int vbs_set_module_perms(const struct module *mod); +int vbs_unload_module(const struct module *mod); +int vbs_add_key(const void *key, size_t key_size, unsigned int flags); +int vbs_revoke_key(const void *key_id, size_t id_size); +int vbs_send_certs(const void *certs, size_t certs_size); +int vbs_kexec_validate(const void *kernel, size_t kernel_size, + const void *sig, size_t sig_size); +int vbs_kexec_invalidate(void); + +#else /* !CONFIG_VBS */ + +static inline bool vbs_available(void) { return false; } +static inline int vbs_protect_memory(unsigned long pfn, + unsigned long nr_pages, unsigned int perms) { return -ENOSYS; } +static inline int vbs_seal_kernel(void) { return -ENOSYS; } +static inline int vbs_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) { return -ENOSYS; } +static inline int vbs_set_module_perms(const struct module *mod) + { return -ENOSYS; } +static inline int vbs_unload_module(const struct module *mod) + { return -ENOSYS; } +static inline int vbs_add_key(const void *key, size_t key_size, + unsigned int flags) { return -ENOSYS; } +static inline int vbs_revoke_key(const void *key_id, size_t id_size) + { return -ENOSYS; } +static inline int vbs_send_certs(const void *certs, size_t certs_size) + { return -ENOSYS; } +static inline int vbs_kexec_validate(const void *kernel, size_t kernel_siz= e, + const void *sig, size_t sig_size) { return -ENOSYS; } +static inline int vbs_kexec_invalidate(void) { return -ENOSYS; } + +#endif /* CONFIG_VBS */ +#endif /* _LINUX_VBS_H */ diff --git a/security/Kconfig b/security/Kconfig index f7bf6cdc6229..31ab9b0fa7d0 100644 --- a/security/Kconfig +++ b/security/Kconfig @@ -299,6 +299,8 @@ config SECURITY_COMMONCAP_KUNIT_TEST =20 If unsure, say N. =20 +source "security/vbs/Kconfig" + source "security/Kconfig.hardening" =20 endmenu diff --git a/security/Makefile b/security/Makefile index 4601230ba442..cd85a7615d01 100644 --- a/security/Makefile +++ b/security/Makefile @@ -27,5 +27,8 @@ obj-$(CONFIG_BPF_LSM) +=3D bpf/ obj-$(CONFIG_SECURITY_LANDLOCK) +=3D landlock/ obj-$(CONFIG_SECURITY_IPE) +=3D ipe/ =20 +# Virtualization-Based Security +obj-$(CONFIG_VBS) +=3D vbs/ + # Object integrity file lists obj-$(CONFIG_INTEGRITY) +=3D integrity/ diff --git a/security/vbs/Kconfig b/security/vbs/Kconfig new file mode 100644 index 000000000000..3d9fb104b1fc --- /dev/null +++ b/security/vbs/Kconfig @@ -0,0 +1,69 @@ +# SPDX-License-Identifier: GPL-2.0-only + +config VBS + bool "Virtualization-Based Security (VBS) support" + depends on (X86_64 || ARM64) && VM_PLANES + help + Enable a transport-agnostic interface between the guest OS + (plane-0 / VTL0 / VMPL2+) and the secure kernel (plane-1 / + VTL1 / VMPL0 / service TD / RMM). + + The core VBS layer dispatches calls from kernel subsystems + (memory protection, module authentication, key management) + to a platform-specific backend such as KVM software planes, + AMD SEV-SNP SVSM, Intel TDX, Hyper-V VSM, or Arm CCA. + + If unsure, say N. + +config VBS_KVM_PLANES + bool "VBS backend: KVM software planes" + depends on VBS && KVM_GUEST + help + VBS backend that uses KVM paravirt hypercalls to communicate + between plane-0 (normal guest) and plane-1 (secure kernel + running in a separate KVM VM plane managed by QEMU). + + Select this if you are running under KVM with VM planes + support enabled. + +config VBS_SEV_SNP + bool "VBS backend: AMD SEV-SNP" + depends on VBS && AMD_MEM_ENCRYPT + help + VBS backend that uses the SVSM (Secure VM Service Module) + protocol to communicate with the SVSM running at VMPL0 + on AMD SEV-SNP platforms. + + Select this if you are running as an SEV-SNP guest with + an SVSM providing security services at VMPL0. + +config VBS_TDX + bool "VBS backend: Intel TDX service TD" + depends on VBS && INTEL_TDX_GUEST + help + VBS backend that uses TDG.VP.VMCALL (TDVMCALL) to communicate + with a service TD providing security services on Intel TDX + platforms. + + Note: Service TD support is still evolving in the TDX + architecture. Select this for development/testing only. + +config VBS_HV_VSM + bool "VBS backend: Hyper-V VSM" + depends on VBS && HYPERV + help + VBS backend that uses native Hyper-V hypercalls to communicate + between VTL0 (normal kernel) and VTL1 (secure kernel). + + Select this if you are running as a Hyper-V guest with + Virtual Secure Mode (VSM) enabled. + +config VBS_ARM_CCA + bool "VBS backend: Arm CCA (Confidential Compute Architecture)" + depends on VBS && ARM64 + help + VBS backend that uses the RSI (Realm Services Interface) to + communicate between a Realm guest and the RMM (Realm Management + Monitor) or a security service on Arm CCA platforms. + + Select this if you are running as a Realm guest under Arm CCA/RME. diff --git a/security/vbs/Makefile b/security/vbs/Makefile new file mode 100644 index 000000000000..4f0f26ef4f71 --- /dev/null +++ b/security/vbs/Makefile @@ -0,0 +1,9 @@ +# SPDX-License-Identifier: GPL-2.0-only +obj-$(CONFIG_VBS) +=3D vbs.o +vbs-y :=3D core.o probe.o + +obj-$(CONFIG_VBS_KVM_PLANES) +=3D kvm_planes.o +obj-$(CONFIG_VBS_SEV_SNP) +=3D sev_snp.o +obj-$(CONFIG_VBS_TDX) +=3D tdx.o +obj-$(CONFIG_VBS_HV_VSM) +=3D hv_vsm.o +obj-$(CONFIG_VBS_ARM_CCA) +=3D arm_cca.o diff --git a/security/vbs/arm_cca.c b/security/vbs/arm_cca.c new file mode 100644 index 000000000000..21c8b00bb1d1 --- /dev/null +++ b/security/vbs/arm_cca.c @@ -0,0 +1,300 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS backend =E2=80=94 Arm CCA (Confidential Compute Architecture) + * + * Uses the RSI (Realm Services Interface) to communicate between the + * Realm guest (plane-0) and the RMM (Realm Management Monitor) or a + * security service running in a higher-privileged realm. + * + * Transport: SMC calls via arm_smccc_smc() using SMC_RSI_HOST_CALL for + * RPC-style requests to the host/monitor, and direct RSI + * commands for memory state management (RIPAS transitions). + * + * Memory model: + * - Protected (RIPAS_RAM): RMM-backed, encrypted, inaccessible to host + * - Shared (RIPAS_EMPTY): Host-backed, used for I/O and communication + * - The highest IPA bit marks shared vs protected pages + */ + +#include "internal.h" + +#include +#include +#include + +#include + +/* =E2=94=80=E2=94=80 VBS host-call command IDs =E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +/* + * VBS requests are sent to the host via SMC_RSI_HOST_CALL. The host + * call structure is placed in a shared (RIPAS_EMPTY) page. The first + * 16 bytes encode the VBS-specific command header. + */ +#define CCA_VBS_MAGIC 0x56425343 /* "VBSC" */ + +struct vbs_cca_host_req { + __u32 magic; /* CCA_VBS_MAGIC */ + __u32 call_id; /* enum vbs_call_id / CCA_VBS_* cmd */ + __u32 arg_size; /* bytes of payload following this hdr */ + __u32 reserved; + __u8 payload[]; +} __packed; + +struct vbs_cca_host_resp { + __s32 status; /* 0 =3D success, negative errno */ + __u32 resp_size; + __u8 payload[]; +} __packed; + +/* VBS sub-commands */ +#define CCA_VBS_INIT 0 +#define CCA_VBS_SHUTDOWN 1 +#define CCA_VBS_PROTECT_MEMORY 2 +#define CCA_VBS_SEAL_KERNEL 3 +#define CCA_VBS_VALIDATE_MODULE 4 +#define CCA_VBS_SET_MODULE_PERMS 5 +#define CCA_VBS_UNLOAD_MODULE 6 +#define CCA_VBS_ADD_KEY 7 +#define CCA_VBS_REVOKE_KEY 8 +#define CCA_VBS_SEND_CERTS 9 +#define CCA_VBS_KEXEC_VALIDATE 10 +#define CCA_VBS_KEXEC_INVALIDATE 11 + +/* Shared pages for host communication (RIPAS_EMPTY / decrypted) */ +static void *cca_req_page; +static void *cca_resp_page; + +/* =E2=94=80=E2=94=80 low-level host call =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80 */ + +static int cca_vbs_host_call(u32 cmd, const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + struct vbs_cca_host_req *req; + struct vbs_cca_host_resp *rsp; + struct arm_smccc_res res; + unsigned long ret; + + if (!cca_req_page || !cca_resp_page) + return -ENOMEM; + + if (arg_size > PAGE_SIZE - sizeof(*req)) + return -E2BIG; + + /* Build request in the shared page */ + req =3D cca_req_page; + memset(req, 0, PAGE_SIZE); + req->magic =3D CCA_VBS_MAGIC; + req->call_id =3D cmd; + req->arg_size =3D arg_size; + if (arg_size && arg) + memcpy(req->payload, arg, arg_size); + + memset(cca_resp_page, 0, PAGE_SIZE); + + /* + * SMC_RSI_HOST_CALL: arg1 =3D IPA of host call structure. + * The host (VMM) reads the request, processes it, writes the + * response into cca_resp_page, then returns control. + */ + arm_smccc_smc(SMC_RSI_HOST_CALL, virt_to_phys(cca_req_page), + 0, 0, 0, 0, 0, 0, &res); + ret =3D res.a0; + if (ret !=3D RSI_SUCCESS) { + pr_err_ratelimited("vbs-cca: RSI host call failed (%lu)\n", + ret); + return -EIO; + } + + /* Read response */ + rsp =3D cca_resp_page; + if (rsp->status) + return rsp->status; + + if (resp && resp_size) { + size_t copy =3D min_t(size_t, resp_size, rsp->resp_size); + + memcpy(resp, rsp->payload, copy); + } + return 0; +} + +static int cca_vbs_vtl_call(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + return cca_vbs_host_call(id, arg, arg_size, resp, resp_size); +} + +/* =E2=94=80=E2=94=80 memory protection =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +/* + * On Arm CCA, memory protection is handled natively via RIPAS transitions. + * The RMM enforces that protected (RIPAS_RAM) pages are inaccessible to + * the host. For VBS-style per-page permission control (R/W/X), we + * forward the request to the security service via a host call. + */ + +struct vbs_cca_protect_args { + __u64 pfn; + __u64 nr_pages; + __u32 perms; +} __packed; + +static int cca_vbs_protect_memory(unsigned long pfn, unsigned long nr_page= s, + unsigned int perms) +{ + struct vbs_cca_protect_args args =3D { + .pfn =3D pfn, + .nr_pages =3D nr_pages, + .perms =3D perms, + }; + + return cca_vbs_host_call(CCA_VBS_PROTECT_MEMORY, + &args, sizeof(args), NULL, 0); +} + +static int cca_vbs_seal_kernel(void) +{ + return cca_vbs_host_call(CCA_VBS_SEAL_KERNEL, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +static int cca_vbs_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) +{ + return cca_vbs_host_call(CCA_VBS_VALIDATE_MODULE, NULL, 0, NULL, 0); +} + +static int cca_vbs_set_module_perms(const struct module *mod) +{ + return cca_vbs_host_call(CCA_VBS_SET_MODULE_PERMS, NULL, 0, NULL, 0); +} + +static int cca_vbs_unload_module(const struct module *mod) +{ + return cca_vbs_host_call(CCA_VBS_UNLOAD_MODULE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 key / certificate management =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int cca_vbs_add_key(const void *key, size_t key_size, + unsigned int flags) +{ + return cca_vbs_host_call(CCA_VBS_ADD_KEY, key, key_size, NULL, 0); +} + +static int cca_vbs_revoke_key(const void *key_id, size_t id_size) +{ + return cca_vbs_host_call(CCA_VBS_REVOKE_KEY, key_id, id_size, NULL, 0); +} + +static int cca_vbs_send_certs(const void *certs, size_t certs_size) +{ + return cca_vbs_host_call(CCA_VBS_SEND_CERTS, + certs, certs_size, NULL, 0); +} + +/* =E2=94=80=E2=94=80 kexec validation =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int cca_vbs_kexec_validate(const void *kernel, size_t kernel_size, + const void *sig, size_t sig_size) +{ + return cca_vbs_host_call(CCA_VBS_KEXEC_VALIDATE, NULL, 0, NULL, 0); +} + +static int cca_vbs_kexec_invalidate(void) +{ + return cca_vbs_host_call(CCA_VBS_KEXEC_INVALIDATE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 lifecycle =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int cca_vbs_init(void) +{ + int ret; + + /* + * Allocate shared pages for host communication. Convert them + * to RIPAS_EMPTY so the host/VMM can access them. + */ + cca_req_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + cca_resp_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + if (!cca_req_page || !cca_resp_page) { + ret =3D -ENOMEM; + goto fail; + } + + /* Mark as shared (RIPAS_EMPTY) for host access */ + ret =3D set_memory_decrypted((unsigned long)cca_req_page, 1); + if (ret) + goto fail; + ret =3D set_memory_decrypted((unsigned long)cca_resp_page, 1); + if (ret) + goto fail_re_encrypt_req; + + ret =3D cca_vbs_host_call(CCA_VBS_INIT, NULL, 0, NULL, 0); + if (ret) { + pr_err("vbs-cca: realm VBS init failed (%d)\n", ret); + goto fail_re_encrypt; + } + + pr_info("vbs-cca: connected to Arm CCA security service\n"); + return 0; + +fail_re_encrypt: + set_memory_encrypted((unsigned long)cca_resp_page, 1); +fail_re_encrypt_req: + set_memory_encrypted((unsigned long)cca_req_page, 1); +fail: + free_page((unsigned long)cca_req_page); + free_page((unsigned long)cca_resp_page); + cca_req_page =3D cca_resp_page =3D NULL; + return ret; +} + +static void cca_vbs_shutdown(void) +{ + cca_vbs_host_call(CCA_VBS_SHUTDOWN, NULL, 0, NULL, 0); + + if (cca_resp_page) { + set_memory_encrypted((unsigned long)cca_resp_page, 1); + free_page((unsigned long)cca_resp_page); + } + if (cca_req_page) { + set_memory_encrypted((unsigned long)cca_req_page, 1); + free_page((unsigned long)cca_req_page); + } + cca_req_page =3D cca_resp_page =3D NULL; +} + +/* =E2=94=80=E2=94=80 ops table & registration =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static const struct vbs_ops cca_vbs_ops =3D { + .name =3D "arm-cca", + .init =3D cca_vbs_init, + .shutdown =3D cca_vbs_shutdown, + .vtl_call =3D cca_vbs_vtl_call, + .protect_memory =3D cca_vbs_protect_memory, + .seal_kernel =3D cca_vbs_seal_kernel, + .validate_module =3D cca_vbs_validate_module, + .set_module_perms =3D cca_vbs_set_module_perms, + .unload_module =3D cca_vbs_unload_module, + .add_key =3D cca_vbs_add_key, + .revoke_key =3D cca_vbs_revoke_key, + .send_certs =3D cca_vbs_send_certs, + .kexec_validate =3D cca_vbs_kexec_validate, + .kexec_invalidate =3D cca_vbs_kexec_invalidate, +}; + +/* =E2=94=80=E2=94=80 detection & probe (called from probe.c) =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +bool __init vbs_cca_detect(void) +{ + return is_realm_world(); +} + +const struct vbs_ops *vbs_cca_get_ops(void) +{ + return &cca_vbs_ops; +} diff --git a/security/vbs/core.c b/security/vbs/core.c new file mode 100644 index 000000000000..352590d88136 --- /dev/null +++ b/security/vbs/core.c @@ -0,0 +1,166 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS =E2=80=94 Virtualization-Based Security core + * + * Dispatches calls from guest kernel subsystems to the active + * platform-specific backend (KVM planes, SVSM, Hyper-V VSM, =E2=80=A6). + */ + +#include "internal.h" + +#include + +static const struct vbs_ops *vbs_backend; +static DEFINE_MUTEX(vbs_lock); + +int vbs_register_backend(const struct vbs_ops *ops) +{ + int ret =3D 0; + + if (!ops || !ops->name) + return -EINVAL; + + mutex_lock(&vbs_lock); + if (vbs_backend) { + pr_err("vbs: backend \"%s\" already registered, rejecting \"%s\"\n", + vbs_backend->name, ops->name); + ret =3D -EBUSY; + } else { + vbs_backend =3D ops; + pr_info("vbs: registered backend \"%s\"\n", ops->name); + } + mutex_unlock(&vbs_lock); + return ret; +} +EXPORT_SYMBOL_GPL(vbs_register_backend); + +bool vbs_available(void) +{ + return READ_ONCE(vbs_backend) !=3D NULL; +} +EXPORT_SYMBOL_GPL(vbs_available); + +/* =E2=94=80=E2=94=80 convenience wrappers =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80 */ + +int vbs_protect_memory(unsigned long pfn, unsigned long nr_pages, + unsigned int perms) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->protect_memory) + return -EOPNOTSUPP; + return ops->protect_memory(pfn, nr_pages, perms); +} +EXPORT_SYMBOL_GPL(vbs_protect_memory); + +int vbs_seal_kernel(void) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->seal_kernel) + return -EOPNOTSUPP; + return ops->seal_kernel(); +} +EXPORT_SYMBOL_GPL(vbs_seal_kernel); + +int vbs_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->validate_module) + return -EOPNOTSUPP; + return ops->validate_module(elf, elf_size, sig, sig_size); +} +EXPORT_SYMBOL_GPL(vbs_validate_module); + +int vbs_set_module_perms(const struct module *mod) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->set_module_perms) + return -EOPNOTSUPP; + return ops->set_module_perms(mod); +} +EXPORT_SYMBOL_GPL(vbs_set_module_perms); + +int vbs_unload_module(const struct module *mod) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->unload_module) + return -EOPNOTSUPP; + return ops->unload_module(mod); +} +EXPORT_SYMBOL_GPL(vbs_unload_module); + +int vbs_add_key(const void *key, size_t key_size, unsigned int flags) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->add_key) + return -EOPNOTSUPP; + return ops->add_key(key, key_size, flags); +} +EXPORT_SYMBOL_GPL(vbs_add_key); + +int vbs_revoke_key(const void *key_id, size_t id_size) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->revoke_key) + return -EOPNOTSUPP; + return ops->revoke_key(key_id, id_size); +} +EXPORT_SYMBOL_GPL(vbs_revoke_key); + +int vbs_send_certs(const void *certs, size_t certs_size) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->send_certs) + return -EOPNOTSUPP; + return ops->send_certs(certs, certs_size); +} +EXPORT_SYMBOL_GPL(vbs_send_certs); + +int vbs_kexec_validate(const void *kernel, size_t kernel_size, + const void *sig, size_t sig_size) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->kexec_validate) + return -EOPNOTSUPP; + return ops->kexec_validate(kernel, kernel_size, sig, sig_size); +} +EXPORT_SYMBOL_GPL(vbs_kexec_validate); + +int vbs_kexec_invalidate(void) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + + if (!ops) + return -ENODEV; + if (!ops->kexec_invalidate) + return -EOPNOTSUPP; + return ops->kexec_invalidate(); +} +EXPORT_SYMBOL_GPL(vbs_kexec_invalidate); diff --git a/security/vbs/hv_vsm.c b/security/vbs/hv_vsm.c new file mode 100644 index 000000000000..981ec7fa95e3 --- /dev/null +++ b/security/vbs/hv_vsm.c @@ -0,0 +1,257 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS backend =E2=80=94 Hyper-V VSM (Virtual Secure Mode) + * + * Uses native Hyper-V hypercalls to communicate between VTL0 (normal + * kernel) and VTL1 (secure kernel / SKCI). + * + * Transport: hv_do_hypercall() with VTL-targeted input pages. + * + * On Hyper-V the VTL architecture is a first-class feature: + * - VTL0 runs the normal OS kernel. + * - VTL1 runs the secure kernel that enforces HVCI / Credential Guard. + * - VTL switches are performed via HvVtlCall / HvVtlReturn hypercalls. + * - Memory protection is enforced per-VTL via the second-level address + * translation (SLAT / EPT / NPT) controlled by the hypervisor. + */ + +#include "internal.h" + +#include +#include + +#include +#include + +/* =E2=94=80=E2=94=80 Hyper-V VTL call / return hypercall numbers =E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +#define HVCALL_VTL_CALL 0x0011 +#define HVCALL_VTL_RETURN 0x0012 + +/* + * VBS-specific hypercall =E2=80=94 used to send structured VBS requests t= o VTL1. + * This sits in the vendor-extension range and is routed by the Hyper-V + * secure kernel to the appropriate VBS service handler. + */ +#define HVCALL_VBS_REQUEST 0x0200 + +/* =E2=94=80=E2=94=80 shared-memory request / response layout =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +struct vbs_hv_request { + __u32 call_id; /* enum vbs_call_id */ + __u32 arg_size; /* bytes of payload following this hdr */ + __u8 payload[]; +} __packed; + +struct vbs_hv_response { + __s32 status; /* 0 =3D success, negative errno */ + __u32 resp_size; + __u8 payload[]; +} __packed; + +/* Hypercall input / output pages (one page each, allocated once) */ +static void *hv_input_page; +static void *hv_output_page; + +/* =E2=94=80=E2=94=80 low-level VTL call =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int hv_vsm_vtl_call(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + struct vbs_hv_request *req; + struct vbs_hv_response *rsp; + u64 status; + + if (!hv_input_page || !hv_output_page) + return -ENOMEM; + + if (arg_size > PAGE_SIZE - sizeof(*req)) + return -E2BIG; + + /* Build request in the hypercall input page */ + req =3D hv_input_page; + memset(req, 0, PAGE_SIZE); + req->call_id =3D id; + req->arg_size =3D arg_size; + if (arg_size && arg) + memcpy(req->payload, arg, arg_size); + + memset(hv_output_page, 0, PAGE_SIZE); + + status =3D hv_do_hypercall(HVCALL_VBS_REQUEST, + hv_input_page, hv_output_page); + if (!hv_result_success(status)) { + pr_err_ratelimited("vbs-hv: hypercall failed (0x%llx)\n", + status); + return -EIO; + } + + /* Read response from the output page */ + rsp =3D hv_output_page; + if (rsp->status) + return rsp->status; + + if (resp && resp_size) { + size_t copy =3D min_t(size_t, resp_size, rsp->resp_size); + + memcpy(resp, rsp->payload, copy); + } + return 0; +} + +/* =E2=94=80=E2=94=80 memory protection =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +struct vbs_hv_protect_args { + __u64 pfn; + __u64 nr_pages; + __u32 perms; +} __packed; + +static int hv_vsm_protect_memory(unsigned long pfn, unsigned long nr_pages, + unsigned int perms) +{ + struct vbs_hv_protect_args args =3D { + .pfn =3D pfn, + .nr_pages =3D nr_pages, + .perms =3D perms, + }; + + return hv_vsm_vtl_call(VBS_CALL_PROTECT_MEMORY, + &args, sizeof(args), NULL, 0); +} + +static int hv_vsm_seal_kernel(void) +{ + return hv_vsm_vtl_call(VBS_CALL_SEAL_KERNEL, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +static int hv_vsm_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) +{ + return hv_vsm_vtl_call(VBS_CALL_VALIDATE_MODULE, NULL, 0, NULL, 0); +} + +static int hv_vsm_set_module_perms(const struct module *mod) +{ + return hv_vsm_vtl_call(VBS_CALL_SET_MODULE_PERMS, NULL, 0, NULL, 0); +} + +static int hv_vsm_unload_module(const struct module *mod) +{ + return hv_vsm_vtl_call(VBS_CALL_UNLOAD_MODULE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 key / certificate management =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int hv_vsm_add_key(const void *key, size_t key_size, + unsigned int flags) +{ + return hv_vsm_vtl_call(VBS_CALL_ADD_KEY, key, key_size, NULL, 0); +} + +static int hv_vsm_revoke_key(const void *key_id, size_t id_size) +{ + return hv_vsm_vtl_call(VBS_CALL_REVOKE_KEY, key_id, id_size, NULL, 0); +} + +static int hv_vsm_send_certs(const void *certs, size_t certs_size) +{ + return hv_vsm_vtl_call(VBS_CALL_SEND_CERTS, + certs, certs_size, NULL, 0); +} + +/* =E2=94=80=E2=94=80 kexec validation =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int hv_vsm_kexec_validate(const void *kernel, size_t kernel_size, + const void *sig, size_t sig_size) +{ + return hv_vsm_vtl_call(VBS_CALL_KEXEC_VALIDATE, NULL, 0, NULL, 0); +} + +static int hv_vsm_kexec_invalidate(void) +{ + return hv_vsm_vtl_call(VBS_CALL_KEXEC_INVALIDATE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 lifecycle =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int hv_vsm_init(void) +{ + int ret; + + /* + * Use the Hyper-V provided hypercall input/output pages. + * Allocate our own pair so we don't conflict with other users. + */ + hv_input_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + hv_output_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + if (!hv_input_page || !hv_output_page) { + ret =3D -ENOMEM; + goto fail; + } + + ret =3D hv_vsm_vtl_call(VBS_CALL_INIT, NULL, 0, NULL, 0); + if (ret) { + pr_err("vbs-hv: VTL1 secure kernel INIT failed (%d)\n", ret); + goto fail; + } + + pr_info("vbs-hv: connected to Hyper-V VTL1 secure kernel\n"); + return 0; + +fail: + free_page((unsigned long)hv_input_page); + free_page((unsigned long)hv_output_page); + hv_input_page =3D hv_output_page =3D NULL; + return ret; +} + +static void hv_vsm_shutdown(void) +{ + hv_vsm_vtl_call(VBS_CALL_SHUTDOWN, NULL, 0, NULL, 0); + free_page((unsigned long)hv_input_page); + free_page((unsigned long)hv_output_page); + hv_input_page =3D hv_output_page =3D NULL; +} + +/* =E2=94=80=E2=94=80 ops table & registration =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static const struct vbs_ops hv_vsm_ops =3D { + .name =3D "hv-vsm", + .init =3D hv_vsm_init, + .shutdown =3D hv_vsm_shutdown, + .vtl_call =3D hv_vsm_vtl_call, + .protect_memory =3D hv_vsm_protect_memory, + .seal_kernel =3D hv_vsm_seal_kernel, + .validate_module =3D hv_vsm_validate_module, + .set_module_perms =3D hv_vsm_set_module_perms, + .unload_module =3D hv_vsm_unload_module, + .add_key =3D hv_vsm_add_key, + .revoke_key =3D hv_vsm_revoke_key, + .send_certs =3D hv_vsm_send_certs, + .kexec_validate =3D hv_vsm_kexec_validate, + .kexec_invalidate =3D hv_vsm_kexec_invalidate, +}; + +/* =E2=94=80=E2=94=80 detection & probe (called from probe.c) =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +bool __init vbs_hv_vsm_detect(void) +{ + if (!hv_is_hyperv_initialized()) + return false; + + if (ms_hyperv.vtl !=3D 0) { + pr_debug("vbs-hv: not at VTL0 (vtl=3D%u), skipping\n", + ms_hyperv.vtl); + return false; + } + + return true; +} + +const struct vbs_ops *vbs_hv_vsm_get_ops(void) +{ + return &hv_vsm_ops; +} diff --git a/security/vbs/internal.h b/security/vbs/internal.h new file mode 100644 index 000000000000..415621c993b8 --- /dev/null +++ b/security/vbs/internal.h @@ -0,0 +1,41 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * VBS internal header =E2=80=94 shared between probe.c and backend implem= entations. + */ +#ifndef _SECURITY_VBS_INTERNAL_H +#define _SECURITY_VBS_INTERNAL_H + +#include +#include +#include +#include +#include + +/* Each backend exports a detect + get_ops pair for the centralized probe.= */ + +#ifdef CONFIG_VBS_SEV_SNP +bool __init vbs_sev_snp_detect(void); +const struct vbs_ops *vbs_sev_snp_get_ops(void); +#endif + +#ifdef CONFIG_VBS_TDX +bool __init vbs_tdx_detect(void); +const struct vbs_ops *vbs_tdx_get_ops(void); +#endif + +#ifdef CONFIG_VBS_ARM_CCA +bool __init vbs_cca_detect(void); +const struct vbs_ops *vbs_cca_get_ops(void); +#endif + +#ifdef CONFIG_VBS_HV_VSM +bool __init vbs_hv_vsm_detect(void); +const struct vbs_ops *vbs_hv_vsm_get_ops(void); +#endif + +#ifdef CONFIG_VBS_KVM_PLANES +bool __init vbs_kvm_planes_detect(void); +const struct vbs_ops *vbs_kvm_planes_get_ops(void); +#endif + +#endif /* _SECURITY_VBS_INTERNAL_H */ diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c new file mode 100644 index 000000000000..3526f7c429c3 --- /dev/null +++ b/security/vbs/kvm_planes.c @@ -0,0 +1,258 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS backend =E2=80=94 KVM software planes + * + * Uses KVM paravirt hypercalls to communicate between plane-0 (normal + * guest kernel) and plane-1 (secure kernel running in a separate KVM + * plane managed by QEMU). + * + * Transport: kvm_hypercall{0..4}() =E2=86=92 KVM_EXIT_HYPERCALL =E2=86=92= QEMU =E2=86=92 plane-1 + * + * The shared-memory VTL-call protocol works as follows: + * 1. Plane-0 fills a request buffer in shared memory. + * 2. Plane-0 issues a KVM hypercall carrying the physical address + * and size of the request. + * 3. QEMU (or the host) delivers the request to the plane-1 vCPU. + * 4. Plane-1 processes the request and writes a response. + * 5. Plane-0 reads the response from shared memory. + */ + +#include "internal.h" + +#include +#include +#include +#include + +/* =E2=94=80=E2=94=80 hypercall numbers for VBS VTL calls (plane-0 =E2=86= =92 plane-1) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ +/* + * These extend the existing KVM_HC_* numbering. The host (KVM + QEMU) + * intercepts them and routes them to the secure-kernel plane. + */ +#define KVM_HC_VBS_VTL_CALL 15 + +/* =E2=94=80=E2=94=80 shared-memory request / response layout =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +struct vbs_kvm_request { + __u32 call_id; /* enum vbs_call_id */ + __u32 arg_size; /* bytes of payload following this hdr */ + __u8 payload[]; /* variable-length argument data */ +} __packed; + +struct vbs_kvm_response { + __s32 status; /* 0 =3D success, negative errno */ + __u32 resp_size; /* bytes of payload following this hdr */ + __u8 payload[]; /* variable-length response data */ +} __packed; + +/* + * A single page is used for each direction. That gives ~4 KiB of + * payload per call, which is enough for all current VBS operations. + */ +static void *kvm_req_page; /* request (plane-0 writes, plane-1 reads) */ +static void *kvm_resp_page; /* response (plane-1 writes, plane-0 reads) */ + +/* =E2=94=80=E2=94=80 low-level VTL call =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int kvm_planes_vtl_call(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + struct vbs_kvm_request *req; + struct vbs_kvm_response *rsp; + long hc_ret; + + if (!kvm_req_page || !kvm_resp_page) + return -ENOMEM; + + if (arg_size > PAGE_SIZE - sizeof(*req)) + return -E2BIG; + + /* Build request in the shared page */ + req =3D kvm_req_page; + req->call_id =3D id; + req->arg_size =3D arg_size; + if (arg_size && arg) + memcpy(req->payload, arg, arg_size); + + /* Issue hypercall: pass physical addresses of req & resp pages */ + hc_ret =3D kvm_hypercall2(KVM_HC_VBS_VTL_CALL, + virt_to_phys(kvm_req_page), + virt_to_phys(kvm_resp_page)); + if (hc_ret) { + pr_err_ratelimited("vbs-kvm: hypercall failed (%ld)\n", hc_ret); + return -EIO; + } + + /* Read response */ + rsp =3D kvm_resp_page; + if (rsp->status) + return rsp->status; + + if (resp && resp_size) { + size_t copy =3D min_t(size_t, resp_size, rsp->resp_size); + + memcpy(resp, rsp->payload, copy); + } + return 0; +} + +/* =E2=94=80=E2=94=80 memory protection =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +struct vbs_protect_args { + __u64 pfn; + __u64 nr_pages; + __u32 perms; +} __packed; + +static int kvm_planes_protect_memory(unsigned long pfn, + unsigned long nr_pages, + unsigned int perms) +{ + struct vbs_protect_args args =3D { + .pfn =3D pfn, + .nr_pages =3D nr_pages, + .perms =3D perms, + }; + + return kvm_planes_vtl_call(VBS_CALL_PROTECT_MEMORY, + &args, sizeof(args), NULL, 0); +} + +static int kvm_planes_seal_kernel(void) +{ + return kvm_planes_vtl_call(VBS_CALL_SEAL_KERNEL, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +static int kvm_planes_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) +{ + /* + * Module blobs can be large =E2=80=94 for the KVM planes backend we pass + * the physical address and size to plane-1 via the VTL call and + * let plane-1 map/read the pages directly from its EPT view. + * For now, a stub that signals "not yet implemented". + */ + return kvm_planes_vtl_call(VBS_CALL_VALIDATE_MODULE, + NULL, 0, NULL, 0); +} + +static int kvm_planes_set_module_perms(const struct module *mod) +{ + return kvm_planes_vtl_call(VBS_CALL_SET_MODULE_PERMS, + NULL, 0, NULL, 0); +} + +static int kvm_planes_unload_module(const struct module *mod) +{ + return kvm_planes_vtl_call(VBS_CALL_UNLOAD_MODULE, + NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 key / certificate management =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int kvm_planes_add_key(const void *key, size_t key_size, + unsigned int flags) +{ + return kvm_planes_vtl_call(VBS_CALL_ADD_KEY, key, key_size, NULL, 0); +} + +static int kvm_planes_revoke_key(const void *key_id, size_t id_size) +{ + return kvm_planes_vtl_call(VBS_CALL_REVOKE_KEY, + key_id, id_size, NULL, 0); +} + +static int kvm_planes_send_certs(const void *certs, size_t certs_size) +{ + return kvm_planes_vtl_call(VBS_CALL_SEND_CERTS, + certs, certs_size, NULL, 0); +} + +/* =E2=94=80=E2=94=80 kexec validation =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int kvm_planes_kexec_validate(const void *kernel, size_t kernel_siz= e, + const void *sig, size_t sig_size) +{ + return kvm_planes_vtl_call(VBS_CALL_KEXEC_VALIDATE, + NULL, 0, NULL, 0); +} + +static int kvm_planes_kexec_invalidate(void) +{ + return kvm_planes_vtl_call(VBS_CALL_KEXEC_INVALIDATE, + NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 lifecycle =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int kvm_planes_init(void) +{ + int ret; + + kvm_req_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + kvm_resp_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + if (!kvm_req_page || !kvm_resp_page) { + ret =3D -ENOMEM; + goto fail; + } + + ret =3D kvm_planes_vtl_call(VBS_CALL_INIT, NULL, 0, NULL, 0); + if (ret) { + pr_err("vbs-kvm: plane-1 INIT call failed (%d)\n", ret); + goto fail; + } + + pr_info("vbs-kvm: connected to plane-1 secure kernel\n"); + return 0; +fail: + free_page((unsigned long)kvm_req_page); + free_page((unsigned long)kvm_resp_page); + kvm_req_page =3D kvm_resp_page =3D NULL; + return ret; +} + +static void kvm_planes_shutdown(void) +{ + kvm_planes_vtl_call(VBS_CALL_SHUTDOWN, NULL, 0, NULL, 0); + free_page((unsigned long)kvm_req_page); + free_page((unsigned long)kvm_resp_page); + kvm_req_page =3D kvm_resp_page =3D NULL; +} + +/* =E2=94=80=E2=94=80 ops table & registration =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static const struct vbs_ops kvm_planes_ops =3D { + .name =3D "kvm-planes", + .init =3D kvm_planes_init, + .shutdown =3D kvm_planes_shutdown, + .vtl_call =3D kvm_planes_vtl_call, + .protect_memory =3D kvm_planes_protect_memory, + .seal_kernel =3D kvm_planes_seal_kernel, + .validate_module =3D kvm_planes_validate_module, + .set_module_perms =3D kvm_planes_set_module_perms, + .unload_module =3D kvm_planes_unload_module, + .add_key =3D kvm_planes_add_key, + .revoke_key =3D kvm_planes_revoke_key, + .send_certs =3D kvm_planes_send_certs, + .kexec_validate =3D kvm_planes_kexec_validate, + .kexec_invalidate =3D kvm_planes_kexec_invalidate, +}; + +/* =E2=94=80=E2=94=80 detection & probe (called from probe.c) =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +bool __init vbs_kvm_planes_detect(void) +{ + if (!kvm_para_available()) { + pr_debug("vbs-kvm: KVM paravirt not available\n"); + return false; + } + return true; +} + +const struct vbs_ops *vbs_kvm_planes_get_ops(void) +{ + return &kvm_planes_ops; +} diff --git a/security/vbs/probe.c b/security/vbs/probe.c new file mode 100644 index 000000000000..292f3663a996 --- /dev/null +++ b/security/vbs/probe.c @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS platform detection and backend selection + * + * Single initcall that probes the platform and registers the appropriate + * VBS backend. Detection order (first match wins): + * + * 1. Hardware CoCo =E2=80=94 these are mutually exclusive by nature: + * a. AMD SEV-SNP with SVSM (VMPL > 0, SVSM at VMPL0) + * b. Intel TDX service TD (running inside a Trust Domain) + * c. Arm CCA (Realm guest with RSI) + * + * 2. Hypervisor-specific: + * d. Hyper-V VSM (Hyper-V guest at VTL0) + * + * 3. Software emulation: + * e. KVM software planes (KVM paravirt guest) + * + * Only one backend can be active. The first successful probe wins. + */ + +#include "internal.h" + +/* Stubs for backends not configured */ +#ifndef CONFIG_VBS_SEV_SNP +static inline bool vbs_sev_snp_detect(void) { return false; } +static inline const struct vbs_ops *vbs_sev_snp_get_ops(void) { return NUL= L; } +#endif +#ifndef CONFIG_VBS_TDX +static inline bool vbs_tdx_detect(void) { return false; } +static inline const struct vbs_ops *vbs_tdx_get_ops(void) { return NULL; } +#endif +#ifndef CONFIG_VBS_ARM_CCA +static inline bool vbs_cca_detect(void) { return false; } +static inline const struct vbs_ops *vbs_cca_get_ops(void) { return NULL; } +#endif +#ifndef CONFIG_VBS_HV_VSM +static inline bool vbs_hv_vsm_detect(void) { return false; } +static inline const struct vbs_ops *vbs_hv_vsm_get_ops(void) { return NULL= ; } +#endif +#ifndef CONFIG_VBS_KVM_PLANES +static inline bool vbs_kvm_planes_detect(void) { return false; } +static inline const struct vbs_ops *vbs_kvm_planes_get_ops(void) { return = NULL; } +#endif + +/* =E2=94=80=E2=94=80 probe table =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +struct vbs_probe_entry { + const char *name; + bool (*detect)(void); + const struct vbs_ops *(*get_ops)(void); +}; + +static const struct vbs_probe_entry vbs_probe_table[] __initconst =3D { + /* + * Hardware confidential-compute backends first. + * These are mutually exclusive =E2=80=94 a machine is SEV-SNP *or* TDX + * *or* CCA, never more than one. + */ + { "AMD SEV-SNP", vbs_sev_snp_detect, vbs_sev_snp_get_ops }, + { "Intel TDX", vbs_tdx_detect, vbs_tdx_get_ops }, + { "Arm CCA", vbs_cca_detect, vbs_cca_get_ops }, + + /* Hypervisor-specific */ + { "Hyper-V VSM", vbs_hv_vsm_detect, vbs_hv_vsm_get_ops }, + + /* Software emulation (lowest priority) */ + { "KVM planes", vbs_kvm_planes_detect, vbs_kvm_planes_get_ops }, +}; + +/* =E2=94=80=E2=94=80 single boot-time probe =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80 */ + +static int __init vbs_probe_init(void) +{ + int i, ret; + + for (i =3D 0; i < ARRAY_SIZE(vbs_probe_table); i++) { + const struct vbs_probe_entry *e =3D &vbs_probe_table[i]; + + if (!e->detect()) + continue; + + pr_info("vbs: detected %s platform\n", e->name); + + ret =3D vbs_register_backend(e->get_ops()); + if (ret) { + pr_err("vbs: failed to register %s backend (%d)\n", + e->name, ret); + return ret; + } + return 0; + } + + pr_debug("vbs: no supported platform detected\n"); + return 0; +} + +/* + * Run at device_initcall level: platform detection (CPUID, MSRs, SMCCC) + * is complete by this point, but subsystems that consume VBS (module + * loading, HEKI) have not yet started. + */ +device_initcall(vbs_probe_init); diff --git a/security/vbs/sev_snp.c b/security/vbs/sev_snp.c new file mode 100644 index 000000000000..510a2245a0e7 --- /dev/null +++ b/security/vbs/sev_snp.c @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS backend =E2=80=94 AMD SEV-SNP + * + * Uses the SVSM (Secure VM Service Module) protocol to communicate + * between the guest (VMPL2+) and the SVSM running at VMPL0. + * + * Transport: VMGEXIT with SVM_VMGEXIT_SNP_RUN_VMPL exit code, + * parameters passed via the SVSM Calling Area (CAA). + * + * The SVSM already provides core services (PVALIDATE, attestation, + * vTPM). This backend extends it with VBS-specific calls for + * memory protection, module authentication, and key management + * using a new VBS SVSM protocol number. + */ + +#include "internal.h" + +#include + +#include + +/* + * VBS SEV-SNP protocol =E2=80=94 extends the existing SVSM protocol numbe= ring. + * Protocol 0 =3D core, 1 =3D attestation, 2 =3D vTPM, 3 =3D VBS. + */ +#define SEV_SNP_VBS_CALL(x) ((3ULL << 32) | (x)) + +/* VBS-specific SEV-SNP call IDs (mapped from enum vbs_call_id) */ +#define SEV_SNP_VBS_INIT 0 +#define SEV_SNP_VBS_SHUTDOWN 1 +#define SEV_SNP_VBS_PROTECT_MEMORY 2 +#define SEV_SNP_VBS_SEAL_KERNEL 3 +#define SEV_SNP_VBS_VALIDATE_MODULE 4 +#define SEV_SNP_VBS_SET_MODULE_PERMS 5 +#define SEV_SNP_VBS_UNLOAD_MODULE 6 +#define SEV_SNP_VBS_ADD_KEY 7 +#define SEV_SNP_VBS_REVOKE_KEY 8 +#define SEV_SNP_VBS_SEND_CERTS 9 +#define SEV_SNP_VBS_KEXEC_VALIDATE 10 +#define SEV_SNP_VBS_KEXEC_INVALIDATE 11 + +/* =E2=94=80=E2=94=80 low-level VTL call via SEV-SNP =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80 */ + +/* + * Issue a VBS call through the SVSM protocol. + * + * The CAA svsm_buffer is used to pass request/response data. + * RAX encodes the protocol (3 =3D VBS) and call ID. + * RCX carries the physical address of any auxiliary data buffer. + * RDX carries the size of the auxiliary data. + */ +static int sev_snp_vbs_call(u32 call_id, const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + struct svsm_call call =3D {}; + int ret; + + call.rax =3D SEV_SNP_VBS_CALL(call_id); + if (arg && arg_size) { + call.rcx =3D __pa(arg); + call.rdx =3D arg_size; + } + if (resp && resp_size) { + call.r8 =3D __pa(resp); + call.r9 =3D resp_size; + } + + ret =3D svsm_perform_call_protocol(&call); + if (ret) + pr_err_ratelimited("vbs-sev-snp: call %u failed (%d)\n", + call_id, ret); + return ret; +} + +static int sev_snp_vbs_vtl_call(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + return sev_snp_vbs_call(id, arg, arg_size, resp, resp_size); +} + +/* =E2=94=80=E2=94=80 memory protection =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +struct vbs_sev_snp_protect_args { + __u64 pfn; + __u64 nr_pages; + __u32 perms; +} __packed; + +static int sev_snp_vbs_protect_memory(unsigned long pfn, + unsigned long nr_pages, + unsigned int perms) +{ + struct vbs_sev_snp_protect_args args =3D { + .pfn =3D pfn, + .nr_pages =3D nr_pages, + .perms =3D perms, + }; + + return sev_snp_vbs_call(SEV_SNP_VBS_PROTECT_MEMORY, + &args, sizeof(args), NULL, 0); +} + +static int sev_snp_vbs_seal_kernel(void) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_SEAL_KERNEL, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +static int sev_snp_vbs_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) +{ + /* + * Module ELF may be large. Pass its physical address and size + * to VMPL0 so the SVSM can read it from the shared address space. + */ + return sev_snp_vbs_call(SEV_SNP_VBS_VALIDATE_MODULE, + NULL, 0, NULL, 0); +} + +static int sev_snp_vbs_set_module_perms(const struct module *mod) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_SET_MODULE_PERMS, + NULL, 0, NULL, 0); +} + +static int sev_snp_vbs_unload_module(const struct module *mod) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_UNLOAD_MODULE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 key / certificate management =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int sev_snp_vbs_add_key(const void *key, size_t key_size, + unsigned int flags) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_ADD_KEY, key, key_size, NULL, 0); +} + +static int sev_snp_vbs_revoke_key(const void *key_id, size_t id_size) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_REVOKE_KEY, key_id, id_size, NULL, 0); +} + +static int sev_snp_vbs_send_certs(const void *certs, size_t certs_size) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_SEND_CERTS, + certs, certs_size, NULL, 0); +} + +/* =E2=94=80=E2=94=80 kexec validation =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int sev_snp_vbs_kexec_validate(const void *kernel, size_t kernel_si= ze, + const void *sig, size_t sig_size) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_KEXEC_VALIDATE, NULL, 0, NULL, 0); +} + +static int sev_snp_vbs_kexec_invalidate(void) +{ + return sev_snp_vbs_call(SEV_SNP_VBS_KEXEC_INVALIDATE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 lifecycle =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int sev_snp_vbs_init(void) +{ + int ret; + + ret =3D sev_snp_vbs_call(SEV_SNP_VBS_INIT, NULL, 0, NULL, 0); + if (ret) { + pr_err("vbs-sev-snp: VBS init failed (%d)\n", ret); + return ret; + } + + pr_info("vbs-sev-snp: connected to SVSM at VMPL0\n"); + return 0; +} + +static void sev_snp_vbs_shutdown(void) +{ + sev_snp_vbs_call(SEV_SNP_VBS_SHUTDOWN, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 ops table & registration =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static const struct vbs_ops sev_snp_vbs_ops =3D { + .name =3D "sev-snp", + .init =3D sev_snp_vbs_init, + .shutdown =3D sev_snp_vbs_shutdown, + .vtl_call =3D sev_snp_vbs_vtl_call, + .protect_memory =3D sev_snp_vbs_protect_memory, + .seal_kernel =3D sev_snp_vbs_seal_kernel, + .validate_module =3D sev_snp_vbs_validate_module, + .set_module_perms =3D sev_snp_vbs_set_module_perms, + .unload_module =3D sev_snp_vbs_unload_module, + .add_key =3D sev_snp_vbs_add_key, + .revoke_key =3D sev_snp_vbs_revoke_key, + .send_certs =3D sev_snp_vbs_send_certs, + .kexec_validate =3D sev_snp_vbs_kexec_validate, + .kexec_invalidate =3D sev_snp_vbs_kexec_invalidate, +}; + +/* =E2=94=80=E2=94=80 detection & probe (called from probe.c) =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +bool __init vbs_sev_snp_detect(void) +{ + if (!cc_platform_has(CC_ATTR_GUEST_SEV_SNP)) + return false; + + if (snp_vmpl =3D=3D 0) { + pr_debug("vbs-sev-snp: running at VMPL0, no SVSM above us\n"); + return false; + } + + return true; +} + +const struct vbs_ops *vbs_sev_snp_get_ops(void) +{ + return &sev_snp_vbs_ops; +} diff --git a/security/vbs/tdx.c b/security/vbs/tdx.c new file mode 100644 index 000000000000..43e636e2b591 --- /dev/null +++ b/security/vbs/tdx.c @@ -0,0 +1,280 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS backend =E2=80=94 Intel TDX service TD + * + * Uses TDG.VP.VMCALL (TDVMCALL) to communicate between the main TD + * (plane-0) and a service TD (plane-1) that provides security services. + * + * Transport: TDVMCALL with a VBS-specific sub-function leaf. The VMM + * (QEMU / KVM) routes the call to the service TD, which + * shares memory with the main TD for request/response data. + * + * Note: Service TD support is still evolving in the TDX architecture. + * This backend provides the framework and will be updated as the + * inter-TD communication spec is finalised. + */ + +#include "internal.h" + +#include +#include + +#include +#include + +/* + * VBS-specific TDVMCALL sub-function. Chosen from the vendor-specific + * range (>=3D 0x10010000) to avoid conflicts with the GHCI-defined leaves. + */ +#define TDVMCALL_VBS 0x10010000ULL + +/* VBS sub-commands passed in R12 */ +#define TDX_VBS_INIT 0 +#define TDX_VBS_SHUTDOWN 1 +#define TDX_VBS_PROTECT_MEMORY 2 +#define TDX_VBS_SEAL_KERNEL 3 +#define TDX_VBS_VALIDATE_MODULE 4 +#define TDX_VBS_SET_MODULE_PERMS 5 +#define TDX_VBS_UNLOAD_MODULE 6 +#define TDX_VBS_ADD_KEY 7 +#define TDX_VBS_REVOKE_KEY 8 +#define TDX_VBS_SEND_CERTS 9 +#define TDX_VBS_KEXEC_VALIDATE 10 +#define TDX_VBS_KEXEC_INVALIDATE 11 + +/* =E2=94=80=E2=94=80 shared-memory buffers =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +/* + * Shared (decrypted) pages for passing request and response data between + * the main TD and the service TD. Marked shared via cc_mkdec() so the + * VMM and service TD can access them. + */ +static void *tdx_req_page; +static void *tdx_resp_page; + +/* =E2=94=80=E2=94=80 low-level VBS TDVMCALL =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80 */ + +/* + * Issue a VBS call to the service TD through the VMM. + * + * Register usage (TDVMCALL convention): + * R11 =3D sub-function leaf (TDVMCALL_VBS) + * R12 =3D VBS command ID + * R13 =3D physical address of request buffer (shared) + * R14 =3D physical address of response buffer (shared) + * R15 =3D request size + */ +static int tdx_vbs_call(u32 cmd, const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + struct tdx_module_args args =3D {}; + u64 ret; + + if (arg && arg_size) { + if (arg_size > PAGE_SIZE || !tdx_req_page) + return -E2BIG; + memcpy(tdx_req_page, arg, arg_size); + } + + args.r11 =3D TDVMCALL_VBS; + args.r12 =3D cmd; + args.r13 =3D tdx_req_page ? cc_mkdec(virt_to_phys(tdx_req_page)) : 0; + args.r14 =3D tdx_resp_page ? cc_mkdec(virt_to_phys(tdx_resp_page)) : 0; + args.r15 =3D arg_size; + + ret =3D __tdx_hypercall(&args); + if (ret) { + pr_err_ratelimited("vbs-tdx: TDVMCALL failed (0x%llx)\n", ret); + return -EIO; + } + + /* R10 holds the VMM return status */ + if (args.r10) { + pr_err_ratelimited("vbs-tdx: service TD returned 0x%llx\n", + args.r10); + return -EREMOTEIO; + } + + if (resp && resp_size && tdx_resp_page) { + size_t copy =3D min_t(size_t, resp_size, PAGE_SIZE); + + memcpy(resp, tdx_resp_page, copy); + } + return 0; +} + +static int tdx_vbs_vtl_call(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + return tdx_vbs_call(id, arg, arg_size, resp, resp_size); +} + +/* =E2=94=80=E2=94=80 memory protection =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +struct vbs_tdx_protect_args { + __u64 pfn; + __u64 nr_pages; + __u32 perms; +} __packed; + +static int tdx_vbs_protect_memory(unsigned long pfn, unsigned long nr_page= s, + unsigned int perms) +{ + struct vbs_tdx_protect_args args =3D { + .pfn =3D pfn, + .nr_pages =3D nr_pages, + .perms =3D perms, + }; + + return tdx_vbs_call(TDX_VBS_PROTECT_MEMORY, + &args, sizeof(args), NULL, 0); +} + +static int tdx_vbs_seal_kernel(void) +{ + return tdx_vbs_call(TDX_VBS_SEAL_KERNEL, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +static int tdx_vbs_validate_module(const void *elf, size_t elf_size, + const void *sig, size_t sig_size) +{ + return tdx_vbs_call(TDX_VBS_VALIDATE_MODULE, NULL, 0, NULL, 0); +} + +static int tdx_vbs_set_module_perms(const struct module *mod) +{ + return tdx_vbs_call(TDX_VBS_SET_MODULE_PERMS, NULL, 0, NULL, 0); +} + +static int tdx_vbs_unload_module(const struct module *mod) +{ + return tdx_vbs_call(TDX_VBS_UNLOAD_MODULE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 key / certificate management =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int tdx_vbs_add_key(const void *key, size_t key_size, + unsigned int flags) +{ + return tdx_vbs_call(TDX_VBS_ADD_KEY, key, key_size, NULL, 0); +} + +static int tdx_vbs_revoke_key(const void *key_id, size_t id_size) +{ + return tdx_vbs_call(TDX_VBS_REVOKE_KEY, key_id, id_size, NULL, 0); +} + +static int tdx_vbs_send_certs(const void *certs, size_t certs_size) +{ + return tdx_vbs_call(TDX_VBS_SEND_CERTS, certs, certs_size, NULL, 0); +} + +/* =E2=94=80=E2=94=80 kexec validation =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int tdx_vbs_kexec_validate(const void *kernel, size_t kernel_size, + const void *sig, size_t sig_size) +{ + return tdx_vbs_call(TDX_VBS_KEXEC_VALIDATE, NULL, 0, NULL, 0); +} + +static int tdx_vbs_kexec_invalidate(void) +{ + return tdx_vbs_call(TDX_VBS_KEXEC_INVALIDATE, NULL, 0, NULL, 0); +} + +/* =E2=94=80=E2=94=80 lifecycle =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ + +static int tdx_vbs_init(void) +{ + int ret; + + /* + * Allocate shared pages for inter-TD communication. These must + * be marked as shared (decrypted) so the service TD can read them. + */ + tdx_req_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + tdx_resp_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + if (!tdx_req_page || !tdx_resp_page) { + ret =3D -ENOMEM; + goto fail; + } + + /* + * Convert to shared pages. set_memory_decrypted() clears the + * encryption bit so the VMM / service TD can access these pages. + */ + ret =3D set_memory_decrypted((unsigned long)tdx_req_page, 1); + if (ret) + goto fail; + ret =3D set_memory_decrypted((unsigned long)tdx_resp_page, 1); + if (ret) + goto fail_re_encrypt_req; + + ret =3D tdx_vbs_call(TDX_VBS_INIT, NULL, 0, NULL, 0); + if (ret) { + pr_err("vbs-tdx: service TD init failed (%d)\n", ret); + goto fail_re_encrypt; + } + + pr_info("vbs-tdx: connected to TDX service TD\n"); + return 0; + +fail_re_encrypt: + set_memory_encrypted((unsigned long)tdx_resp_page, 1); +fail_re_encrypt_req: + set_memory_encrypted((unsigned long)tdx_req_page, 1); +fail: + free_page((unsigned long)tdx_req_page); + free_page((unsigned long)tdx_resp_page); + tdx_req_page =3D tdx_resp_page =3D NULL; + return ret; +} + +static void tdx_vbs_shutdown(void) +{ + tdx_vbs_call(TDX_VBS_SHUTDOWN, NULL, 0, NULL, 0); + + if (tdx_resp_page) { + set_memory_encrypted((unsigned long)tdx_resp_page, 1); + free_page((unsigned long)tdx_resp_page); + } + if (tdx_req_page) { + set_memory_encrypted((unsigned long)tdx_req_page, 1); + free_page((unsigned long)tdx_req_page); + } + tdx_req_page =3D tdx_resp_page =3D NULL; +} + +/* =E2=94=80=E2=94=80 ops table & registration =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static const struct vbs_ops tdx_vbs_ops =3D { + .name =3D "tdx-service-td", + .init =3D tdx_vbs_init, + .shutdown =3D tdx_vbs_shutdown, + .vtl_call =3D tdx_vbs_vtl_call, + .protect_memory =3D tdx_vbs_protect_memory, + .seal_kernel =3D tdx_vbs_seal_kernel, + .validate_module =3D tdx_vbs_validate_module, + .set_module_perms =3D tdx_vbs_set_module_perms, + .unload_module =3D tdx_vbs_unload_module, + .add_key =3D tdx_vbs_add_key, + .revoke_key =3D tdx_vbs_revoke_key, + .send_certs =3D tdx_vbs_send_certs, + .kexec_validate =3D tdx_vbs_kexec_validate, + .kexec_invalidate =3D tdx_vbs_kexec_invalidate, +}; + +/* =E2=94=80=E2=94=80 detection & probe (called from probe.c) =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ + +bool __init vbs_tdx_detect(void) +{ + return cc_platform_has(CC_ATTR_GUEST_TDX); +} + +const struct vbs_ops *vbs_tdx_get_ops(void) +{ + return &tdx_vbs_ops; +} --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 757A343A81A; Wed, 5 Aug 2026 11:03:55 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927836; cv=none; b=mucJldog9iaXyIeDHOoEPwca5BJFmODRcUxOQG2JD/ZUMaHZG6swjJ30pHf8QsLEVtGU6IU/9C71/4qMETpzREvlFFD86WxruGgykQ88S3T+Sb+ki1HxmKpDEp3aT+a5yH7mOOaN4pX7ThjBF+92Tfy/KXWBpl3/1Wwg7uP54Hw= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927836; c=relaxed/simple; bh=vam/t1N6IT97N17AD+wC9on+SHw5SQJ31wu2TPXglJM=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=aS2k0uHULwfCol7lQ93fAqBalM3JZIQyehwS7EZXx5WEXuFhG2gdsXgmqDC1Cxadwgzf0fzFDDJaQXzo+WMWkfFeYjQr0uSBPd7Y4b1alDB/6yu/njz0TUns3j5NY7PrS/4UQ6lelOJudY9yI2Qb642nFthis8Is6C3cE1HtUYs= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=hB9aaRmH; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="hB9aaRmH" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 81BBD20B7169; Wed, 5 Aug 2026 04:03:34 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 81BBD20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927814; bh=gTX5HFZBUVieVTMJ1ys2qCD3TJ8fG6Q7VQSgAosmpCo=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=hB9aaRmH7LyqbXuaUC3l7VARW2yXvcqkixuA8otsc69FAKM7d40o+wyokxSFC6WYp 96+V/xvfNnhy0QahZnfd17jIHdmrzDWjecl7NMr8SiJ5aH+aOtIrv3GZreOsQo4l/N SeeiIyABjOtNkEzahaZxgEmGtV7HtcfDJDXekA1g= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 13/42] Add a inter-plane communication mechanism through KVM. - model this to use a single page similar to SEV-SNP Date: Wed, 5 Aug 2026 04:02:55 -0700 Message-ID: <20260805110324.25067-14-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable --- security/vbs/kvm_planes.c | 102 +++++++++++++++++++------------------- 1 file changed, 51 insertions(+), 51 deletions(-) diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c index 3526f7c429c3..3eec3abb56ee 100644 --- a/security/vbs/kvm_planes.c +++ b/security/vbs/kvm_planes.c @@ -31,26 +31,30 @@ */ #define KVM_HC_VBS_VTL_CALL 15 =20 -/* =E2=94=80=E2=94=80 shared-memory request / response layout =E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80 */ +/* =E2=94=80=E2=94=80 shared-memory calling area (modelled after the SVSM = CAA) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ =20 -struct vbs_kvm_request { - __u32 call_id; /* enum vbs_call_id */ - __u32 arg_size; /* bytes of payload following this hdr */ - __u8 payload[]; /* variable-length argument data */ +/* + * Single shared page used for both request and response data. + * The protocol is synchronous: plane-0 writes the request, issues a + * hypercall, blocks until QEMU returns, then reads the response from + * the same page. No concurrent access is possible. + * + * Layout (within one 4 KiB page): + * [ call_pending | call_id | status | arg_size | resp_size | buffer ] + */ +struct vbs_kvm_ca { + __u8 call_pending; /* 1 while call is in flight */ + __u8 rsvd[3]; + __u32 call_id; /* enum vbs_call_id (set by caller) */ + __s32 status; /* return code (set by responder) */ + __u32 arg_size; /* request payload size */ + __u32 resp_size; /* response payload size */ + __u8 buffer[]; /* request data in, response data out */ } __packed; =20 -struct vbs_kvm_response { - __s32 status; /* 0 =3D success, negative errno */ - __u32 resp_size; /* bytes of payload following this hdr */ - __u8 payload[]; /* variable-length response data */ -} __packed; +#define VBS_CA_BUF_SIZE (PAGE_SIZE - sizeof(struct vbs_kvm_ca)) =20 -/* - * A single page is used for each direction. That gives ~4 KiB of - * payload per call, which is enough for all current VBS operations. - */ -static void *kvm_req_page; /* request (plane-0 writes, plane-1 reads) */ -static void *kvm_resp_page; /* response (plane-1 writes, plane-0 reads) */ +static void *kvm_ca_page; /* single calling-area page */ =20 /* =E2=94=80=E2=94=80 low-level VTL call =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80 */ =20 @@ -58,41 +62,44 @@ static int kvm_planes_vtl_call(enum vbs_call_id id, const void *arg, size_t arg_size, void *resp, size_t resp_size) { - struct vbs_kvm_request *req; - struct vbs_kvm_response *rsp; + struct vbs_kvm_ca *ca; long hc_ret; =20 - if (!kvm_req_page || !kvm_resp_page) + if (!kvm_ca_page) return -ENOMEM; =20 - if (arg_size > PAGE_SIZE - sizeof(*req)) + if (arg_size > VBS_CA_BUF_SIZE) return -E2BIG; =20 - /* Build request in the shared page */ - req =3D kvm_req_page; - req->call_id =3D id; - req->arg_size =3D arg_size; + ca =3D kvm_ca_page; + + /* Build request */ + ca->call_id =3D id; + ca->arg_size =3D arg_size; + ca->status =3D 0; + ca->resp_size =3D 0; if (arg_size && arg) - memcpy(req->payload, arg, arg_size); + memcpy(ca->buffer, arg, arg_size); + ca->call_pending =3D 1; + + /* Issue hypercall: pass physical address of the calling area */ + hc_ret =3D kvm_hypercall1(KVM_HC_VBS_VTL_CALL, + virt_to_phys(kvm_ca_page)); + ca->call_pending =3D 0; =20 - /* Issue hypercall: pass physical addresses of req & resp pages */ - hc_ret =3D kvm_hypercall2(KVM_HC_VBS_VTL_CALL, - virt_to_phys(kvm_req_page), - virt_to_phys(kvm_resp_page)); if (hc_ret) { pr_err_ratelimited("vbs-kvm: hypercall failed (%ld)\n", hc_ret); return -EIO; } =20 - /* Read response */ - rsp =3D kvm_resp_page; - if (rsp->status) - return rsp->status; + if (ca->status) + return ca->status; =20 - if (resp && resp_size) { - size_t copy =3D min_t(size_t, resp_size, rsp->resp_size); + /* Read response from the same buffer */ + if (resp && resp_size && ca->resp_size) { + size_t copy =3D min_t(size_t, resp_size, ca->resp_size); =20 - memcpy(resp, rsp->payload, copy); + memcpy(resp, ca->buffer, copy); } return 0; } @@ -192,34 +199,27 @@ static int kvm_planes_init(void) { int ret; =20 - kvm_req_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); - kvm_resp_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); - if (!kvm_req_page || !kvm_resp_page) { - ret =3D -ENOMEM; - goto fail; - } + kvm_ca_page =3D (void *)__get_free_page(GFP_KERNEL | __GFP_ZERO); + if (!kvm_ca_page) + return -ENOMEM; =20 ret =3D kvm_planes_vtl_call(VBS_CALL_INIT, NULL, 0, NULL, 0); if (ret) { pr_err("vbs-kvm: plane-1 INIT call failed (%d)\n", ret); - goto fail; + free_page((unsigned long)kvm_ca_page); + kvm_ca_page =3D NULL; + return ret; } =20 pr_info("vbs-kvm: connected to plane-1 secure kernel\n"); return 0; -fail: - free_page((unsigned long)kvm_req_page); - free_page((unsigned long)kvm_resp_page); - kvm_req_page =3D kvm_resp_page =3D NULL; - return ret; } =20 static void kvm_planes_shutdown(void) { kvm_planes_vtl_call(VBS_CALL_SHUTDOWN, NULL, 0, NULL, 0); - free_page((unsigned long)kvm_req_page); - free_page((unsigned long)kvm_resp_page); - kvm_req_page =3D kvm_resp_page =3D NULL; + free_page((unsigned long)kvm_ca_page); + kvm_ca_page =3D NULL; } =20 /* =E2=94=80=E2=94=80 ops table & registration =E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 2CBE943B3F6; Wed, 5 Aug 2026 11:03:56 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927838; cv=none; b=IS5LvE6DeGXnkV2SiDKTUpX4pmwBv9zv0isLWGvwHTxit5EJkFKBQPpdiocOQZfcVCz4IMhZDikWvSfrHVZT6YsyN4ZHVTNPNghHQCR3ojKTJ2/rsJFAjGNXEOZbF9JfuZgkItpg6icBeccxSjd15tmo3SV27/CpO7Fg5kdzWho= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927838; c=relaxed/simple; bh=fw/3m2gOCBD7tgMqLnrxTv3DXia3LDwXiPtZQbSdaRA=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=pQkmynk4PAx0PBlKS8aG3NY8O2sLadhETuiRcVZevz6ZYCdGyY+ZG8shAatV7ewNfeSgDC84HTUglapeOoQOJVaywgRrSw8bTbiG+kI7vjK0UHi0cgw9gzvgUwGx9SVNJ5BEzK90XWKz8LN1YV6YDeZEXUu+1BcwaJICAKGHT+M= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=Xos5SIf9; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="Xos5SIf9" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 23F4E20B716B; Wed, 5 Aug 2026 04:03:35 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 23F4E20B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927815; bh=TSibJLeuOC2oppqd3yILeAKE37LPNTC/8+9Siv/5++Q=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=Xos5SIf9TUNAuQYOC5PmtCgbbdWDojfZmqkibp8DwnZFSQpWxLBGoi9UEkfTDL11v egDWDVYsEl1eyrwILA/I3VmomXkKEbEK9jr/2uHtkJ9HxFneZ+Kw8m03VYAU4+GuxL CK5J90dvQQhZLj1ZdWuPUO8DWUCZ3MThM9x30Hhk= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 14/42] KVM: Add per-plane memory attribute support for cross-plane EPT protection Date: Wed, 5 Aug 2026 04:02:56 -0700 Message-ID: <20260805110324.25067-15-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Extend the KVM memory attributes framework to support per-plane R/W/X permission control, enabling a higher-privilege plane (e.g., plane-1 / secure kernel) to restrict a lower-privilege plane's (e.g., plane-0) EPT permissions. This is the KVM equivalent of the AMD SEV-SNP RMP (Reverse Map Table): each plane has its own mem_attr_array, and attributes like NO_WRITE and NO_EXEC are enforced by filtering pte_access bits during SPTE creation. Changes: - include/uapi/linux/kvm.h: Add KVM_MEMORY_ATTRIBUTE_NO_WRITE (bit 4) and KVM_MEMORY_ATTRIBUTE_NO_EXEC (bit 5). Add struct kvm_plane_memory_attributes and KVM_SET_PLANE_MEMORY_ATTRIBUTES ioctl (0xd6) for targeting a specific plane's address space. - virt/kvm/kvm_main.c: Extend kvm_supported_mem_attributes() to return NO_WRITE|NO_EXEC when CONFIG_KVM_MAX_NR_VCPU_PLANES is enabled. Add KVM_SET_PLANE_MEMORY_ATTRIBUTES ioctl handler that validates the target plane and delegates to the existing kvm_vm_ioctl_set_mem_attributes() infrastructure. - arch/x86/kvm/mmu/spte.h: Add kvm_plane_filter_pte_access() helper that reads the plane's mem_attr_array for a GFN and strips W/X from pte_access when NO_WRITE/NO_EXEC are set. - arch/x86/kvm/mmu/tdp_mmu.c, arch/x86/kvm/mmu/mmu.c: Wire kvm_plane_filter_pte_access() into both TDP and shadow MMU SPTE creation paths, filtering pte_access before calling make_spte(). --- arch/x86/kvm/mmu/mmu.c | 4 +++- arch/x86/kvm/mmu/spte.h | 33 +++++++++++++++++++++++++++++++++ arch/x86/kvm/mmu/tdp_mmu.c | 4 +++- include/uapi/linux/kvm.h | 23 +++++++++++++++++++++++ 4 files changed, 62 insertions(+), 2 deletions(-) diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c index a0d9a0a33c5f..3b861a42a712 100644 --- a/arch/x86/kvm/mmu/mmu.c +++ b/arch/x86/kvm/mmu/mmu.c @@ -3107,7 +3107,9 @@ static int mmu_set_spte(struct kvm_vcpu *vcpu, struct= kvm_memory_slot *slot, return RET_PF_EMULATE; } =20 - wrprot =3D make_spte(vcpu, sp, slot, pte_access, gfn, pfn, *sptep, prefet= ch, + wrprot =3D make_spte(vcpu, sp, slot, + kvm_plane_filter_pte_access(vcpu, gfn, pte_access), + gfn, pfn, *sptep, prefetch, false, host_writable, &spte); =20 if (*sptep =3D=3D spte) { diff --git a/arch/x86/kvm/mmu/spte.h b/arch/x86/kvm/mmu/spte.h index 13eea94dd212..421836fd3932 100644 --- a/arch/x86/kvm/mmu/spte.h +++ b/arch/x86/kvm/mmu/spte.h @@ -579,4 +579,37 @@ static inline u64 restore_acc_track_spte(u64 spte) void __init kvm_mmu_spte_module_init(void); void kvm_mmu_reset_all_pte_masks(void); =20 +/* + * Apply per-plane memory protection attributes to pte_access. + * If the plane's mem_attr_array has NO_WRITE or NO_EXEC set for a GFN, + * strip the corresponding access bits before building the SPTE. + */ +#ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES +static inline unsigned int kvm_plane_filter_pte_access(struct kvm_vcpu *vc= pu, + gfn_t gfn, + unsigned int pte_access) +{ + struct kvm_plane *plane =3D vcpu_to_plane(vcpu); + unsigned long attrs; + + if (!plane) + return pte_access; + + attrs =3D kvm_get_plane_memory_attributes(plane, gfn); + if (attrs & KVM_MEMORY_ATTRIBUTE_NO_WRITE) + pte_access &=3D ~ACC_WRITE_MASK; + if (attrs & KVM_MEMORY_ATTRIBUTE_NO_EXEC) + pte_access &=3D ~ACC_EXEC_MASK; + + return pte_access; +} +#else +static inline unsigned int kvm_plane_filter_pte_access(struct kvm_vcpu *vc= pu, + gfn_t gfn, + unsigned int pte_access) +{ + return pte_access; +} +#endif + #endif diff --git a/arch/x86/kvm/mmu/tdp_mmu.c b/arch/x86/kvm/mmu/tdp_mmu.c index 4503558211fd..0603445377aa 100644 --- a/arch/x86/kvm/mmu/tdp_mmu.c +++ b/arch/x86/kvm/mmu/tdp_mmu.c @@ -1140,7 +1140,9 @@ static int tdp_mmu_map_handle_target_level(struct kvm= _vcpu *vcpu, if (unlikely(!fault->slot)) new_spte =3D make_mmio_spte(vcpu, iter->gfn, sp->role.access); else - wrprot =3D make_spte(vcpu, sp, fault->slot, sp->role.access, iter->gfn, + wrprot =3D make_spte(vcpu, sp, fault->slot, + kvm_plane_filter_pte_access(vcpu, iter->gfn, sp->role.access), + iter->gfn, fault->pfn, iter->old_spte, fault->prefetch, false, fault->map_writable, &new_spte); =20 diff --git a/include/uapi/linux/kvm.h b/include/uapi/linux/kvm.h index de670bd836bf..82189353ef35 100644 --- a/include/uapi/linux/kvm.h +++ b/include/uapi/linux/kvm.h @@ -1687,6 +1687,29 @@ struct kvm_memory_attributes { =20 #define KVM_MEMORY_ATTRIBUTE_PRIVATE (1ULL << 3) =20 +/* + * Per-plane memory protection attributes (VM planes / VBS). + * These control EPT R/W/X permissions enforced by the hypervisor on + * behalf of a higher-privilege plane (e.g., plane-1 restricting plane-0). + */ +#define KVM_MEMORY_ATTRIBUTE_NO_WRITE (1ULL << 4) +#define KVM_MEMORY_ATTRIBUTE_NO_EXEC (1ULL << 5) + +/* + * Set memory attributes on a specific plane's address space. + * Used by a higher-privilege plane to restrict a lower-privilege plane's + * EPT permissions (e.g., plane-1 making plane-0 kernel text read-only). + */ +struct kvm_plane_memory_attributes { + __u32 plane; /* target plane index */ + __u32 flags; /* must be 0 */ + __u64 address; /* GPA (page-aligned) */ + __u64 size; /* size in bytes (page-aligned) */ + __u64 attributes; /* KVM_MEMORY_ATTRIBUTE_NO_WRITE / NO_EXEC */ +}; + +#define KVM_SET_PLANE_MEMORY_ATTRIBUTES _IOW(KVMIO, 0xd6, struct kvm_plane= _memory_attributes) + #define KVM_CREATE_GUEST_MEMFD _IOWR(KVMIO, 0xd4, struct kvm_create_guest= _memfd) #define GUEST_MEMFD_FLAG_MMAP (1ULL << 0) #define GUEST_MEMFD_FLAG_INIT_SHARED (1ULL << 1) --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id A54404334D3; Wed, 5 Aug 2026 11:03:56 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927838; cv=none; b=tL5OJRiwON3ps9XW90nTWqAmhB996hQ8CHMkd3bn4pktVrDBQ3QVJWeYPHe4LY0erDgkAyBg1/a+//kWnTXh//exczPSoQXhs9I8+8jyjtRWwYlTtdU4lo4nrPlMx5k6VEG5g7/yRKi0K4LMTNLPLWAw0crsesACbSp7vY+ct5Y= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927838; c=relaxed/simple; bh=cWUXBZZ/0t1QBYaf1rn4JIUxtpfPjOf2Zk+P9DLSsb4=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=szfOHU46VEBz1XmLVJSRME5TQbCd22Auck0aKzaLsTI+jYAy7DYV4HSWFJxgyCgPmN3+OYTpcXxwQfe8R6mO59lw14zcaAWnJXJMkf6RLFT3KekiebCY10qpgSzDVobxwmyttj7C3kMG9/WDxIflYAPLIzwQZgf9QBiHE4KehmI= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=J1Dmf0S5; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="J1Dmf0S5" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id CDE8F20B716C; Wed, 5 Aug 2026 04:03:35 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com CDE8F20B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927815; bh=ggyXrQg2nON7P/s3Q3W09ZXdgsDVh5eVj4ANYdmwBr8=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=J1Dmf0S55BuZZ4flbJsRtWWCiK/ZCX7sqLRBrHtxMaU0WkJxc1w0Xxz3zFeX/NfRq Y4WARdfwCfLwjo8Skuh5x/Wn9x+EEoAOrar8akVDV99rSKDVYbrFCeGNnHSpLxhdi9 9kagAwn3vp1ktKfgZBQ6sUnlOVzOvvrq+fdVXTgw= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 15/42] KVM: x86: Add KVM_HC_VBS_VTL_CALL hypercall for VBS inter-plane calls Date: Wed, 5 Aug 2026 04:02:57 -0700 Message-ID: <20260805110324.25067-16-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Define KVM_HC_VBS_VTL_CALL (hypercall 15) in the UAPI header and wire it into the KVM x86 hypercall exit path so it reaches QEMU userspace. This hypercall is used by the plane-0 guest VBS subsystem to issue synchronous calls to the plane-1 secure kernel via a shared calling-area (CAA) page, following the same pattern as KVM_HC_VM_PLANES_CONFIG/ ACTIVATE. Changes: - include/uapi/linux/kvm_para.h: Define KVM_HC_VBS_VTL_CALL =3D 15 - arch/x86/kvm/x86.c: Add to KVM_EXIT_HYPERCALL_VALID_MASK and to the userspace-exit case in ____kvm_emulate_hypercall() - security/vbs/kvm_planes.c: Remove local #define of KVM_HC_VBS_VTL_CALL, add #include to pick up the UAPI definition --- arch/x86/kvm/x86.c | 6 ++++-- include/uapi/linux/kvm_para.h | 1 + security/vbs/kvm_planes.c | 8 +------- 3 files changed, 6 insertions(+), 9 deletions(-) diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index b7256f155bea..4b99016fe536 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -121,7 +121,8 @@ static u64 __read_mostly efer_reserved_bits =3D ~((u64)= EFER_SCE); =20 #define KVM_EXIT_HYPERCALL_VALID_MASK (BIT(KVM_HC_MAP_GPA_RANGE) | \ BIT(KVM_HC_VM_PLANES_CONFIG) | \ - BIT(KVM_HC_VM_PLANES_ACTIVATE)) + BIT(KVM_HC_VM_PLANES_ACTIVATE) | \ + BIT(KVM_HC_VBS_VTL_CALL)) =20 #define KVM_CAP_PMU_VALID_MASK KVM_PMU_CAP_DISABLE =20 @@ -10533,7 +10534,8 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vcpu= , int cpl, return 0; } case KVM_HC_VM_PLANES_CONFIG: - case KVM_HC_VM_PLANES_ACTIVATE: { + case KVM_HC_VM_PLANES_ACTIVATE: + case KVM_HC_VBS_VTL_CALL: { ret =3D -KVM_ENOSYS; if (!user_exit_on_hypercall(vcpu->kvm, nr)) break; diff --git a/include/uapi/linux/kvm_para.h b/include/uapi/linux/kvm_para.h index 1b097f7ed937..1703238952fb 100644 --- a/include/uapi/linux/kvm_para.h +++ b/include/uapi/linux/kvm_para.h @@ -32,6 +32,7 @@ #define KVM_HC_MAP_GPA_RANGE 12 #define KVM_HC_VM_PLANES_CONFIG 13 #define KVM_HC_VM_PLANES_ACTIVATE 14 +#define KVM_HC_VBS_VTL_CALL 15 =20 /* * hypercalls use architecture specific diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c index 3eec3abb56ee..07a004712e9f 100644 --- a/security/vbs/kvm_planes.c +++ b/security/vbs/kvm_planes.c @@ -22,15 +22,9 @@ #include #include #include +#include #include =20 -/* =E2=94=80=E2=94=80 hypercall numbers for VBS VTL calls (plane-0 =E2=86= =92 plane-1) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ -/* - * These extend the existing KVM_HC_* numbering. The host (KVM + QEMU) - * intercepts them and routes them to the secure-kernel plane. - */ -#define KVM_HC_VBS_VTL_CALL 15 - /* =E2=94=80=E2=94=80 shared-memory calling area (modelled after the SVSM = CAA) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ =20 /* --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id E9E42430CE4; Wed, 5 Aug 2026 11:03:57 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927840; cv=none; b=HiFsdx9TS1gTcRLsCKOFuv9AvUBY8Fi49pg3RmtVWxtKhq9FfIit6gCFAYQeqAr1Lset+AMip5hJbZ15Z152P4yMIHnw6UP9S3nbK1FHSvavXt0dULJMfkLczGN470p/PTepqE0AgQKtl8QBtSKFDAbpBbJbCR2+/8ezN4vpJms= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927840; c=relaxed/simple; bh=UmUWVtSuVwvlzujYbsuuA85G+zuIQuxu2AqHP9g/B4E=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=kwwvCON8VFHI+nyb+SHLH7WL0j8Zt1JjdQep0NsyQJOykLjc5vFeDlPpPrSRX7RWBoCcbJymrKBR/CXwJXybon1BXQHJmBehsQHAW47PprMwl0ly979uf5wBuGi1971Lo9jCtgUQhqybw3iX+InU53gnKnyeAMB+VZ62XTDx+9Q= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=LQG66M5N; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="LQG66M5N" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 701D020B716D; Wed, 5 Aug 2026 04:03:36 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 701D020B716D DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927816; bh=qS0R4suTYYgDZhfRy1IzPe8WLVp7bZFCgZd3OyOa0lw=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=LQG66M5Nzg/62vEtn9XJy+wAxSsB7Ds2It5CipJ51S1UapXxoYY1OnGdwlVvkH6Vl TMRcBlBlWNiOtiQwOQ7inOE12vxeHpzA0zoBzMI3LlNlCDXUjprfiXO4EMHK3P+aTj 3xA7WBfAkzuu1gCChGCwqUTVzZx8gkRN/THW+aOI= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 16/42] vbs: Add HEKI kernel sealing and fix KVM plane memory attribute guards Date: Wed, 5 Aug 2026 04:02:58 -0700 Message-ID: <20260805110324.25067-17-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Implement Hypervisor-Enforced Kernel Integrity (HEKI) =E2=80=94 the plane-0 guest kernel automatically seals its text and rodata sections at late_initcall time by sending their GPAs to QEMU via the VBS VTL call mechanism. Guest-side changes: - security/vbs/core.c: Add vbs_heki_late_init() as a late_initcall that calls ops->init() to set up the VBS backend (allocate the shared CAA page, send VBS_CALL_INIT), then calls vbs_seal_kernel() to request kernel text/rodata protection. - security/vbs/kvm_planes.c: Implement kvm_planes_seal_kernel() to build a vbs_seal_kernel_req with page-aligned text/rodata GPAs and CR3, sent via VBS_CALL_SEAL_KERNEL to QEMU. - security/vbs/heki.h (new): Shared HEKI data structures (vbs_seal_kernel_req, vbs_protect_memory_req) and x86-64 page table walker callback interface. - security/vbs/heki.c (new): x86-64 4-level page table walker for plane-1 auditing of plane-0 mappings. Classifies pages as TEXT/RODATA/DATA_RW/DATA_RX based on PTE permission bits. - security/vbs/Kconfig: Add CONFIG_VBS_HEKI option. - security/vbs/Makefile: Build heki.o when CONFIG_VBS_HEKI=3Dy. Host-side fix: - virt/kvm/kvm_main.c: Replace CONFIG_KVM_MAX_NR_VCPU_PLANES (which had no Kconfig definition and was never set) with CONFIG_VM_PLANES in the three #ifdef guards protecting KVM_SET_PLANE_MEMORY_ATTRIBUTES ioctl and NO_WRITE/NO_EXEC attribute support. Without this fix the ioctl returned -ENOTTY. Signed-off-by: Sriram Nambakam --- security/vbs/Kconfig | 14 ++ security/vbs/Makefile | 1 + security/vbs/core.c | 32 +++++ security/vbs/heki.c | 287 ++++++++++++++++++++++++++++++++++++++ security/vbs/heki.h | 83 +++++++++++ security/vbs/kvm_planes.c | 19 ++- 6 files changed, 435 insertions(+), 1 deletion(-) create mode 100644 security/vbs/heki.c create mode 100644 security/vbs/heki.h diff --git a/security/vbs/Kconfig b/security/vbs/Kconfig index 3d9fb104b1fc..0fdbfbc7a795 100644 --- a/security/vbs/Kconfig +++ b/security/vbs/Kconfig @@ -15,6 +15,20 @@ config VBS =20 If unsure, say N. =20 +config VBS_HEKI + bool "HEKI: Hypervisor-Enforced Kernel Integrity" + depends on VBS && X86_64 + help + Enable the HEKI subsystem which provides: + - x86-64 page table walker for auditing guest kernel mappings + - Kernel seal support (make kernel text/rodata immutable via + EPT permission enforcement) + + This code runs in plane-1 (secure kernel) to inspect and + protect plane-0's address space. + + If unsure, say N. + config VBS_KVM_PLANES bool "VBS backend: KVM software planes" depends on VBS && KVM_GUEST diff --git a/security/vbs/Makefile b/security/vbs/Makefile index 4f0f26ef4f71..e33052ccde2d 100644 --- a/security/vbs/Makefile +++ b/security/vbs/Makefile @@ -2,6 +2,7 @@ obj-$(CONFIG_VBS) +=3D vbs.o vbs-y :=3D core.o probe.o =20 +vbs-$(CONFIG_VBS_HEKI) +=3D heki.o obj-$(CONFIG_VBS_KVM_PLANES) +=3D kvm_planes.o obj-$(CONFIG_VBS_SEV_SNP) +=3D sev_snp.o obj-$(CONFIG_VBS_TDX) +=3D tdx.o diff --git a/security/vbs/core.c b/security/vbs/core.c index 352590d88136..16b5329964f9 100644 --- a/security/vbs/core.c +++ b/security/vbs/core.c @@ -164,3 +164,35 @@ int vbs_kexec_invalidate(void) return ops->kexec_invalidate(); } EXPORT_SYMBOL_GPL(vbs_kexec_invalidate); + +/* =E2=94=80=E2=94=80 HEKI: automatic kernel sealing at late init =E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int __init vbs_heki_late_init(void) +{ + const struct vbs_ops *ops =3D READ_ONCE(vbs_backend); + int ret; + + if (!ops) { + pr_debug("vbs: HEKI: no backend, skipping kernel seal\n"); + return 0; + } + + /* Initialize the backend (allocates shared memory, etc.) */ + if (ops->init) { + ret =3D ops->init(); + if (ret) { + pr_warn("vbs: HEKI: backend init failed (%d)\n", ret); + return 0; + } + } + + pr_info("vbs: HEKI: sealing kernel text and rodata\n"); + ret =3D vbs_seal_kernel(); + if (ret) + pr_warn("vbs: HEKI: seal_kernel failed (%d)\n", ret); + else + pr_info("vbs: HEKI: kernel sealed successfully\n"); + + return 0; +} +late_initcall(vbs_heki_late_init); diff --git a/security/vbs/heki.c b/security/vbs/heki.c new file mode 100644 index 000000000000..8b4c4e3b170b --- /dev/null +++ b/security/vbs/heki.c @@ -0,0 +1,287 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * HEKI =E2=80=94 Hypervisor-Enforced Kernel Integrity + * + * x86-64 page table walker and kernel protection logic. + * + * The page table walker is designed to be called from plane-1 (the secure + * kernel) to audit plane-0's page tables. It is parameterised with a + * read_gpa() callback so it can work both in-kernel (for plane-1 with + * direct GPA access) and from QEMU (future, for host-side auditing). + * + * The seal_kernel helper runs in plane-0 and sends the kernel text/rodata + * GPA ranges to the secure side via the VBS VTL call mechanism. + */ + +#include "heki.h" +#include "internal.h" + +#include +#include +#include + +#ifdef CONFIG_X86_64 +#include + +/* =E2=94=80=E2=94=80 x86-64 page table constants =E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +#define PT_ENTRIES 512 +#define PT_ENTRY_SIZE 8 + +/* PTE bit positions */ +#define PTE_PRESENT BIT_ULL(0) +#define PTE_WRITABLE BIT_ULL(1) +#define PTE_USER BIT_ULL(2) +#define PTE_PS BIT_ULL(7) /* page size (huge page) */ +#define PTE_NX BIT_ULL(63) /* no-execute */ + +/* Physical address mask for 4-level paging (bits 12..51) */ +#define PTE_ADDR_MASK 0x000FFFFFFFFFF000ULL + +/* Page sizes */ +#define PAGE_SIZE_4K (1UL << 12) +#define PAGE_SIZE_2M (1UL << 21) +#define PAGE_SIZE_1G (1UL << 30) + +/* Virtual address extraction helpers */ +static inline unsigned int pml4_index(unsigned long va) +{ + return (va >> 39) & 0x1FF; +} + +static inline unsigned int pdpt_index(unsigned long va) +{ + return (va >> 30) & 0x1FF; +} + +static inline unsigned int pd_index(unsigned long va) +{ + return (va >> 21) & 0x1FF; +} + +static inline unsigned int pt_index(unsigned long va) +{ + return (va >> 12) & 0x1FF; +} + +/* + * Classify a page based on its PTE permission bits. + */ +static enum heki_page_class classify_pte(u64 pte) +{ + bool writable =3D !!(pte & PTE_WRITABLE); + bool executable =3D !(pte & PTE_NX); + + if (executable && !writable) + return HEKI_PAGE_TEXT; + if (!executable && !writable) + return HEKI_PAGE_RODATA; + if (!executable && writable) + return HEKI_PAGE_DATA_RW; + /* executable + writable =E2=80=94 W^X violation */ + return HEKI_PAGE_DATA_RX; +} + +/* + * Read a single page table entry from guest physical memory. + */ +static int read_pte(u64 table_gpa, unsigned int index, + int (*read_gpa)(u64, void *, size_t, void *), + void *ctx, u64 *pte_out) +{ + u64 entry_gpa =3D table_gpa + (u64)index * PT_ENTRY_SIZE; + + return read_gpa(entry_gpa, pte_out, sizeof(*pte_out), ctx); +} + +/* + * Walk a page table (PT) level =E2=80=94 4K pages. + */ +static int walk_pt(u64 pt_gpa, unsigned long va_base, + int (*read_gpa)(u64, void *, size_t, void *), void *ctx, + unsigned long va_start, unsigned long va_end, + heki_walk_cb cb, void *priv) +{ + unsigned int start_idx, end_idx, i; + int ret; + + start_idx =3D (va_start > va_base) ? pt_index(va_start) : 0; + end_idx =3D (va_end && va_end < va_base + PT_ENTRIES * PAGE_SIZE_4K) + ? pt_index(va_end - 1) : PT_ENTRIES - 1; + + for (i =3D start_idx; i <=3D end_idx; i++) { + u64 pte; + unsigned long va =3D va_base + (unsigned long)i * PAGE_SIZE_4K; + + ret =3D read_pte(pt_gpa, i, read_gpa, ctx, &pte); + if (ret) + return ret; + if (!(pte & PTE_PRESENT)) + continue; + + ret =3D cb(va, pte & PTE_ADDR_MASK, PAGE_SIZE_4K, + classify_pte(pte), priv); + if (ret) + return ret; + } + return 0; +} + +/* + * Walk a page directory (PD) level =E2=80=94 2M huge pages or recurse int= o PT. + */ +static int walk_pd(u64 pd_gpa, unsigned long va_base, + int (*read_gpa)(u64, void *, size_t, void *), void *ctx, + unsigned long va_start, unsigned long va_end, + heki_walk_cb cb, void *priv) +{ + unsigned int start_idx, end_idx, i; + int ret; + + start_idx =3D (va_start > va_base) ? pd_index(va_start) : 0; + end_idx =3D (va_end && va_end < va_base + (unsigned long)PT_ENTRIES * P= AGE_SIZE_2M) + ? pd_index(va_end - 1) : PT_ENTRIES - 1; + + for (i =3D start_idx; i <=3D end_idx; i++) { + u64 pde; + unsigned long va =3D va_base + (unsigned long)i * PAGE_SIZE_2M; + + ret =3D read_pte(pd_gpa, i, read_gpa, ctx, &pde); + if (ret) + return ret; + if (!(pde & PTE_PRESENT)) + continue; + + if (pde & PTE_PS) { + /* 2M huge page */ + ret =3D cb(va, pde & PTE_ADDR_MASK, PAGE_SIZE_2M, + classify_pte(pde), priv); + if (ret) + return ret; + } else { + ret =3D walk_pt(pde & PTE_ADDR_MASK, va, + read_gpa, ctx, va_start, va_end, + cb, priv); + if (ret) + return ret; + } + } + return 0; +} + +/* + * Walk a page directory pointer table (PDPT) =E2=80=94 1G huge pages or r= ecurse. + */ +static int walk_pdpt(u64 pdpt_gpa, unsigned long va_base, + int (*read_gpa)(u64, void *, size_t, void *), void *ctx, + unsigned long va_start, unsigned long va_end, + heki_walk_cb cb, void *priv) +{ + unsigned int start_idx, end_idx, i; + int ret; + + start_idx =3D (va_start > va_base) ? pdpt_index(va_start) : 0; + end_idx =3D (va_end && va_end < va_base + (unsigned long)PT_ENTRIES * P= AGE_SIZE_1G) + ? pdpt_index(va_end - 1) : PT_ENTRIES - 1; + + for (i =3D start_idx; i <=3D end_idx; i++) { + u64 pdpte; + unsigned long va =3D va_base + (unsigned long)i * PAGE_SIZE_1G; + + ret =3D read_pte(pdpt_gpa, i, read_gpa, ctx, &pdpte); + if (ret) + return ret; + if (!(pdpte & PTE_PRESENT)) + continue; + + if (pdpte & PTE_PS) { + /* 1G huge page */ + ret =3D cb(va, pdpte & PTE_ADDR_MASK, PAGE_SIZE_1G, + classify_pte(pdpte), priv); + if (ret) + return ret; + } else { + ret =3D walk_pd(pdpte & PTE_ADDR_MASK, va, + read_gpa, ctx, va_start, va_end, + cb, priv); + if (ret) + return ret; + } + } + return 0; +} + +/** + * heki_walk_x86_tables - walk x86-64 4-level page tables + * @cr3: value of CR3 (page table root physical address) + * @read_gpa: callback to read bytes from a guest physical address + * @read_ctx: opaque context passed to read_gpa + * @va_start: start of virtual address range (0 =3D from beginning) + * @va_end: end of virtual address range (0 =3D to end) + * @cb: callback invoked for each present page + * @priv: opaque context passed to cb + * + * Walks the full PML4 =E2=86=92 PDPT =E2=86=92 PD =E2=86=92 PT hierarchy,= invoking @cb for + * every present page (4K, 2M, or 1G) within [va_start, va_end). + * + * Returns 0 on success, or the first non-zero return from @cb / @read_gpa. + */ +int heki_walk_x86_tables(unsigned long cr3, + int (*read_gpa)(u64 gpa, void *buf, size_t len, + void *ctx), + void *read_ctx, + unsigned long va_start, unsigned long va_end, + heki_walk_cb cb, void *priv) +{ + u64 pml4_gpa =3D cr3 & PTE_ADDR_MASK; + unsigned int i; + int ret; + + if (!read_gpa || !cb) + return -EINVAL; + + /* + * Walk PML4 entries. Each PML4 entry covers 512 GB. + * For the kernel half of the address space on x86-64, + * entries 256..511 map the kernel virtual addresses + * (0xffff800000000000 and above). + */ + for (i =3D 0; i < PT_ENTRIES; i++) { + u64 pml4e; + /* Each PML4 entry covers 512 GiB */ + unsigned long va_base =3D (unsigned long)i << 39; + + /* + * Sign-extend for canonical addresses: entries 256..511 + * map the upper half (kernel space). + */ + if (i >=3D 256) + va_base |=3D 0xFFFF000000000000UL; + + /* Skip entries outside the requested range */ + if (va_end && va_base >=3D va_end) + break; + if (va_start) { + unsigned long entry_end =3D va_base + + (1UL << 39) - 1; + if (entry_end < va_start) + continue; + } + + ret =3D read_pte(pml4_gpa, i, read_gpa, read_ctx, &pml4e); + if (ret) + return ret; + if (!(pml4e & PTE_PRESENT)) + continue; + + ret =3D walk_pdpt(pml4e & PTE_ADDR_MASK, va_base, + read_gpa, read_ctx, va_start, va_end, + cb, priv); + if (ret) + return ret; + } + + return 0; +} + +#endif /* CONFIG_X86_64 */ diff --git a/security/vbs/heki.h b/security/vbs/heki.h new file mode 100644 index 000000000000..fee986de351a --- /dev/null +++ b/security/vbs/heki.h @@ -0,0 +1,83 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * HEKI =E2=80=94 Hypervisor-Enforced Kernel Integrity + * + * Shared data structures between the guest kernel (plane-0) and the + * VBS secure kernel / QEMU dispatcher. These structs are placed in + * the VBS CAA page buffer and must be kept in sync with the QEMU-side + * definitions. + */ +#ifndef _VBS_HEKI_H +#define _VBS_HEKI_H + +#include + +/* + * VBS_CALL_PROTECT_MEMORY payload =E2=80=94 request EPT permission change= s on + * a contiguous GPA range from the perspective of the calling plane. + */ +struct vbs_protect_memory_req { + __u64 gpa; /* guest-physical address (page-aligned) */ + __u64 size; /* region size in bytes (page-aligned) */ + __u32 perms; /* desired permissions: VBS_MEM_* flags */ + __u32 flags; /* reserved, must be 0 */ +} __packed; + +/* + * VBS_CALL_SEAL_KERNEL payload =E2=80=94 plane-0 sends the GPAs of its ke= rnel + * text and rodata sections so that the secure side can make them + * immutable (NO_WRITE in the lower plane's EPT). + */ +struct vbs_seal_kernel_req { + __u64 text_gpa; /* _stext physical address */ + __u64 text_size; /* _etext - _stext */ + __u64 rodata_gpa; /* __start_rodata physical address */ + __u64 rodata_size; /* __end_rodata - __start_rodata */ + __u64 cr3; /* plane-0 kernel CR3 for verification */ +} __packed; + +/* =E2=94=80=E2=94=80 x86-64 page table walker (for plane-1 auditing) =E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80 */ + +/* Classification of a guest-physical page based on page table walk */ +enum heki_page_class { + HEKI_PAGE_UNMAPPED =3D 0, + HEKI_PAGE_TEXT =3D 1, /* executable, read-only (kernel text) */ + HEKI_PAGE_RODATA =3D 2, /* non-executable, read-only */ + HEKI_PAGE_DATA_RW =3D 3, /* non-executable, read-write */ + HEKI_PAGE_DATA_RX =3D 4, /* executable, read-write (DANGEROUS) */ +}; + +/* + * Callback invoked for each mapped page during a page table walk. + * @va: virtual address of the page + * @pa: guest-physical address of the page + * @size: page size (4K, 2M, or 1G) + * @pclass: classification based on PTE permission bits + * @priv: opaque context from the caller + * + * Return 0 to continue walking, non-zero to stop. + */ +typedef int (*heki_walk_cb)(unsigned long va, unsigned long pa, + unsigned long size, enum heki_page_class pclass, + void *priv); + +#ifdef CONFIG_X86_64 +/* + * Walk x86-64 4-level page tables starting from @cr3. + * @read_gpa: function to read @len bytes from guest physical address @gpa + * into @buf. Returns 0 on success. + * @va_start, @va_end: virtual address range to walk (0 for full walk) + * @cb: callback invoked for each mapped page + * @priv: opaque context passed to the callback + * + * Returns 0 on success, negative errno on failure. + */ +int heki_walk_x86_tables(unsigned long cr3, + int (*read_gpa)(u64 gpa, void *buf, size_t len, + void *ctx), + void *read_ctx, + unsigned long va_start, unsigned long va_end, + heki_walk_cb cb, void *priv); +#endif /* CONFIG_X86_64 */ + +#endif /* _VBS_HEKI_H */ diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c index 07a004712e9f..293c960c0968 100644 --- a/security/vbs/kvm_planes.c +++ b/security/vbs/kvm_planes.c @@ -23,7 +23,11 @@ #include #include #include +#include #include +#include + +#include "heki.h" =20 /* =E2=94=80=E2=94=80 shared-memory calling area (modelled after the SVSM = CAA) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ =20 @@ -122,7 +126,20 @@ static int kvm_planes_protect_memory(unsigned long pfn, =20 static int kvm_planes_seal_kernel(void) { - return kvm_planes_vtl_call(VBS_CALL_SEAL_KERNEL, NULL, 0, NULL, 0); + struct vbs_seal_kernel_req req =3D { + .text_gpa =3D __pa_symbol(_stext), + .text_size =3D PAGE_ALIGN((u64)(_etext - _stext)), + .rodata_gpa =3D __pa_symbol(__start_rodata), + .rodata_size =3D PAGE_ALIGN((u64)(__end_rodata - __start_rodata)), + .cr3 =3D read_cr3_pa(), + }; + + pr_info("vbs-kvm: seal_kernel text=3D[0x%llx+0x%llx] rodata=3D[0x%llx+0x%= llx] cr3=3D0x%llx\n", + req.text_gpa, req.text_size, + req.rodata_gpa, req.rodata_size, req.cr3); + + return kvm_planes_vtl_call(VBS_CALL_SEAL_KERNEL, + &req, sizeof(req), NULL, 0); } =20 /* =E2=94=80=E2=94=80 module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id C99F943CEC7; Wed, 5 Aug 2026 11:03:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927840; cv=none; b=p4sGIgTkyCGT7Fh2BJ+6vddNf/Q9FpehJ707TfqYWzDN+5nfZAS+wmdYdcAzu/qgy0P5CVa8aRKqR3Rdh7ehXyOo3fR5tkG8yAH1MliM8CDj6eSk+Deaj0OuT3ZNOM+FZyDBahVUjTgrekDA3GGoQ6g6+7DkWjJcx7nW0hPOytI= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927840; c=relaxed/simple; bh=jzdjyo1cSHBrg+vBvTCJ+uT+giL04yTaNxS8M7ZI0Ho=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=kNej4EpuUx+VH7+KLKy1Yp9Oal53A5xOS9HLZRWjlH5X7ed2kSywebuqtv+gxVatVjzPA5mA6xafL6bF+CsY3zDvZ1zosRR7qdfF1yQzuSgnevfB691XdEsN5JH5cN0yTtibaNoQHclJcCctm6gzvrUcgyQDjyxkpw3tyr2g/HE= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=KClYnQBQ; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="KClYnQBQ" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 906A420B7169; Wed, 5 Aug 2026 04:03:37 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 906A420B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927817; bh=duFM8Vmsa5KFjEUjTCphJphVogV9Eyn1mCYidCnYXik=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=KClYnQBQrHA2HwzBUerI61Mej1MeEKx0hNOr1mQAJiiZFBNftf0tcKXG6a5/63j/f j50s1nIWIvVJBOOw5hJYzL5fiDNLjDlcD4WRHR9ovuRXu4VVHkJ6oHbsxD8rS9+2ZV RKjBnlLu88o+yJy00rXFVD1kByAt/Rlq5jEQ5Qg0= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 17/42] vbs: Add module authentication via VBS/HEKI Date: Wed, 5 Aug 2026 04:02:59 -0700 Message-ID: <20260805110324.25067-18-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Hook the kernel module loader to send module validation requests to the secure kernel (plane-1 / QEMU) before allowing modules to load, and to set per-section EPT permissions after module formation. kernel/module/main.c: - After add_unformed_module(): call vbs_validate_module() with the module ELF blob GPA and the kernel's own sig_ok result from module_sig_check(). If the secure side rejects, loading is aborted. - After complete_formation(): call vbs_set_module_perms() to apply EPT permissions per section (text=3DR+X, rodata=3DR, data=3DR+W). Failure is non-fatal to avoid breaking module loading on ioctl errors. - In free_module(): call vbs_unload_module() so the secure side can release EPT overrides for the freed module. - All hooks are guarded by vbs_available() and are no-ops when VBS is not active. security/vbs/heki.h: - Add vbs_validate_module_req with module name, ELF GPA/size, and sig_ok flag (kernel's signature verification result). - Add vbs_module_section and vbs_set_module_perms_req for per-section GPA + permissions. - Add vbs_unload_module_req for module unload notification. security/vbs/kvm_planes.c: - Implement kvm_planes_validate_module(): converts vmalloc ELF pointer to GPA, sends sig_ok flag via VBS_CALL_VALIDATE_MODULE. - Implement kvm_planes_set_module_perms(): iterates mod->mem[] array, maps each section type to VBS_MEM_* permissions (TEXT=E2=86=92R+X, RODATA=E2=86=92R, DATA=E2=86=92R+W), sends via VBS_CALL_SET_MODULE_PERMS. - Implement kvm_planes_unload_module(): sends module name via VBS_CALL_UNLOAD_MODULE. Signed-off-by: Sriram Nambakam --- kernel/module/main.c | 37 ++++++++++++++ security/vbs/heki.h | 46 +++++++++++++++++ security/vbs/kvm_planes.c | 102 +++++++++++++++++++++++++++++++++++--- 3 files changed, 178 insertions(+), 7 deletions(-) diff --git a/kernel/module/main.c b/kernel/module/main.c index 46dd8d25a605..2d0232fccf18 100644 --- a/kernel/module/main.c +++ b/kernel/module/main.c @@ -39,6 +39,7 @@ #include #include #include +#include #include #include #include @@ -1418,6 +1419,11 @@ static void free_module(struct module *mod) { trace_module_free(mod); =20 + /* Notify the secure kernel that this module is being unloaded + * so it can release any EPT permission overrides. */ + if (vbs_available()) + vbs_unload_module(mod); + codetag_unload_module(mod); =20 mod_sysfs_teardown(mod); @@ -3472,6 +3478,22 @@ static int load_module(struct load_info *info, const= char __user *uargs, if (err) goto free_module; =20 + /* + * If VBS is available, ask the secure kernel (plane-1) to + * validate this module. We pass the module name and the + * sig_ok flag from the kernel's own signature check. + * Plane-1 can enforce additional policy (e.g., allowlist). + */ + if (vbs_available()) { + err =3D vbs_validate_module(info->hdr, info->len, + NULL, info->sig_ok ? 1 : 0); + if (err) { + pr_warn("vbs: module '%s' rejected by secure kernel (%ld)\n", + mod->name, err); + goto unlink_mod; + } + } + /* * We are tainting your kernel if your module gets into * the modules linked list somehow. @@ -3539,6 +3561,21 @@ static int load_module(struct load_info *info, const= char __user *uargs, if (err) goto ddebug_cleanup; =20 + /* + * If VBS is available, send the module's per-section layout + * to the secure kernel so it can enforce EPT permissions: + * text =E2=86=92 R+X (NO_WRITE), rodata =E2=86=92 R (NO_WRITE|NO_EXEC), + * data =E2=86=92 R+W (no restrictions). + */ + if (vbs_available()) { + err =3D vbs_set_module_perms(mod); + if (err) + pr_warn("vbs: set_module_perms for %s failed (%ld)\n", + mod->name, err); + /* Non-fatal: continue loading even if protection fails */ + err =3D 0; + } + err =3D prepare_coming_module(mod); if (err) goto bug_cleanup; diff --git a/security/vbs/heki.h b/security/vbs/heki.h index fee986de351a..5b7fa92bce21 100644 --- a/security/vbs/heki.h +++ b/security/vbs/heki.h @@ -36,6 +36,52 @@ struct vbs_seal_kernel_req { __u64 cr3; /* plane-0 kernel CR3 for verification */ } __packed; =20 +/* =E2=94=80=E2=94=80 Module authentication =E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80 */ + +/* + * VBS_CALL_VALIDATE_MODULE payload =E2=80=94 plane-0 sends the GPA of the= module + * ELF blob and its appended PKCS#7 signature for plane-1 verification. + * The module blob is in guest physical memory; the secure side reads it + * directly via the GPA (no copy through the CAA page). + */ +struct vbs_validate_module_req { + char name[56]; /* module name (null-terminated) */ + __u64 elf_gpa; /* GPA of the module ELF data */ + __u64 elf_size; /* size of the ELF data (excl. signature) */ + __u32 sig_ok; /* 1 if kernel's sig check passed */ + __u32 reserved; /* padding */ +} __packed; + +/* + * Per-section descriptor for VBS_CALL_SET_MODULE_PERMS. + * Sent as an array in the CAA buffer after the module name. + */ +struct vbs_module_section { + __u64 gpa; /* section GPA (page-aligned) */ + __u64 size; /* section size (page-aligned) */ + __u32 perms; /* VBS_MEM_* permission flags */ + __u32 type; /* enum mod_mem_type */ +} __packed; + +/* + * VBS_CALL_SET_MODULE_PERMS payload =E2=80=94 after relocation, plane-0 s= ends + * the per-section layout so plane-1 can set EPT permissions. + * Sections follow immediately after this header in the buffer. + */ +struct vbs_set_module_perms_req { + char name[56]; /* module name (null-terminated) */ + __u32 nr_sections; /* number of vbs_module_section entries */ + __u32 flags; /* reserved, must be 0 */ + /* struct vbs_module_section sections[]; follows in buffer */ +} __packed; + +/* + * VBS_CALL_UNLOAD_MODULE payload =E2=80=94 module is being freed. + */ +struct vbs_unload_module_req { + char name[56]; /* module name (null-terminated) */ +} __packed; + /* =E2=94=80=E2=94=80 x86-64 page table walker (for plane-1 auditing) =E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80 */ =20 /* Classification of a guest-physical page based on page table walk */ diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c index 293c960c0968..1114adfbd46c 100644 --- a/security/vbs/kvm_planes.c +++ b/security/vbs/kvm_planes.c @@ -23,6 +23,9 @@ #include #include #include +#include +#include +#include #include #include #include @@ -147,26 +150,111 @@ static int kvm_planes_seal_kernel(void) static int kvm_planes_validate_module(const void *elf, size_t elf_size, const void *sig, size_t sig_size) { + struct vbs_validate_module_req req =3D {}; + struct page *elf_page; + const Elf64_Ehdr *ehdr; + + if (!elf || !elf_size) + return -EINVAL; + /* - * Module blobs can be large =E2=80=94 for the KVM planes backend we pass - * the physical address and size to plane-1 via the VTL call and - * let plane-1 map/read the pages directly from its EPT view. - * For now, a stub that signals "not yet implemented". + * sig_size is repurposed: 1 =3D kernel's own sig check passed, + * 0 =3D module is unsigned or sig check failed. */ + req.sig_ok =3D sig_size ? 1 : 0; + + /* Try to extract the module name from the ELF .modinfo section. + * For now, just use a placeholder =E2=80=94 the name is available at + * the call site in load_module() but not passed through the + * vbs_ops interface which takes (elf, elf_size, sig, sig_size). + */ + ehdr =3D elf; + if (elf_size >=3D sizeof(*ehdr) && ehdr->e_ident[0] =3D=3D 0x7f) + strscpy(req.name, "module", sizeof(req.name)); + else + strscpy(req.name, "unknown", sizeof(req.name)); + + /* Get GPA of the ELF blob */ + elf_page =3D vmalloc_to_page(elf); + if (elf_page) { + req.elf_gpa =3D page_to_phys(elf_page) + + offset_in_page(elf); + req.elf_size =3D elf_size; + } + + pr_debug("vbs-kvm: validate_module elf_gpa=3D0x%llx size=3D0x%llx sig_ok= =3D%u\n", + req.elf_gpa, req.elf_size, req.sig_ok); + return kvm_planes_vtl_call(VBS_CALL_VALIDATE_MODULE, - NULL, 0, NULL, 0); + &req, sizeof(req), NULL, 0); } =20 static int kvm_planes_set_module_perms(const struct module *mod) { + struct { + struct vbs_set_module_perms_req hdr; + struct vbs_module_section sections[MOD_MEM_NUM_TYPES]; + } __packed req =3D {}; + int i, n =3D 0; + + strscpy(req.hdr.name, mod->name, sizeof(req.hdr.name)); + + for (i =3D 0; i < MOD_MEM_NUM_TYPES; i++) { + const struct module_memory *mem =3D &mod->mem[i]; + struct vbs_module_section *sec; + unsigned long gpa; + struct page *p; + + if (!mem->base || !mem->size) + continue; + + p =3D vmalloc_to_page(mem->base); + if (!p) + continue; + + gpa =3D page_to_phys(p) + offset_in_page(mem->base); + sec =3D &req.sections[n]; + sec->gpa =3D gpa; + sec->size =3D PAGE_ALIGN(mem->size); + sec->type =3D i; + + /* Set permissions based on section type */ + switch (i) { + case MOD_TEXT: + case MOD_INIT_TEXT: + sec->perms =3D VBS_MEM_READ | VBS_MEM_EXEC; + break; + case MOD_RODATA: + case MOD_RO_AFTER_INIT: + case MOD_INIT_RODATA: + sec->perms =3D VBS_MEM_READ; + break; + default: /* MOD_DATA, MOD_INIT_DATA */ + sec->perms =3D VBS_MEM_READ | VBS_MEM_WRITE; + break; + } + n++; + } + + req.hdr.nr_sections =3D n; + + pr_debug("vbs-kvm: set_module_perms %s: %d sections\n", + mod->name, n); + return kvm_planes_vtl_call(VBS_CALL_SET_MODULE_PERMS, - NULL, 0, NULL, 0); + &req, + sizeof(req.hdr) + n * sizeof(req.sections[0]), + NULL, 0); } =20 static int kvm_planes_unload_module(const struct module *mod) { + struct vbs_unload_module_req req =3D {}; + + strscpy(req.name, mod->name, sizeof(req.name)); + return kvm_planes_vtl_call(VBS_CALL_UNLOAD_MODULE, - NULL, 0, NULL, 0); + &req, sizeof(req), NULL, 0); } =20 /* =E2=94=80=E2=94=80 key / certificate management =E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80 */ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id ABF9443E9C5; Wed, 5 Aug 2026 11:03:59 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927841; cv=none; b=rDzMauVESzCqEIEEpKilmfBavuS6kxJyCTOPboFqPzKE1cZ457enAwf0MGFqM2NDBKywPK5ZMKCcWZKQzsgJEUor3F7Z/LxPieXabuSY9pI+WxfsigUb3CwDpGsIq8aFomvl/1kKQLZ327rBZweQucvavATarBl6kAvAn48ZDJU= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927841; c=relaxed/simple; bh=NYoFCJs7hcwY6WkT5Rnwoy1VBXTXZnunw8e1HMzEKBs=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=hXKj2MTZw2K/fk5WZmWLJqlP0i6kUNMdroW3il74Evo3IWa2ZUdHzB88HxeuFT84L1qpfWr6z8iZwgTa3PZToCVNQiP6vkFgsTVBBIVpSby/vp1TkW12fFP5fDXFDT2c9qGeII+8YVZESNjra4MTOJysXuRc7kbivmsqUcaAPr0= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=AqmyHShP; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="AqmyHShP" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 7FD0020B716B; Wed, 5 Aug 2026 04:03:38 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 7FD0020B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927818; bh=dTRcKNCHeBGgLTvtUT3mS37PqtwLmSq7VZWMho7DmDA=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=AqmyHShPKsLZWg6hrykMQ0zUwrOUwVSlyzMGzd+NJLJggTIxLPWosaHXkI2axthP+ DTuokXAWrTsMBQ9ZaXPz/s/JtFEerPvNcWQYItuwlRhLo1mQR1vNwBEUa8u4ui/Rk7 dvuZLS4pBMAGZTTXW8KrnY08eG7+TJug4c/YlgW8= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 18/42] vbs: Add kexec validation and make module auth non-fatal Date: Wed, 5 Aug 2026 04:03:00 -0700 Message-ID: <20260805110324.25067-19-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Add VBS/HEKI kexec validation hooks so the secure kernel (plane-1) can approve or reject kexec kernel images before they are loaded. kexec_file.c: - After signature verification passes, call vbs_kexec_validate() to send the kernel image GPA, size, and sig_ok flag to the secure kernel via the VTL call interface. - If the secure kernel rejects the image, kexec_file_load fails. kexec_core.c: - In kimage_free(), call vbs_kexec_invalidate() to notify the secure kernel that a previously validated kexec image is being freed. security/vbs/heki.h: - Add struct vbs_kexec_validate_req (kernel_gpa, kernel_size, sig_ok, flags). security/vbs/kvm_planes.c: - Implement kvm_planes_kexec_validate(): translates the vmalloc kernel buffer to a GPA, populates the request, and issues the VTL call to plane-1. - Implement kvm_planes_kexec_invalidate(): issues the VTL call with no payload. kernel/module/main.c: - Change VBS module validation from fatal to non-fatal. If the secure kernel rejects a module, log a warning but allow loading to continue. This prevents unsigned modules (common at boot) from blocking the system. A strict policy can be enforced later. Signed-off-by: Sriram Nambakam --- kernel/kexec_core.c | 5 +++++ kernel/kexec_file.c | 20 ++++++++++++++++++++ kernel/module/main.c | 9 +++++---- security/vbs/heki.h | 14 ++++++++++++++ security/vbs/kvm_planes.c | 24 +++++++++++++++++++++++- 5 files changed, 67 insertions(+), 5 deletions(-) diff --git a/kernel/kexec_core.c b/kernel/kexec_core.c index dc770b9a6d05..a7bdbfaf68c8 100644 --- a/kernel/kexec_core.c +++ b/kernel/kexec_core.c @@ -43,6 +43,7 @@ #include #include #include +#include =20 #include #include @@ -580,6 +581,10 @@ void kimage_free(struct kimage *image) if (!image) return; =20 + /* Notify the secure kernel that a kexec image is being freed */ + if (vbs_available()) + vbs_kexec_invalidate(); + #ifdef CONFIG_CRASH_DUMP if (image->vmcoreinfo_data_copy) { crash_update_vmcoreinfo_safecopy(NULL); diff --git a/kernel/kexec_file.c b/kernel/kexec_file.c index 2bfbb2d144e6..81cf454ab516 100644 --- a/kernel/kexec_file.c +++ b/kernel/kexec_file.c @@ -27,6 +27,7 @@ #include #include #include +#include #include "kexec_internal.h" =20 #ifdef CONFIG_KEXEC_SIG @@ -243,6 +244,25 @@ kimage_file_prepare_segments(struct kimage *image, int= kernel_fd, int initrd_fd, if (ret) goto out; #endif + + /* + * If VBS is available, ask the secure kernel (plane-1) to + * validate the kexec kernel image. Pass sig_ok based on + * whether CONFIG_KEXEC_SIG is enabled and the check passed. + */ + if (vbs_available()) { + int sig_ok =3D 0; +#ifdef CONFIG_KEXEC_SIG + sig_ok =3D 1; /* we got here, so sig check passed */ +#endif + ret =3D vbs_kexec_validate(image->kernel_buf, + image->kernel_buf_len, + NULL, sig_ok); + if (ret) { + pr_warn("vbs: kexec kernel rejected by secure kernel (%d)\n", ret); + goto out; + } + } /* It is possible that there no initramfs is being loaded */ if (!(flags & KEXEC_FILE_NO_INITRAMFS)) { ret =3D kernel_read_file_from_fd(initrd_fd, 0, &image->initrd_buf, diff --git a/kernel/module/main.c b/kernel/module/main.c index 2d0232fccf18..3b46d6c0fb41 100644 --- a/kernel/module/main.c +++ b/kernel/module/main.c @@ -3487,11 +3487,12 @@ static int load_module(struct load_info *info, cons= t char __user *uargs, if (vbs_available()) { err =3D vbs_validate_module(info->hdr, info->len, NULL, info->sig_ok ? 1 : 0); - if (err) { - pr_warn("vbs: module '%s' rejected by secure kernel (%ld)\n", + if (err) + pr_warn("vbs: module '%s' validation returned (%ld) =E2=80=94 continuin= g\n", mod->name, err); - goto unlink_mod; - } + /* Non-fatal: allow loading to continue even if VBS rejects. + * A strict policy can be enforced later by changing this. */ + err =3D 0; } =20 /* diff --git a/security/vbs/heki.h b/security/vbs/heki.h index 5b7fa92bce21..fb485f171045 100644 --- a/security/vbs/heki.h +++ b/security/vbs/heki.h @@ -82,6 +82,20 @@ struct vbs_unload_module_req { char name[56]; /* module name (null-terminated) */ } __packed; =20 +/* =E2=94=80=E2=94=80 Kexec validation =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +/* + * VBS_CALL_KEXEC_VALIDATE payload =E2=80=94 plane-0 sends the GPA and siz= e of + * the kexec kernel image for plane-1 validation before allowing the + * kexec to proceed. + */ +struct vbs_kexec_validate_req { + __u64 kernel_gpa; /* GPA of the kernel image buffer */ + __u64 kernel_size; /* size of the kernel image */ + __u32 sig_ok; /* 1 if kernel's sig check passed */ + __u32 flags; /* reserved, must be 0 */ +} __packed; + /* =E2=94=80=E2=94=80 x86-64 page table walker (for plane-1 auditing) =E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80 */ =20 /* Classification of a guest-physical page based on page table walk */ diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c index 1114adfbd46c..061163a4d303 100644 --- a/security/vbs/kvm_planes.c +++ b/security/vbs/kvm_planes.c @@ -282,12 +282,34 @@ static int kvm_planes_send_certs(const void *certs, s= ize_t certs_size) static int kvm_planes_kexec_validate(const void *kernel, size_t kernel_siz= e, const void *sig, size_t sig_size) { + struct vbs_kexec_validate_req req =3D {}; + struct page *page; + + if (!kernel || !kernel_size) + return -EINVAL; + + /* + * sig_size is repurposed: 1 =3D kernel's sig check passed, + * 0 =3D unsigned or failed (same pattern as module validation). + */ + req.sig_ok =3D sig_size ? 1 : 0; + req.kernel_size =3D kernel_size; + + /* Get GPA of the kernel image buffer (first page) */ + page =3D vmalloc_to_page(kernel); + if (page) + req.kernel_gpa =3D page_to_phys(page) + offset_in_page(kernel); + + pr_info("vbs-kvm: kexec_validate gpa=3D0x%llx size=3D0x%llx sig_ok=3D%u\n= ", + req.kernel_gpa, req.kernel_size, req.sig_ok); + return kvm_planes_vtl_call(VBS_CALL_KEXEC_VALIDATE, - NULL, 0, NULL, 0); + &req, sizeof(req), NULL, 0); } =20 static int kvm_planes_kexec_invalidate(void) { + pr_info("vbs-kvm: kexec_invalidate\n"); return kvm_planes_vtl_call(VBS_CALL_KEXEC_INVALIDATE, NULL, 0, NULL, 0); } --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 8511D43FD04; Wed, 5 Aug 2026 11:04:00 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927841; cv=none; b=edPvV/gVMDjC61YM++pwPLFpx7SDHC6tv7dGmzA9eXFUYCVGg0uFwsUXXkKMu5ZGHGQFthx5HLw/pcT3rS5FiVDf7G6DiSqxTEK8gMcssqIK9npOCQTqHMwJfArP0Vijym5O94xRw7ghVt+E9TmPNpRaccUbRzN/8C1YOQKFKUM= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927841; c=relaxed/simple; bh=D9VOq56B9kZmsm/OCIMpL/keYWy0pJvDDZkuomsAm3Y=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=B1s95ooSHlrJEUr+ZXfITLQQu6gw3bg7PUHrgWd9EBdGa0G1y10x6qLcIZ68etnBnWGq+ToiCMoqreDfdhJraJr1dWgvzej8GycknXBeibSbhF0LPehJCEAdrv7dlzA1Dez6GAZUP/HmnswMhF08mGQX3kHE8C+2ZJxSTsCoK9A= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=eva/5TAa; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="eva/5TAa" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 6854120B716C; Wed, 5 Aug 2026 04:03:39 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 6854120B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927819; bh=KADTfjk6xMNFkykpB6ZFJP7LLPsxtzvPwQF0sjffH+E=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=eva/5TAa9sagOk/0+z0Q89BEuXl0wyBzL4Le9cVirI9YYNj5FHHoLGc6AhPXOx+cb Luuou3TmfgJIIqTkGrg8AHevoxP2TCNEI5grvaKVP9JzvGteqn798i+O+p+7RRWQYb QWIvok70Cjz2SP5mUqsuA/SnHNDBsiwU6t0J48Pg= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 19/42] Merge branch 'master' into vm-planes Date: Wed, 5 Aug 2026 04:03:01 -0700 Message-ID: <20260805110324.25067-20-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- arch/x86/kvm/svm/svm.c | 2 +- arch/x86/kvm/x86.c | 2 +- kernel/kexec_file.c | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/x86/kvm/svm/svm.c b/arch/x86/kvm/svm/svm.c index ce242e86c5ea..bc281fbc54c2 100644 --- a/arch/x86/kvm/svm/svm.c +++ b/arch/x86/kvm/svm/svm.c @@ -3240,7 +3240,7 @@ static int interrupt_window_interception(struct kvm_v= cpu *vcpu) kvm_make_request(KVM_REQ_EVENT, vcpu); svm_clear_vintr(to_svm(vcpu)); =20 - ++vcpu->stat.irq_window_exits; + ++vcpu->stat->irq_window_exits; return 1; } =20 diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index 4b99016fe536..d4210053e6b8 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -11129,7 +11129,7 @@ void kvm_inc_or_dec_irq_window_inhibit(struct kvm *= kvm, bool inc) */ guard(rwsem_write)(&kvm->arch.apicv_update_lock); if (atomic_add_return(add, &kvm->arch.apicv_nr_irq_window_req) =3D=3D inc) - __kvm_set_or_clear_apicv_inhibit(kvm, APICV_INHIBIT_REASON_IRQWIN, inc); + __kvm_set_or_clear_apicv_inhibit(kvm->planes[0], APICV_INHIBIT_REASON_IR= QWIN, inc); } EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_inc_or_dec_irq_window_inhibit); =20 diff --git a/kernel/kexec_file.c b/kernel/kexec_file.c index 81cf454ab516..da44bac7df74 100644 --- a/kernel/kexec_file.c +++ b/kernel/kexec_file.c @@ -259,7 +259,7 @@ kimage_file_prepare_segments(struct kimage *image, int = kernel_fd, int initrd_fd, image->kernel_buf_len, NULL, sig_ok); if (ret) { - pr_warn("vbs: kexec kernel rejected by secure kernel (%d)\n", ret); + pr_warn("vbs: kexec kernel rejected by secure kernel (%zd)\n", ret); goto out; } } --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id F027A443E4E; Wed, 5 Aug 2026 11:04:01 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927844; cv=none; b=qLl7vqGQRFlu/notkBk46uPYVkB9bHrkAND3NSDnInpMfqu3rX+22aepvsO16oe3Zc4auQ4plSm6TQq2W0UKOZvpnH50BwO9B6oms249I7nBNv0MzGjHx9vVXNW2dCg36b1yzDLeppIFRtgQLk39wQyVGpCW0DkV7KavTMmuOVk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927844; c=relaxed/simple; bh=2f9BkvlL9Yr3+PSZYxQVGcsGqhBZBFCCZJwatJO8m4E=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=O3a/vQwfrIE7B3myTvTxBC2QaLTyHhqcNJ+N3V+wmxorsRxiAU7AAvlq8bZEh+6FDrtIrBg3N3NI06Jg222O+uaKDbd2UmZM+9kUfQ650+Rr/q7/9S1/OKS0pCamYl/cO9SzSAmaOiSAvqipb5kQS8NLG8q4XUgwdZz2WePBJYM= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=Yt33GxiV; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="Yt33GxiV" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 5D9DB20B7169; Wed, 5 Aug 2026 04:03:40 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 5D9DB20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927820; bh=9JtBsdQ4lK3gWV7uhv378cf5+zPWsZh7yqwQpTe6NDs=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=Yt33GxiVUzBR7+o/rUOqXRAnXT/Op8OLmf+d6s0LXcrqW7p2ask83Tu96DAjLNUFu hyNxfnVzHA6vDBW2a3w41rzCo5j50AOdRIoAcyCNDcc3fcGcb3Ioe00o/sEHPbL3Tx beiARPrflufI3ROsthiKv2dk4ezqoMEmRW0SeHgA= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 20/42] kvm: x86: fix merged plane API/stat build regressions Date: Wed, 5 Aug 2026 04:03:02 -0700 Message-ID: <20260805110324.25067-21-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- arch/x86/kvm/debugfs.c | 2 +- arch/x86/kvm/hyperv.c | 8 +- arch/x86/kvm/kvm_cache_regs.h | 249 ++++++++++++++++++++++++++++++++++ arch/x86/kvm/mmu/mmu.c | 41 +++--- arch/x86/kvm/mmu/spte.h | 10 +- arch/x86/kvm/mmu/tdp_mmu.c | 2 +- arch/x86/kvm/svm/avic.c | 2 +- arch/x86/kvm/svm/sev.c | 4 +- arch/x86/kvm/vmx/vmx.c | 20 +-- arch/x86/kvm/xen.c | 2 - include/uapi/linux/kvm.h | 2 + virt/kvm/guest_memfd.c | 3 +- 12 files changed, 291 insertions(+), 54 deletions(-) create mode 100644 arch/x86/kvm/kvm_cache_regs.h diff --git a/arch/x86/kvm/debugfs.c b/arch/x86/kvm/debugfs.c index 192cc7228197..0074a56e45b4 100644 --- a/arch/x86/kvm/debugfs.c +++ b/arch/x86/kvm/debugfs.c @@ -24,7 +24,7 @@ DEFINE_SIMPLE_ATTRIBUTE(vcpu_timer_advance_ns_fops, vcpu_= get_timer_advance_ns, N static int vcpu_get_guest_mode(void *data, u64 *val) { struct kvm_vcpu *vcpu =3D (struct kvm_vcpu *) data; - *val =3D vcpu->stat->guest_mode; + *val =3D vcpu->stat.guest_mode; return 0; } =20 diff --git a/arch/x86/kvm/hyperv.c b/arch/x86/kvm/hyperv.c index 75d5d7f7994e..ee6b32d2a5cb 100644 --- a/arch/x86/kvm/hyperv.c +++ b/arch/x86/kvm/hyperv.c @@ -145,7 +145,7 @@ static void synic_update_vector(struct kvm_vcpu_hv_syni= c *synic, * Inhibit APICv if any vCPU is using SynIC's AutoEOI, which relies on * the hypervisor to manually inject IRQs. */ - __kvm_set_or_clear_apicv_inhibit(vcpu_to_plane(vcpu), + __kvm_set_or_clear_apicv_inhibit(vcpu->kvm, APICV_INHIBIT_REASON_HYPERV, !!hv->synic_auto_eoi_used); =20 @@ -491,8 +491,6 @@ static int synic_set_irq(struct kvm_vcpu_hv_synic *syni= c, u32 sint) irq.delivery_mode =3D APIC_DM_FIXED; irq.vector =3D vector; irq.level =3D 1; - irq.plane =3D vcpu->plane; - ret =3D kvm_irq_delivery_to_apic(vcpu->plane, vcpu->arch.apic, &irq); trace_kvm_hv_synic_set_irq(vcpu->vcpu_id, sint, irq.vector, ret); return ret; @@ -1999,7 +1997,7 @@ int kvm_hv_vcpu_flush_tlb(struct kvm_vcpu *vcpu) kvm_x86_call(flush_tlb_gva)(vcpu, gva + j * PAGE_SIZE); } =20 - ++vcpu->stat->tlb_flush; + ++vcpu->stat.tlb_flush; } return 0; =20 @@ -2403,7 +2401,7 @@ static int kvm_hv_hypercall_complete(struct kvm_vcpu = *vcpu, u64 result) =20 trace_kvm_hv_hypercall_done(result); kvm_hv_hypercall_set_result(vcpu, result); - ++vcpu->stat->hypercalls; + ++vcpu->stat.hypercalls; =20 ret =3D kvm_skip_emulated_instruction(vcpu); =20 diff --git a/arch/x86/kvm/kvm_cache_regs.h b/arch/x86/kvm/kvm_cache_regs.h new file mode 100644 index 000000000000..8ddb01191d6f --- /dev/null +++ b/arch/x86/kvm/kvm_cache_regs.h @@ -0,0 +1,249 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef ASM_KVM_CACHE_REGS_H +#define ASM_KVM_CACHE_REGS_H + +#include + +#define KVM_POSSIBLE_CR0_GUEST_BITS (X86_CR0_TS | X86_CR0_WP) +#define KVM_POSSIBLE_CR4_GUEST_BITS \ + (X86_CR4_PVI | X86_CR4_DE | X86_CR4_PCE | X86_CR4_OSFXSR \ + | X86_CR4_OSXMMEXCPT | X86_CR4_PGE | X86_CR4_TSD | X86_CR4_FSGSBASE \ + | X86_CR4_CET) + +#define X86_CR0_PDPTR_BITS (X86_CR0_CD | X86_CR0_NW | X86_CR0_PG) +#define X86_CR4_TLBFLUSH_BITS (X86_CR4_PGE | X86_CR4_PCIDE | X86_CR4_PAE |= X86_CR4_SMEP) +#define X86_CR4_PDPTR_BITS (X86_CR4_PGE | X86_CR4_PSE | X86_CR4_PAE | X= 86_CR4_SMEP) + +static_assert(!(KVM_POSSIBLE_CR0_GUEST_BITS & X86_CR0_PDPTR_BITS)); + +#define BUILD_KVM_GPR_ACCESSORS(lname, uname) \ +static __always_inline unsigned long kvm_##lname##_read(struct kvm_vcpu *v= cpu)\ +{ \ + return vcpu->arch.regs[VCPU_REGS_##uname]; \ +} \ +static __always_inline void kvm_##lname##_write(struct kvm_vcpu *vcpu, = \ + unsigned long val) \ +{ \ + vcpu->arch.regs[VCPU_REGS_##uname] =3D val; \ +} +BUILD_KVM_GPR_ACCESSORS(rax, RAX) +BUILD_KVM_GPR_ACCESSORS(rbx, RBX) +BUILD_KVM_GPR_ACCESSORS(rcx, RCX) +BUILD_KVM_GPR_ACCESSORS(rdx, RDX) +BUILD_KVM_GPR_ACCESSORS(rbp, RBP) +BUILD_KVM_GPR_ACCESSORS(rsi, RSI) +BUILD_KVM_GPR_ACCESSORS(rdi, RDI) +#ifdef CONFIG_X86_64 +BUILD_KVM_GPR_ACCESSORS(r8, R8) +BUILD_KVM_GPR_ACCESSORS(r9, R9) +BUILD_KVM_GPR_ACCESSORS(r10, R10) +BUILD_KVM_GPR_ACCESSORS(r11, R11) +BUILD_KVM_GPR_ACCESSORS(r12, R12) +BUILD_KVM_GPR_ACCESSORS(r13, R13) +BUILD_KVM_GPR_ACCESSORS(r14, R14) +BUILD_KVM_GPR_ACCESSORS(r15, R15) +#endif + +/* + * Using the register cache from interrupt context is generally not allowe= d, as + * caching a register and marking it available/dirty can't be done atomica= lly, + * i.e. accesses from interrupt context may clobber state or read stale da= ta if + * the vCPU task is in the process of updating the cache. The exception i= s if + * KVM is handling a PMI IRQ/NMI VM-Exit, as that bound code sequence does= n't + * touch the cache, it runs after the cache is reset (post VM-Exit), and P= MIs + * need to access several registers that are cacheable. + */ +#define kvm_assert_register_caching_allowed(vcpu) \ + lockdep_assert_once(in_task() || kvm_arch_pmi_in_guest(vcpu)) + +/* + * avail dirty + * 0 0 register in VMCS/VMCB + * 0 1 *INVALID* + * 1 0 register in vcpu->arch + * 1 1 register in vcpu->arch, needs to be stored back + */ +static inline bool kvm_register_is_available(struct kvm_vcpu *vcpu, + enum kvm_reg reg) +{ + kvm_assert_register_caching_allowed(vcpu); + return test_bit(reg, (unsigned long *)&vcpu->arch.regs_avail); +} + +static inline bool kvm_register_is_dirty(struct kvm_vcpu *vcpu, + enum kvm_reg reg) +{ + kvm_assert_register_caching_allowed(vcpu); + return test_bit(reg, (unsigned long *)&vcpu->arch.regs_dirty); +} + +static inline void kvm_register_mark_available(struct kvm_vcpu *vcpu, + enum kvm_reg reg) +{ + kvm_assert_register_caching_allowed(vcpu); + __set_bit(reg, (unsigned long *)&vcpu->arch.regs_avail); +} + +static inline void kvm_register_mark_dirty(struct kvm_vcpu *vcpu, + enum kvm_reg reg) +{ + kvm_assert_register_caching_allowed(vcpu); + __set_bit(reg, (unsigned long *)&vcpu->arch.regs_avail); + __set_bit(reg, (unsigned long *)&vcpu->arch.regs_dirty); +} + +/* + * kvm_register_test_and_mark_available() is a special snowflake that uses= an + * arch bitop directly to avoid the explicit instrumentation that comes wi= th + * the generic bitops. This allows code that cannot be instrumented (noin= str + * functions), e.g. the low level VM-Enter/VM-Exit paths, to cache registe= rs. + */ +static __always_inline bool kvm_register_test_and_mark_available(struct kv= m_vcpu *vcpu, + enum kvm_reg reg) +{ + kvm_assert_register_caching_allowed(vcpu); + return arch___test_and_set_bit(reg, (unsigned long *)&vcpu->arch.regs_ava= il); +} + +/* + * The "raw" register helpers are only for cases where the full 64 bits of= a + * register are read/written irrespective of current vCPU mode. In other = words, + * odds are good you shouldn't be using the raw variants. + */ +static inline unsigned long kvm_register_read_raw(struct kvm_vcpu *vcpu, i= nt reg) +{ + if (WARN_ON_ONCE((unsigned int)reg >=3D NR_VCPU_REGS)) + return 0; + + if (!kvm_register_is_available(vcpu, reg)) + kvm_x86_call(cache_reg)(vcpu, reg); + + return vcpu->arch.regs[reg]; +} + +static inline void kvm_register_write_raw(struct kvm_vcpu *vcpu, int reg, + unsigned long val) +{ + if (WARN_ON_ONCE((unsigned int)reg >=3D NR_VCPU_REGS)) + return; + + vcpu->arch.regs[reg] =3D val; + kvm_register_mark_dirty(vcpu, reg); +} + +static inline unsigned long kvm_rip_read(struct kvm_vcpu *vcpu) +{ + return kvm_register_read_raw(vcpu, VCPU_REGS_RIP); +} + +static inline void kvm_rip_write(struct kvm_vcpu *vcpu, unsigned long val) +{ + kvm_register_write_raw(vcpu, VCPU_REGS_RIP, val); +} + +static inline unsigned long kvm_rsp_read(struct kvm_vcpu *vcpu) +{ + return kvm_register_read_raw(vcpu, VCPU_REGS_RSP); +} + +static inline void kvm_rsp_write(struct kvm_vcpu *vcpu, unsigned long val) +{ + kvm_register_write_raw(vcpu, VCPU_REGS_RSP, val); +} + +static inline u64 kvm_pdptr_read(struct kvm_vcpu *vcpu, int index) +{ + might_sleep(); /* on svm */ + + if (!kvm_register_is_available(vcpu, VCPU_EXREG_PDPTR)) + kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_PDPTR); + + return vcpu->arch.walk_mmu->pdptrs[index]; +} + +static inline void kvm_pdptr_write(struct kvm_vcpu *vcpu, int index, u64 v= alue) +{ + vcpu->arch.walk_mmu->pdptrs[index] =3D value; +} + +static inline ulong kvm_read_cr0_bits(struct kvm_vcpu *vcpu, ulong mask) +{ + ulong tmask =3D mask & KVM_POSSIBLE_CR0_GUEST_BITS; + if ((tmask & vcpu->arch.cr0_guest_owned_bits) && + !kvm_register_is_available(vcpu, VCPU_EXREG_CR0)) + kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_CR0); + return vcpu->arch.cr0 & mask; +} + +static __always_inline bool kvm_is_cr0_bit_set(struct kvm_vcpu *vcpu, + unsigned long cr0_bit) +{ + BUILD_BUG_ON(!is_power_of_2(cr0_bit)); + + return !!kvm_read_cr0_bits(vcpu, cr0_bit); +} + +static inline ulong kvm_read_cr0(struct kvm_vcpu *vcpu) +{ + return kvm_read_cr0_bits(vcpu, ~0UL); +} + +static inline ulong kvm_read_cr4_bits(struct kvm_vcpu *vcpu, ulong mask) +{ + ulong tmask =3D mask & KVM_POSSIBLE_CR4_GUEST_BITS; + if ((tmask & vcpu->arch.cr4_guest_owned_bits) && + !kvm_register_is_available(vcpu, VCPU_EXREG_CR4)) + kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_CR4); + return vcpu->arch.cr4 & mask; +} + +static __always_inline bool kvm_is_cr4_bit_set(struct kvm_vcpu *vcpu, + unsigned long cr4_bit) +{ + BUILD_BUG_ON(!is_power_of_2(cr4_bit)); + + return !!kvm_read_cr4_bits(vcpu, cr4_bit); +} + +static inline ulong kvm_read_cr3(struct kvm_vcpu *vcpu) +{ + if (!kvm_register_is_available(vcpu, VCPU_EXREG_CR3)) + kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_CR3); + return vcpu->arch.cr3; +} + +static inline ulong kvm_read_cr4(struct kvm_vcpu *vcpu) +{ + return kvm_read_cr4_bits(vcpu, ~0UL); +} + +static inline u64 kvm_read_edx_eax(struct kvm_vcpu *vcpu) +{ + return (kvm_rax_read(vcpu) & -1u) + | ((u64)(kvm_rdx_read(vcpu) & -1u) << 32); +} + +static inline void enter_guest_mode(struct kvm_vcpu *vcpu) +{ + vcpu->arch.hflags |=3D HF_GUEST_MASK; + vcpu->stat.guest_mode =3D 1; +} + +static inline void leave_guest_mode(struct kvm_vcpu *vcpu) +{ + vcpu->arch.hflags &=3D ~HF_GUEST_MASK; + + if (vcpu->arch.load_eoi_exitmap_pending) { + vcpu->arch.load_eoi_exitmap_pending =3D false; + kvm_make_request(KVM_REQ_LOAD_EOI_EXITMAP, vcpu); + } + + vcpu->stat.guest_mode =3D 0; +} + +static inline bool is_guest_mode(struct kvm_vcpu *vcpu) +{ + return vcpu->arch.hflags & HF_GUEST_MASK; +} + +#endif diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c index 3b861a42a712..6e41c5df72ed 100644 --- a/arch/x86/kvm/mmu/mmu.c +++ b/arch/x86/kvm/mmu/mmu.c @@ -3100,7 +3100,7 @@ static int mmu_set_spte(struct kvm_vcpu *vcpu, struct= kvm_memory_slot *slot, } =20 if (unlikely(is_noslot_pfn(pfn))) { - vcpu->stat->pf_mmio_spte_created++; + vcpu->stat.pf_mmio_spte_created++; mark_mmio_spte(vcpu, sptep, gfn, pte_access); if (flush) kvm_flush_remote_tlbs_gfn(vcpu->kvm, gfn, level); @@ -3809,7 +3809,7 @@ static int fast_page_fault(struct kvm_vcpu *vcpu, str= uct kvm_page_fault *fault) walk_shadow_page_lockless_end(vcpu); =20 if (ret !=3D RET_PF_INVALID) - vcpu->stat->pf_fast++; + vcpu->stat.pf_fast++; =20 return ret; } @@ -4602,7 +4602,7 @@ void kvm_arch_async_page_ready(struct kvm_vcpu *vcpu,= struct kvm_async_pf *work) * truly spurious and never trigger emulation */ if (r =3D=3D RET_PF_FIXED) - vcpu->stat->pf_fixed++; + vcpu->stat.pf_fixed++; } =20 static void kvm_mmu_finish_page_fault(struct kvm_vcpu *vcpu, @@ -6529,7 +6529,7 @@ int noinline kvm_mmu_page_fault(struct kvm_vcpu *vcpu= , gpa_t cr2_or_gpa, u64 err } =20 if (r =3D=3D RET_PF_INVALID) { - vcpu->stat->pf_taken++; + vcpu->stat.pf_taken++; =20 r =3D kvm_mmu_do_page_fault(vcpu, cr2_or_gpa, error_code, false, &emulation_type, NULL); @@ -6545,11 +6545,11 @@ int noinline kvm_mmu_page_fault(struct kvm_vcpu *vc= pu, gpa_t cr2_or_gpa, u64 err &emulation_type); =20 if (r =3D=3D RET_PF_FIXED) - vcpu->stat->pf_fixed++; + vcpu->stat.pf_fixed++; else if (r =3D=3D RET_PF_EMULATE) - vcpu->stat->pf_emulate++; + vcpu->stat.pf_emulate++; else if (r =3D=3D RET_PF_SPURIOUS) - vcpu->stat->pf_spurious++; + vcpu->stat.pf_spurious++; =20 /* * None of handle_mmio_page_fault(), kvm_mmu_do_page_fault(), or @@ -6663,7 +6663,7 @@ void kvm_mmu_invlpg(struct kvm_vcpu *vcpu, gva_t gva) * done here for them. */ kvm_mmu_invalidate_addr(vcpu, vcpu->arch.walk_mmu, gva, KVM_MMU_ROOTS_ALL= ); - ++vcpu->stat->invlpg; + ++vcpu->stat.invlpg; } EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_mmu_invlpg); =20 @@ -6685,7 +6685,7 @@ void kvm_mmu_invpcid_gva(struct kvm_vcpu *vcpu, gva_t= gva, unsigned long pcid) =20 if (roots) kvm_mmu_invalidate_addr(vcpu, mmu, gva, roots); - ++vcpu->stat->invlpg; + ++vcpu->stat.invlpg; =20 /* * Mappings not reachable via the current cr3 or the prev_roots will be @@ -8024,14 +8024,12 @@ static void hugepage_set_mixed(struct kvm_memory_sl= ot *slot, gfn_t gfn, lpage_info_slot(gfn, slot, level)->disallow_lpage |=3D KVM_LPAGE_MIXED_FL= AG; } =20 -bool kvm_arch_pre_set_memory_attributes(struct kvm_plane *plane, +bool kvm_arch_pre_set_memory_attributes(struct kvm *kvm, struct kvm_gfn_range *range) { struct kvm_memory_slot *slot =3D range->slot; int level; =20 - struct kvm *kvm =3D plane->kvm; - /* * Zap SPTEs even if the slot can't be mapped PRIVATE. KVM x86 only * supports KVM_MEMORY_ATTRIBUTE_PRIVATE, and so it *seems* like KVM @@ -8087,27 +8085,26 @@ bool kvm_arch_pre_set_memory_attributes(struct kvm_= plane *plane, return kvm_unmap_gfn_range(kvm, range); } =20 -static bool hugepage_has_attrs(struct kvm_plane *plane, struct kvm_memory_= slot *slot, +static bool hugepage_has_attrs(struct kvm *kvm, struct kvm_memory_slot *sl= ot, gfn_t gfn, int level, unsigned long attrs) { const unsigned long start =3D gfn; const unsigned long end =3D start + KVM_PAGES_PER_HPAGE(level); =20 if (level =3D=3D PG_LEVEL_2M) - return kvm_range_has_memory_attributes(plane, start, end, ~0, attrs); + return kvm_range_has_memory_attributes(kvm, start, end, ~0, attrs); =20 for (gfn =3D start; gfn < end; gfn +=3D KVM_PAGES_PER_HPAGE(level - 1)) { if (hugepage_test_mixed(slot, gfn, level - 1) || - attrs !=3D kvm_get_plane_memory_attributes(plane, gfn)) + attrs !=3D kvm_get_memory_attributes(kvm, gfn)) return false; } return true; } =20 -bool kvm_arch_post_set_memory_attributes(struct kvm_plane *plane, +bool kvm_arch_post_set_memory_attributes(struct kvm *kvm, struct kvm_gfn_range *range) { - struct kvm *kvm =3D plane->kvm; unsigned long attrs =3D range->arg.attributes; struct kvm_memory_slot *slot =3D range->slot; int level; @@ -8141,7 +8138,7 @@ bool kvm_arch_post_set_memory_attributes(struct kvm_p= lane *plane, */ if (gfn >=3D slot->base_gfn && gfn + nr_pages <=3D slot->base_gfn + slot->npages) { - if (hugepage_has_attrs(plane, slot, gfn, level, attrs)) + if (hugepage_has_attrs(kvm, slot, gfn, level, attrs)) hugepage_clear_mixed(slot, gfn, level); else hugepage_set_mixed(slot, gfn, level); @@ -8163,7 +8160,7 @@ bool kvm_arch_post_set_memory_attributes(struct kvm_p= lane *plane, */ if (gfn < range->end && (gfn + nr_pages) <=3D (slot->base_gfn + slot->npages)) { - if (hugepage_has_attrs(plane, slot, gfn, level, attrs)) + if (hugepage_has_attrs(kvm, slot, gfn, level, attrs)) hugepage_clear_mixed(slot, gfn, level); else hugepage_set_mixed(slot, gfn, level); @@ -8175,13 +8172,11 @@ bool kvm_arch_post_set_memory_attributes(struct kvm= _plane *plane, void kvm_mmu_init_memslot_memory_attributes(struct kvm *kvm, struct kvm_memory_slot *slot) { - struct kvm_plane *plane0; int level; =20 if (!kvm_arch_has_private_mem(kvm)) return; =20 - plane0 =3D kvm->planes[0]; for (level =3D PG_LEVEL_2M; level <=3D KVM_MAX_HUGEPAGE_LEVEL; level++) { /* * Don't bother tracking mixed attributes for pages that can't @@ -8201,9 +8196,9 @@ void kvm_mmu_init_memslot_memory_attributes(struct kv= m *kvm, * be manually checked as the attributes may already be mixed. */ for (gfn =3D start; gfn < end; gfn +=3D nr_pages) { - unsigned long attrs =3D kvm_get_plane_memory_attributes(plane0, gfn); + unsigned long attrs =3D kvm_get_memory_attributes(kvm, gfn); =20 - if (hugepage_has_attrs(plane0, slot, gfn, level, attrs)) + if (hugepage_has_attrs(kvm, slot, gfn, level, attrs)) hugepage_clear_mixed(slot, gfn, level); else hugepage_set_mixed(slot, gfn, level); diff --git a/arch/x86/kvm/mmu/spte.h b/arch/x86/kvm/mmu/spte.h index 421836fd3932..144f7c5a1040 100644 --- a/arch/x86/kvm/mmu/spte.h +++ b/arch/x86/kvm/mmu/spte.h @@ -580,8 +580,8 @@ void __init kvm_mmu_spte_module_init(void); void kvm_mmu_reset_all_pte_masks(void); =20 /* - * Apply per-plane memory protection attributes to pte_access. - * If the plane's mem_attr_array has NO_WRITE or NO_EXEC set for a GFN, + * Apply memory protection attributes to pte_access. + * If memory attributes have NO_WRITE or NO_EXEC set for a GFN, * strip the corresponding access bits before building the SPTE. */ #ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES @@ -589,13 +589,9 @@ static inline unsigned int kvm_plane_filter_pte_access= (struct kvm_vcpu *vcpu, gfn_t gfn, unsigned int pte_access) { - struct kvm_plane *plane =3D vcpu_to_plane(vcpu); unsigned long attrs; =20 - if (!plane) - return pte_access; - - attrs =3D kvm_get_plane_memory_attributes(plane, gfn); + attrs =3D kvm_get_memory_attributes(vcpu->kvm, gfn); if (attrs & KVM_MEMORY_ATTRIBUTE_NO_WRITE) pte_access &=3D ~ACC_WRITE_MASK; if (attrs & KVM_MEMORY_ATTRIBUTE_NO_EXEC) diff --git a/arch/x86/kvm/mmu/tdp_mmu.c b/arch/x86/kvm/mmu/tdp_mmu.c index 0603445377aa..83bee43a3f67 100644 --- a/arch/x86/kvm/mmu/tdp_mmu.c +++ b/arch/x86/kvm/mmu/tdp_mmu.c @@ -1165,7 +1165,7 @@ static int tdp_mmu_map_handle_target_level(struct kvm= _vcpu *vcpu, =20 /* If a MMIO SPTE is installed, the MMIO will need to be emulated. */ if (unlikely(is_mmio_spte(vcpu->kvm, new_spte))) { - vcpu->stat->pf_mmio_spte_created++; + vcpu->stat.pf_mmio_spte_created++; trace_mark_mmio_spte(rcu_dereference(iter->sptep), iter->gfn, new_spte); ret =3D RET_PF_EMULATE; diff --git a/arch/x86/kvm/svm/avic.c b/arch/x86/kvm/svm/avic.c index 251e36f5f0f7..58e493a80cb0 100644 --- a/arch/x86/kvm/svm/avic.c +++ b/arch/x86/kvm/svm/avic.c @@ -404,7 +404,7 @@ static int avic_init_backing_page(struct kvm_vcpu *vcpu) * fully initialized AVIC. */ if (id > max_id) { - kvm_set_apicv_inhibit(vcpu->kvm->planes[0], APICV_INHIBIT_REASON_PHYSICA= L_ID_TOO_BIG); + kvm_set_apicv_inhibit(vcpu->kvm, APICV_INHIBIT_REASON_PHYSICAL_ID_TOO_BI= G); vcpu->arch.apic->apicv_active =3D false; return 0; } diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index 79fee7ebc19b..53e76d22eb08 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -569,7 +569,7 @@ static int __sev_guest_init(struct kvm *kvm, struct kvm= _sev_cmd *argp, INIT_LIST_HEAD(&sev->mirror_vms); sev->need_init =3D false; =20 - kvm_set_apicv_inhibit(kvm->planes[0], APICV_INHIBIT_REASON_SEV); + kvm_set_apicv_inhibit(kvm, APICV_INHIBIT_REASON_SEV); =20 return 0; =20 @@ -4832,7 +4832,7 @@ int sev_handle_vmgexit(struct kvm_vcpu *vcpu) svm->sev_es.ghcb_sa); } case SVM_VMGEXIT_NMI_COMPLETE: - ++vcpu->stat->nmi_window_exits; + ++vcpu->stat.nmi_window_exits; svm->nmi_masked =3D false; kvm_make_request(KVM_REQ_EVENT, vcpu); return 1; diff --git a/arch/x86/kvm/vmx/vmx.c b/arch/x86/kvm/vmx/vmx.c index ee1d606e3314..cdb320169f5c 100644 --- a/arch/x86/kvm/vmx/vmx.c +++ b/arch/x86/kvm/vmx/vmx.c @@ -417,7 +417,7 @@ static noinstr void vmx_l1d_flush(struct kvm_vcpu *vcpu) kvm_clear_cpu_l1tf_flush_l1d(); } =20 - vcpu->stat->l1d_flush++; + vcpu->stat.l1d_flush++; =20 if (static_cpu_has(X86_FEATURE_FLUSH_L1D)) { native_wrmsrq(MSR_IA32_FLUSH_CMD, L1D_FLUSH); @@ -1399,7 +1399,7 @@ static void vmx_prepare_switch_to_host(struct vcpu_vm= x *vmx) =20 host_state =3D &vmx->loaded_vmcs->host_state; =20 - ++vmx->vcpu.stat->host_state_reload; + ++vmx->vcpu.stat.host_state_reload; =20 #ifdef CONFIG_X86_64 rdmsrq(MSR_KERNEL_GS_BASE, vmx->msr_guest_kernel_gs_base); @@ -5111,7 +5111,7 @@ void vmx_inject_irq(struct kvm_vcpu *vcpu, bool reinj= ected) =20 trace_kvm_inj_virq(irq, vcpu->arch.interrupt.soft, reinjected); =20 - ++vcpu->stat->irq_injections; + ++vcpu->stat.irq_injections; if (vmx->rmode.vm86_active) { int inc_eip =3D 0; if (vcpu->arch.interrupt.soft) @@ -5148,7 +5148,7 @@ void vmx_inject_nmi(struct kvm_vcpu *vcpu) vmx->loaded_vmcs->vnmi_blocked_time =3D 0; } =20 - ++vcpu->stat->nmi_injections; + ++vcpu->stat.nmi_injections; vmx->loaded_vmcs->nmi_known_unmasked =3D false; =20 if (vmx->rmode.vm86_active) { @@ -5560,7 +5560,7 @@ static int handle_exception_nmi(struct kvm_vcpu *vcpu) =20 static __always_inline int handle_external_interrupt(struct kvm_vcpu *vcpu) { - ++vcpu->stat->irq_exits; + ++vcpu->stat.irq_exits; return 1; } =20 @@ -5580,7 +5580,7 @@ static int handle_io(struct kvm_vcpu *vcpu) exit_qualification =3D vmx_get_exit_qual(vcpu); string =3D (exit_qualification & 16) !=3D 0; =20 - ++vcpu->stat->io_exits; + ++vcpu->stat.io_exits; =20 if (string) return kvm_emulate_instruction(vcpu, 0); @@ -5834,7 +5834,7 @@ static int handle_interrupt_window(struct kvm_vcpu *v= cpu) =20 kvm_make_request(KVM_REQ_EVENT, vcpu); =20 - ++vcpu->stat->irq_window_exits; + ++vcpu->stat.irq_window_exits; return 1; } =20 @@ -6012,7 +6012,7 @@ static int handle_nmi_window(struct kvm_vcpu *vcpu) return -EIO; =20 exec_controls_clearbit(to_vmx(vcpu), CPU_BASED_NMI_WINDOW_EXITING); - ++vcpu->stat->nmi_window_exits; + ++vcpu->stat.nmi_window_exits; kvm_make_request(KVM_REQ_EVENT, vcpu); =20 return 1; @@ -6276,7 +6276,7 @@ static int handle_notify(struct kvm_vcpu *vcpu) unsigned long exit_qual =3D vmx_get_exit_qual(vcpu); bool context_invalid =3D exit_qual & NOTIFY_VM_CONTEXT_INVALID; =20 - ++vcpu->stat->notify_window_exits; + ++vcpu->stat.notify_window_exits; =20 /* * Notify VM exit happened while executing iret from NMI, @@ -7637,7 +7637,7 @@ fastpath_t vmx_vcpu_run(struct kvm_vcpu *vcpu, u64 ru= n_flags) */ if (vcpu->arch.nested_run_pending && !vmx_get_exit_reason(vcpu).failed_vmentry) - ++vcpu->stat->nested_run; + ++vcpu->stat.nested_run; =20 vcpu->arch.nested_run_pending =3D 0; } diff --git a/arch/x86/kvm/xen.c b/arch/x86/kvm/xen.c index b66e292c80d6..4527f04c6617 100644 --- a/arch/x86/kvm/xen.c +++ b/arch/x86/kvm/xen.c @@ -625,8 +625,6 @@ void kvm_xen_inject_vcpu_vector(struct kvm_vcpu *v) irq.shorthand =3D APIC_DEST_NOSHORT; irq.delivery_mode =3D APIC_DM_FIXED; irq.level =3D 1; - irq.plane =3D v->plane; - kvm_irq_delivery_to_apic(v->plane, NULL, &irq); } =20 diff --git a/include/uapi/linux/kvm.h b/include/uapi/linux/kvm.h index 82189353ef35..dfc9c7dab21e 100644 --- a/include/uapi/linux/kvm.h +++ b/include/uapi/linux/kvm.h @@ -1686,6 +1686,8 @@ struct kvm_memory_attributes { }; =20 #define KVM_MEMORY_ATTRIBUTE_PRIVATE (1ULL << 3) +#define KVM_MEMORY_ATTRIBUTE_NO_WRITE (1ULL << 4) +#define KVM_MEMORY_ATTRIBUTE_NO_EXEC (1ULL << 5) =20 /* * Per-plane memory protection attributes (VM planes / VBS). diff --git a/virt/kvm/guest_memfd.c b/virt/kvm/guest_memfd.c index 229154e06cd0..db57c5766ab6 100644 --- a/virt/kvm/guest_memfd.c +++ b/virt/kvm/guest_memfd.c @@ -827,7 +827,6 @@ static long __kvm_gmem_populate(struct kvm *kvm, struct= kvm_memory_slot *slot, struct file *file, gfn_t gfn, struct page *src_page, kvm_gmem_populate_cb post_populate, void *opaque) { - struct kvm_plane *plane0 =3D kvm->planes[0]; pgoff_t index =3D kvm_gmem_get_index(slot, gfn); struct folio *folio; kvm_pfn_t pfn; @@ -843,7 +842,7 @@ static long __kvm_gmem_populate(struct kvm *kvm, struct= kvm_memory_slot *slot, =20 folio_unlock(folio); =20 - if (!kvm_range_has_memory_attributes(plane0, gfn, gfn + 1, + if (!kvm_range_has_memory_attributes(kvm, gfn, gfn + 1, KVM_MEMORY_ATTRIBUTE_PRIVATE, KVM_MEMORY_ATTRIBUTE_PRIVATE)) { ret =3D -EINVAL; --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 7527743B3F6; Wed, 5 Aug 2026 11:04:02 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927843; cv=none; b=A05d/PKTZDE028MLuA6wdFoWLSEmF1wLooxWNaEeC61PH4D0I6axYjoQi6i6RgcZYITPbvNHm23vQlBgQ7NPjnV2wCVdkRud/1oHzvmXzdYcNwz9RE33sg1fUrienP1UP+btp0V7Vn2NXzxsfoCzY7Cc1EOFS+wcpzFD5MLhXYo= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927843; c=relaxed/simple; bh=fd1fZGkKlRI0hsVdzhb2l4RtI1jIs/V9mpyhR9PBfEY=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=YJJBhkMiSLoIvq0wLEQulaXyjF93DBWD1QavLg22/i/l34cJUuyOPa9xePHIh5LPTMyoOADfbg8i3WUreuFJx/B2y5aIzYf7aTlp/GwErdJGWdL9dxfxK+Wq8dcPFEjHMEnylQhKvp5SIL0OdFh/UI9kt7vJWk8shdk+pseB1QE= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=T5DO/7Vt; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="T5DO/7Vt" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id C617D20B716A; Wed, 5 Aug 2026 04:03:41 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com C617D20B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927821; bh=67aOwmHOj/Q52K+V5Ek1yk9wbvv+XpwwjutWPwsyzDg=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=T5DO/7VtUathSgXOq+CnarGbw7kLvSYGcG/eiyvh26+fgo7zbwYm9s/4rPSKvKWe1 3KsN04n5qtQ+fzvEDEu14uGRZMz6Mx9wXkWcFFy0Hg96AL1Bw924uDzH68BYAEPIHa R7cBrXwOVShJWgHqJNmOBfoIYeIZtcLHLzeF6NiU= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 21/42] KVM: x86: exit VM planes and VBS hypercalls to userspace Date: Wed, 5 Aug 2026 04:03:03 -0700 Message-ID: <20260805110324.25067-22-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- arch/x86/kvm/x86.c | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index d4210053e6b8..00dcdd0e22a4 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -119,10 +119,11 @@ u64 __read_mostly efer_reserved_bits =3D ~((u64)(EFER= _SCE | EFER_LME | EFER_LMA)); static u64 __read_mostly efer_reserved_bits =3D ~((u64)EFER_SCE); #endif =20 -#define KVM_EXIT_HYPERCALL_VALID_MASK (BIT(KVM_HC_MAP_GPA_RANGE) | \ - BIT(KVM_HC_VM_PLANES_CONFIG) | \ - BIT(KVM_HC_VM_PLANES_ACTIVATE) | \ - BIT(KVM_HC_VBS_VTL_CALL)) +#define KVM_EXIT_HYPERCALL_VALID_MASK \ + ((1 << KVM_HC_MAP_GPA_RANGE) | \ + (1 << KVM_HC_VM_PLANES_CONFIG) | \ + (1 << KVM_HC_VM_PLANES_ACTIVATE) | \ + (1 << KVM_HC_VBS_VTL_CALL)) =20 #define KVM_CAP_PMU_VALID_MASK KVM_PMU_CAP_DISABLE =20 @@ -478,7 +479,7 @@ static unsigned int num_msr_based_features; =20 unsigned kvm_x86_default_max_planes(struct kvm *kvm) { - return 1; + return 2; } EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_x86_default_max_planes); =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id E2CFC446832; Wed, 5 Aug 2026 11:04:02 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927844; cv=none; b=LrSdDh7ShKw4H7CBO8j2RyPxxX4Idni3LWvAJOo2jXaNkKv2/o40/Qbw7JLhRrdgpEUe6GAB7SwGrHJ0nootIpLNbcnXnu3mV47Dzh8xWYJtiRXfDsp0mXksyr8Vm2yXh3JoVAjK154AnShHoIfCNudN4Mp4al0O228as4/XI38= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927844; c=relaxed/simple; bh=S6ow6u37Nze3gz7SP5x0FBpghH++jtM7x4giyEkyIlo=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=aAWBAtzCUwOMeTbgkVlY7x+E7tvMaegecKhieAg8qhsYsAMbTmC+gQLYZis6nNJ8VtT6jzwwmKo8PQ6OqF7euGoeW2ufNHUaAt0YnE3WjWjdYbXW6wkoDNqO+0XTc8ZOqQpK4NxKPl9SZADYnugxnDClWh8sIlQoCF8ivNrrhy8= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=TpGaLzwr; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="TpGaLzwr" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 396DA20B716B; Wed, 5 Aug 2026 04:03:42 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 396DA20B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927822; bh=grQj36QjcJA5uMMqT2x4FOxuTqkiGj6jowV9c9yKi9U=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=TpGaLzwrd/4GBHLJHYK9vQ72OQY3iu75uIj354S4IM8VTSOaJPCSuzAPWb8JPCIh9 olb/PPMSRB/Tvk+/mwDn9kPWlBGmMhk1kij9bZ6QmybY8/6AsagJMYi9E1X6K6H1rc MhzrGy+PBOOU6tnEWlRzeCQzfe4cX6YvNw5aprb0= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 22/42] kexec: block legacy kexec_load when VBS is active Date: Wed, 5 Aug 2026 04:03:04 -0700 Message-ID: <20260805110324.25067-23-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Legacy kexec_load accepts raw memory segments and bypasses the file-based VBS validation path. Reject non-crash usage when a VBS backend is registered to prevent untrusted payload staging. Crash dumps (KEXEC_ON_CRASH) are still permitted since they serve a different purpose and do not replace the running kernel. Returns -EKEYREJECTED so userspace can distinguish VBS policy denial from permission errors. --- kernel/kexec.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/kernel/kexec.c b/kernel/kexec.c index 90756dc6339b..049afe1e1f5d 100644 --- a/kernel/kexec.c +++ b/kernel/kexec.c @@ -16,6 +16,7 @@ #include #include #include +#include =20 #include "kexec_internal.h" =20 @@ -205,6 +206,15 @@ static inline int kexec_load_check(unsigned long nr_se= gments, int image_type =3D (flags & KEXEC_ON_CRASH) ? KEXEC_TYPE_CRASH : KEXEC_TYPE_DEFAULT; int result; + bool crash_kexec =3D !!(flags & KEXEC_ON_CRASH); + + /* + * Legacy kexec_load accepts raw memory segments and bypasses the + * file-based VBS validation path. Reject non-crash usage when VBS + * is active to prevent untrusted payload staging. + */ + if (vbs_available() && !crash_kexec) + return -EKEYREJECTED; =20 /* We only trust the superuser with rebooting the system. */ if (!kexec_load_permitted(image_type)) --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id A7104448BA2; Wed, 5 Aug 2026 11:04:03 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927845; cv=none; b=NalUMgLeC1hls1b/FTXv/z6r6cVLBetIGKMCOg27NIuolVJWJO8yV41losYD9rPn2HGJ1VtB/uXUnJiVcEvEARuSMgAugABOMty1pfhcvMQtsq8dFtMHG8l0lDF7rgQjzd9a4T4OD124Kptqq3RdWetSRZHPaVvCnB0CfOxeW5w= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927845; c=relaxed/simple; bh=gBFUIbOHk+eDSzs7cfdaQX/rG5t5JvZrEJ4GdSpcVpI=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=ggJgrRmMKG4ZUzysMGR3WsJv7KcNQs6ysJ5A10NovOO43LGS3fO7K7BkoJS/gYyvUlo6AmkfKVl42xKfY4I/TNv9YVd0Fp6Q8TWmq44EnX3CadqXWj9eux4HagDKv8Dn13+GlfnJZU1+4NGAp9z9dxe2i6V+MpMMPbx3lzmc3U0= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=Ov08LaRf; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="Ov08LaRf" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id AAC2C20B716C; Wed, 5 Aug 2026 04:03:42 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com AAC2C20B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927822; bh=xsyECdDxm8Pd/K3eGkOY89BjFdC98SR2phH0+4eP6n8=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=Ov08LaRftX1wiqBv+JCdeESgu2eVnNjEtJqK1gCqT2rdhAC+7g6FR/1vKv3lkA+Z1 jRDIMucJY6Mrc7L8AQz97wtxZbmWFb6YkdB+lRVoDBHsM7tN2a0Xo3oFLldOUUN4ZS Ftxls2G5AyXoxF2XO7niRQniM5VINujqntD2qPz8= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 23/42] kvm: x86: fix merged plane API/stat build regressions Date: Wed, 5 Aug 2026 04:03:05 -0700 Message-ID: <20260805110324.25067-24-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" --- include/uapi/linux/kvm.h | 23 ----------------------- 1 file changed, 23 deletions(-) diff --git a/include/uapi/linux/kvm.h b/include/uapi/linux/kvm.h index dfc9c7dab21e..348628c7b17e 100644 --- a/include/uapi/linux/kvm.h +++ b/include/uapi/linux/kvm.h @@ -1689,29 +1689,6 @@ struct kvm_memory_attributes { #define KVM_MEMORY_ATTRIBUTE_NO_WRITE (1ULL << 4) #define KVM_MEMORY_ATTRIBUTE_NO_EXEC (1ULL << 5) =20 -/* - * Per-plane memory protection attributes (VM planes / VBS). - * These control EPT R/W/X permissions enforced by the hypervisor on - * behalf of a higher-privilege plane (e.g., plane-1 restricting plane-0). - */ -#define KVM_MEMORY_ATTRIBUTE_NO_WRITE (1ULL << 4) -#define KVM_MEMORY_ATTRIBUTE_NO_EXEC (1ULL << 5) - -/* - * Set memory attributes on a specific plane's address space. - * Used by a higher-privilege plane to restrict a lower-privilege plane's - * EPT permissions (e.g., plane-1 making plane-0 kernel text read-only). - */ -struct kvm_plane_memory_attributes { - __u32 plane; /* target plane index */ - __u32 flags; /* must be 0 */ - __u64 address; /* GPA (page-aligned) */ - __u64 size; /* size in bytes (page-aligned) */ - __u64 attributes; /* KVM_MEMORY_ATTRIBUTE_NO_WRITE / NO_EXEC */ -}; - -#define KVM_SET_PLANE_MEMORY_ATTRIBUTES _IOW(KVMIO, 0xd6, struct kvm_plane= _memory_attributes) - #define KVM_CREATE_GUEST_MEMFD _IOWR(KVMIO, 0xd4, struct kvm_create_guest= _memfd) #define GUEST_MEMFD_FLAG_MMAP (1ULL << 0) #define GUEST_MEMFD_FLAG_INIT_SHARED (1ULL << 1) --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 50AFD449EAB; Wed, 5 Aug 2026 11:04:04 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927846; cv=none; b=VR7bOKQ3+cHQW6bKMkljPWSrEdD8OisR+lwtNXSgNGY/MHd6psqFnOu2TUGUVn+0fumAdCtVIF5nRDJIcpNV46v5FTqnmYViD+Idwmu21CS8+s9aSmyE7A62eODihFn3x++NVnwxYsJag9HWHrFPy3aINUGccsXqtCqxOoyLRzk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927846; c=relaxed/simple; bh=QNE1jXe9wLUbPbFsj3SZsanKd2JF8BapyQA3tvBij80=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=Kks9fDCpA7TUyIrGGO5bPB4AZJEVllmFAM0ernBAROlHbEKX54OojYCXglawDHZkfSD3hnZvnJu53Trhj/mQ175GPwEb/UbxtGufjr17zPUy6yQZSCBTA2ZxjKVj/wSqT+t4YmqDcPCwvnyM4eDK2g44Yiw6Njed/g80IpQczn0= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=izn8bY4p; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="izn8bY4p" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 4A1C920B716D; Wed, 5 Aug 2026 04:03:43 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 4A1C920B716D DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927823; bh=CkiEMfyYTCwYRnWSxWlgCszilANo9OcAYdfL66ziJaM=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=izn8bY4pZkTYwnfxHPAk8bon9vJEPLOMtvb9be+6RTtOhKiv6RmX0YSRjG0tB9MLi k0We94/1064cA9q4akZTC6zPU/DWLCqgCWiiwBzzHaCkmsMdyYT9f8wH8p4+iQ/BEV bSptqC/u1rTadOP2SO4zqnHbvfi5b/hiMkXvnZ+g= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 24/42] KVM: planes: expose memory-attribute setting to in-kernel callers Date: Wed, 5 Aug 2026 04:03:06 -0700 Message-ID: <20260805110324.25067-25-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" A higher-privilege plane needs to restrict a lower plane's access to guest memory by setting NO_WRITE / NO_EXEC EPT attributes (e.g. HEKI sealing plane-0 text/rodata). The enforcement already lives in kvm_plane_filter_pte_access(); wire up the set side: - advertise KVM_MEMORY_ATTRIBUTE_NO_WRITE / NO_EXEC from kvm_supported_mem_attributes() when CONFIG_VM_PLANES is enabled, so userspace and in-kernel callers know the attributes are available. - make kvm_vm_set_mem_attributes() non-static and declare it in kvm_host.h so an in-kernel secure-plane caller can apply attributes without going through the ioctl path. No functional change for non-plane builds. Signed-off-by: Sriram Nambakam --- include/linux/kvm_host.h | 2 ++ virt/kvm/kvm_main.c | 21 +++++++++++++++++---- 2 files changed, 19 insertions(+), 4 deletions(-) diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index e989b293a34a..82e557e66152 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -2703,6 +2703,8 @@ static inline unsigned long kvm_get_memory_attributes= (struct kvm *kvm, gfn_t gfn =20 bool kvm_range_has_memory_attributes(struct kvm *kvm, gfn_t start, gfn_t e= nd, unsigned long mask, unsigned long attrs); +int kvm_vm_set_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t end, + unsigned long attributes); bool kvm_arch_pre_set_memory_attributes(struct kvm *kvm, struct kvm_gfn_range *range); bool kvm_arch_post_set_memory_attributes(struct kvm *kvm, diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index f703545a7e80..9623ab8ebd9e 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -2603,10 +2603,23 @@ static int kvm_vm_ioctl_clear_dirty_log(struct kvm = *kvm, #ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES static u64 kvm_supported_mem_attributes(struct kvm *kvm) { + u64 attrs =3D 0; + if (!kvm || kvm_arch_has_private_mem(kvm)) - return KVM_MEMORY_ATTRIBUTE_PRIVATE; + attrs |=3D KVM_MEMORY_ATTRIBUTE_PRIVATE; =20 - return 0; +#ifdef CONFIG_VM_PLANES + /* + * Cross-plane EPT protection: a higher-privilege plane may restrict + * a lower plane's access via NO_WRITE / NO_EXEC (e.g. HEKI sealing + * plane-0 kernel text and rodata). The enforcement path lives in + * kvm_plane_filter_pte_access(); advertise the attributes here so + * KVM_SET_MEMORY_ATTRIBUTES accepts them. + */ + attrs |=3D KVM_MEMORY_ATTRIBUTE_NO_WRITE | KVM_MEMORY_ATTRIBUTE_NO_EXEC; +#endif + + return attrs; } =20 /* @@ -2716,8 +2729,8 @@ static bool kvm_pre_set_memory_attributes(struct kvm = *kvm, } =20 /* Set @attributes for the gfn range [@start, @end). */ -static int kvm_vm_set_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t e= nd, - unsigned long attributes) +int kvm_vm_set_mem_attributes(struct kvm *kvm, gfn_t start, gfn_t end, + unsigned long attributes) { struct kvm_mmu_notifier_range pre_set_range =3D { .start =3D start, --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id E8E5B44AB9C; Wed, 5 Aug 2026 11:04:04 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927846; cv=none; b=Af+bHsRP0GQWMA6Y5cb2ck1bHwFvhL8s/P20BetiElQGvnC5v+zsFWnNko21IGFu/8h4vFMWuk+TG09oEs7Zbr8xMSKOQJmlWqapSYa9cjznfbkfzDLzyzaRrtQ/chDEqSOK/wuha3o0acedjtUNu3I/jXwIJ30f4yjFJVknYFY= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927846; c=relaxed/simple; bh=LV39BhhTwMHsbhRAOxbAOb/hdcXDE/lnyj1zgA65Kus=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=CMq0T+L7XeI/Y6w8RThuVp5PoGmmg3tzs3hCh/9tuco2HeVvKlULQGkrD2ZLl9vM59pp2V5l/jnWf1Ea0qLx5NOkC6CF660tE7S5sqmyZVt+UwLVCx8Lh4rKYjPSkaz64Qqf9agOUUjZFWCebjptlUl/DgNoLAK1acPpOtWuF4k= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=O9jhX7nS; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="O9jhX7nS" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 2225A20B716A; Wed, 5 Aug 2026 04:03:44 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 2225A20B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927824; bh=mt1T9Dvgk+QbHOEIbidfgXuyLYVMdeZq2rT8kRKiGTs=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=O9jhX7nSrnEqRUppaIS0rhFWR/bxu36elJTyFkDP3TAXH6EPT4YS4/Er2oUojlHip 8UkSfVSjqNVcYyi4/NZJXbekDbzOYNvb5UU/3XfyAVF4qZEhuBAH8riYAF+KGGk0Jn e3txE/1T1N/aK2ehlNWvb08hMmNQB6lsXlkZX/9k= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 25/42] vm_planes: drop unused per-plane vcpu_count Date: Wed, 5 Aug 2026 04:03:07 -0700 Message-ID: <20260805110324.25067-26-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Under the in-kernel "Option B" secure-plane model the secure plane runs on plane-0's existing vCPU thread and is left stopped until the first VTL call, so a plane never owns a distinct set of vCPUs. The vcpu_count field in struct vm_plane_config (and the matching parse-state copy) is no longer consulted by anything in the kernel: drop the field, its VCPU_COUNT config key parsing, its zero-init and its presence check. No functional change; the value was already unused on the boot path. Signed-off-by: Sriram Nambakam --- include/linux/vm_planes.h | 1 - init/vm_planes.c | 11 ----------- 2 files changed, 12 deletions(-) diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index 47f05fa80039..1130557cf5aa 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -20,7 +20,6 @@ struct vm_plane_config { phys_addr_t load_offset; phys_addr_t memory_size; phys_addr_t entry_point; - unsigned int vcpu_count; unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; char cmdline[VM_PLANE_CMDLINE_MAX]; diff --git a/init/vm_planes.c b/init/vm_planes.c index 274c0015fe76..10c7facdb1af 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -25,7 +25,6 @@ static bool __initdata enable_vm_planes_requested; struct vm_plane_parse_state { phys_addr_t load_offset; phys_addr_t memory_size; - unsigned int vcpu_count; unsigned int kernel_format; char kernel[VM_PLANE_KERNEL_NAME_MAX]; char cmdline[VM_PLANE_CMDLINE_MAX]; @@ -275,14 +274,6 @@ static int __init parse_plane_cfg_line(const char *lin= e, size_t len, return 0; } =20 - if (!strcmp(key, "VCPU_COUNT")) { - if (parsed_u64 =3D=3D 0 || parsed_u64 > UINT_MAX) - return -EINVAL; - plane_cfg[plane_id].vcpu_count =3D (unsigned int)parsed_u64; - state[plane_id].vcpu_count =3D (unsigned int)parsed_u64; - return 0; - } - return -ENOENT; } =20 @@ -314,7 +305,6 @@ static int __init parse_vm_planes_kconfig(const char *b= uf, size_t len, for (i =3D 0; i < *plane_count; i++) { state[i].load_offset =3D VM_PLANES_UNSET_VALUE; state[i].memory_size =3D VM_PLANES_UNSET_VALUE; - state[i].vcpu_count =3D 0; state[i].kernel[0] =3D '\0'; state[i].cmdline[0] =3D '\0'; } @@ -336,7 +326,6 @@ static int __init parse_vm_planes_kconfig(const char *b= uf, size_t len, for (i =3D 1; i < *plane_count; i++) { if (state[i].load_offset =3D=3D VM_PLANES_UNSET_VALUE || state[i].memory_size =3D=3D VM_PLANES_UNSET_VALUE || - !state[i].vcpu_count || !state[i].kernel[0]) return -EINVAL; } --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id B6B9F44E043; Wed, 5 Aug 2026 11:04:05 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927847; cv=none; b=JIwnGm5UaFiJO7Gh0sO7nkF/apIL1ETVaRNh/Xvjui0+lGGq/Fr702T0N0LDCoV7CO/McrQUAj/ZONdwwUOp1HNRdHajCCiZSssrjxVd8ekEJQNTMEv0S84UYZZ58Dz6hdRtWOOpgnM2difeyK7AUob9gYUHExcd1/esWjrBoi4= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927847; c=relaxed/simple; bh=4dPfLjvwm2dCi1BpCMDr3OP9+uCo2/wqxW9zUls3bwQ=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=WybgL+mf6uQx3Xbx5IOIPFHZvyy6b/od/lg3W2zomQVuLy5CwdxUNL6VdnAb17FsYbrbjCkPL+FK1PzBlzYlVAtINhIRQdcYlHLT9gBzAUwV8U96rcR13b653CExyPq3yRDreuF4T7sgySGCu8u3RFCeVbU2YZgSy9lG3WmiJnk= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=McK/7jlH; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="McK/7jlH" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id B1C0820B7169; Wed, 5 Aug 2026 04:03:44 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com B1C0820B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927824; bh=YC/fxgTEyFQzMd6QeSrs/+l70c1o6pfZQI4vPOn5O4s=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=McK/7jlH+ZQ6BTAv7M9GKKYFCIpIGiF7OBWPc4hD5x/M/GkXhkX4mmnYkqfjFWx4s CU5Zz/4yLZKczsVWIQ3eyLVvMklhUE0QeKrbDt6xHRb8CImgbVJl7gs+fzafVaPZEv 7eWY0gzTQ4PGWTjAkGYE+XzZi/7wWTKrodZLGi3w= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 26/42] drivers/virt: add VBS secure-plane park loop Date: Wed, 5 Aug 2026 04:03:08 -0700 Message-ID: <20260805110324.25067-27-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Add a minimal, self-contained driver that lets an otherwise ordinary kernel act as the secure plane (plane >0) of a KVM VM-planes guest. When the "vbs_park" command-line option is present, a kernel thread hands control back to the normal plane via KVM_HC_VBS_VTL_RETURN and then services VTL calls from a shared calling area, acknowledging each as a no-op. This is deliberately independent of the full VBS/HEKI stack (CONFIG_VBS): it implements only the park/dispatch handshake so that any secure kernel (or a future SVSM) can act as plane 1. Gated behind CONFIG_VBS_PARK. Signed-off-by: Sriram Nambakam --- drivers/virt/Kconfig | 15 +++++ drivers/virt/Makefile | 1 + drivers/virt/vbs_park.c | 136 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 152 insertions(+) create mode 100644 drivers/virt/vbs_park.c diff --git a/drivers/virt/Kconfig b/drivers/virt/Kconfig index 52eb7e4ba71f..88e40eaba1c2 100644 --- a/drivers/virt/Kconfig +++ b/drivers/virt/Kconfig @@ -13,6 +13,21 @@ menuconfig VIRT_DRIVERS =20 if VIRT_DRIVERS =20 +config VBS_PARK + bool "KVM VM-planes secure-plane park loop" + depends on X86 && KVM_GUEST + help + Minimal in-kernel handler for the secure plane (plane >0) of a KVM + VM-planes guest. When enabled and the "vbs_park" kernel command-line + option is present, a kernel thread hands control back to the normal + plane via the KVM_HC_VBS_VTL_RETURN hypercall and then services VTL + calls from a shared calling area. + + This is independent of the full VBS/HEKI stack (CONFIG_VBS): it + implements only the park/dispatch handshake so that any secure kernel + can act as plane 1. Calls are acknowledged as no-ops. Say N unless + this kernel is used as a VM-planes secure plane. + config VMGENID tristate "Virtual Machine Generation ID driver" default y diff --git a/drivers/virt/Makefile b/drivers/virt/Makefile index f29901bd7820..fa91899a356d 100644 --- a/drivers/virt/Makefile +++ b/drivers/virt/Makefile @@ -5,6 +5,7 @@ =20 obj-$(CONFIG_FSL_HV_MANAGER) +=3D fsl_hypervisor.o obj-$(CONFIG_VMGENID) +=3D vmgenid.o +obj-$(CONFIG_VBS_PARK) +=3D vbs_park.o obj-y +=3D vboxguest/ =20 obj-$(CONFIG_NITRO_ENCLAVES) +=3D nitro_enclaves/ diff --git a/drivers/virt/vbs_park.c b/drivers/virt/vbs_park.c new file mode 100644 index 000000000000..fabb6beeea7b --- /dev/null +++ b/drivers/virt/vbs_park.c @@ -0,0 +1,136 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * vbs_park - minimal KVM VM-planes secure-plane park loop + * + * This provides only the secure-plane (plane >0) side of the VM-planes + * park/dispatch handshake so that an otherwise ordinary kernel can act as + * plane 1. It is deliberately independent of the full VBS/HEKI stack + * (CONFIG_VBS): it implements no security policy. Its single job is to h= and + * control back to the normal plane (plane 0) via the KVM_HC_VBS_VTL_RETURN + * hypercall and then service VTL calls from the shared calling area. + * + * Control flow (all within plane 0's single KVM_RUN; see + * arch/x86/kvm/x86.c __kvm_emulate_hypercall): + * + * plane 0 KVM plane 1 (her= e) + * ------- --- ------------= -- + * fill calling area + * HC_VBS_VTL_CALL(ca_gpa) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=96=B6 switch_plane =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=96=B6 resume in + * (RAX :=3D ca_gpa) vtl_retur= n() + * handle call_= id + * write ca->st= atus + * resume after VTL_CALL =E2=97=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80 switch_plane =E2=97=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= HC_VBS_VTL_RETURN + * + * Activated by the "vbs_park" kernel command-line option; without it this + * kernel boots normally and never parks. + */ + +#define pr_fmt(fmt) "vbs-park: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +/* + * Shared-memory calling area. MUST match struct vbs_kvm_ca in + * security/vbs/kvm_planes.c (the normal-plane <-> secure-plane wire ABI): + * + * [ call_pending | call_id | status | arg_size | resp_size | buffer ] + */ +struct vtl_ca { + __u8 call_pending; /* 1 while call is in flight */ + __u8 rsvd[3]; + __u32 call_id; /* request id (set by caller) */ + __s32 status; /* return code (set by responder) */ + __u32 arg_size; /* request payload size */ + __u32 resp_size; /* response payload size */ + __u8 buffer[]; /* request data in, response data out */ +} __packed; + +/* Set from the "vbs_park" kernel command-line option. */ +static bool vbs_park_active __ro_after_init; + +static int __init vbs_park_setup(char *str) +{ + vbs_park_active =3D true; + return 1; +} +__setup("vbs_park", vbs_park_setup); + +/* + * Park the secure plane and hand control back to the normal plane. On the + * next VTL call KVM resumes us here with the calling-area GPA in the + * hypercall return value (RAX). @status is carried for tracing only; the + * real result is already in the calling area. + */ +static u64 vtl_return(long status) +{ + return kvm_hypercall1(KVM_HC_VBS_VTL_RETURN, (unsigned long)status); +} + +static int vbs_park_fn(void *unused) +{ + long status =3D 0; + + pr_info("secure-plane park loop started\n"); + + for (;;) { + struct vtl_ca *ca; + u64 ca_gpa; + + /* Park; resume with the next request's calling-area GPA. */ + ca_gpa =3D vtl_return(status); + if (!ca_gpa) { + status =3D -EINVAL; + continue; + } + + ca =3D memremap(ca_gpa, PAGE_SIZE, MEMREMAP_WB); + if (!ca) { + pr_err_ratelimited("failed to map calling area 0x%llx\n", + ca_gpa); + status =3D -EFAULT; + continue; + } + + /* + * No security policy lives here: acknowledge the call as a + * no-op so the normal plane can make progress. Replace this + * with real handlers (or move plane 1 to a dedicated SVSM) to + * enforce actual VBS semantics. + */ + pr_info_ratelimited("VTL call id=3D0x%x arg_size=3D%u (no-op)\n", + ca->call_id, ca->arg_size); + ca->status =3D 0; + ca->resp_size =3D 0; + status =3D 0; + + memunmap(ca); + } + + return 0; +} + +static int __init vbs_park_init(void) +{ + struct task_struct *t; + + if (!vbs_park_active) + return 0; + + t =3D kthread_run(vbs_park_fn, NULL, "vbs-park"); + if (IS_ERR(t)) { + pr_err("failed to start park loop: %ld\n", PTR_ERR(t)); + return PTR_ERR(t); + } + + return 0; +} +late_initcall(vbs_park_init); --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 6300C4503FF; Wed, 5 Aug 2026 11:04:06 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927848; cv=none; b=ATdyR8trI1E3f+udL0iH7qonBAvDiuXHO8n4Ie2NjiHYxwlPYSnhc9O0ykimbEKzBNoj11rNeuNySI/phY1HTThmZj/640QOimt1l8EMWo3/YY4l3XZrMf/9946/SLgdEjQODAWZwgHC4Ch+BK6cdW1pQ6CJB7NOnpzgGLN+LVc= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927848; c=relaxed/simple; bh=kpkgZl0Tcw92oo543975qMU66f9hJ6QM9ee1afXBaws=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=fdCdXcgAtqmOPdhHooXaF+BbyW8RTrAgYZeaY3sPPjT7SJSEzp/P3Pu20t1l0cwQnV3eepjJBgyPKNKD7PXK9t8Q6z0GUnPMg8Z1sD06gmfbtpSH9F/oBFBqQZuN9USCEZpoV+MdNBjWSLm7l7hdAgKc07WFvnV61Gu9D1lnQF0= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=Cf3aD486; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="Cf3aD486" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 871C420B716B; Wed, 5 Aug 2026 04:03:45 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 871C420B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927825; bh=m4DXXDJSWrtuCFV6AD4mQm8ZCb5skNiU0wv/knEqjsA=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=Cf3aD486bf0MQTJDeii5vYS8j3k5l8EsuGlzg38ch2TmPmq/NtypvT9PxXDjT4QvO 8fU/Y57P59/iBcJ1AunXewWG9wJFwITnkNVEM1nmlOivyInmTvSEwOQcPTSdF/N4sO 94gYroOQvmAGaG7dtuA6y9K/CyMP0w0EnJvfSc2E= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 27/42] KVM: planes: add arch-neutral in-kernel plane switch helper Date: Wed, 5 Aug 2026 04:03:09 -0700 Message-ID: <20260805110324.25067-28-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Factor the in-kernel plane switch out of the SEV-SNP VMPL path into a generic kvm_vcpu_switch_plane(). Both vCPUs share the same vcpu->common, so the switch only validates the sibling relationship, flips the per-plane runnable/stopped state, and returns 1 to keep the caller inside KVM_RUN; the run loop then re-selects the target plane via kvm_vcpu_select_plane(). This is the common core shared by all secure-plane backends (SEV-SNP VMPL today, VBS/VTL next); vendor-specific state preparation stays in the caller. Convert __sev_snp_run_vmpl() to use it. Signed-off-by: Sriram Nambakam --- arch/x86/kvm/svm/sev.c | 13 ++++++------- include/linux/kvm_host.h | 1 + virt/kvm/kvm_main.c | 27 +++++++++++++++++++++++++++ 3 files changed, 34 insertions(+), 7 deletions(-) diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index 53e76d22eb08..b9b0bbb72394 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -4507,21 +4507,20 @@ static int __sev_snp_run_vmpl(struct vcpu_svm *svm,= unsigned int vmpl) { struct kvm_vcpu *vcpu =3D &svm->vcpu; struct kvm_vcpu *target =3D vcpu->common->vcpus[vmpl]; - struct vcpu_svm *target_svm =3D to_svm(target); + struct vcpu_svm *target_svm; =20 if (!target) return -EINVAL; =20 - /* Mark current plane as stopped so it is not selected */ + target_svm =3D to_svm(target); + + /* SEV-specific preparation for the target VMPL before switching. */ kvm_set_mp_state(target, KVM_MP_STATE_RUNNABLE); /* In case KVM_REQ_UPDATE_PROTECTED_GUEST_STATE is set - mark the new VMS= A as runnable */ target_svm->sev_es.snp_ap_runnable =3D true; - kvm_vcpu_set_plane_runnable(target); - kvm_vcpu_set_plane_stopped(vcpu); - - kvm_make_request(KVM_REQ_PLANE_RESCHED, vcpu); =20 - return 1; + /* Perform the arch-neutral in-kernel plane switch. */ + return kvm_vcpu_switch_plane(vcpu, target); } =20 static int sev_snp_run_vmpl(struct vcpu_svm *svm) diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index 82e557e66152..c6cf2b6c0076 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -455,6 +455,7 @@ struct kvm_vcpu { =20 void kvm_vcpu_set_plane_runnable(struct kvm_vcpu *vcpu); void kvm_vcpu_set_plane_stopped(struct kvm_vcpu *vcpu); +int kvm_vcpu_switch_plane(struct kvm_vcpu *vcpu, struct kvm_vcpu *target); struct kvm_vcpu *kvm_vcpu_select_plane(struct kvm_vcpu *vcpu); =20 static inline bool kvm_vcpu_wants_to_run(struct kvm_vcpu *vcpu) diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 9623ab8ebd9e..553c282500fd 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -5039,6 +5039,33 @@ void kvm_vcpu_set_plane_stopped(struct kvm_vcpu *vcp= u) } EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_vcpu_set_plane_stopped); =20 +/* + * Switch the logical CPU from the currently-running plane (@vcpu) to a si= bling + * plane (@target) without leaving KVM_RUN. Both vCPUs share the same + * vcpu->common, so this only flips the per-plane runnable/stopped state a= nd + * requests a plane reschedule; the run loop in kvm_arch_vcpu_ioctl_run() = then + * re-selects @target via kvm_vcpu_select_plane() and re-enters the guest. + * + * This is the arch-neutral core of the in-kernel plane switch shared by a= ll + * secure-plane backends (SEV-SNP VMPL, VBS/VTL on Intel and AMD, and, in = the + * future, Arm stage-2). Any vendor-specific state preparation must be do= ne by + * the caller before invoking this helper. + * + * Returns 1 to keep the caller inside KVM_RUN, or -EINVAL if @target is n= ot a + * valid sibling plane of @vcpu. + */ +int kvm_vcpu_switch_plane(struct kvm_vcpu *vcpu, struct kvm_vcpu *target) +{ + if (!target || target->common !=3D vcpu->common) + return -EINVAL; + + kvm_vcpu_set_plane_runnable(target); + kvm_vcpu_set_plane_stopped(vcpu); + + return 1; +} +EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_vcpu_switch_plane); + struct kvm_vcpu *kvm_vcpu_select_plane(struct kvm_vcpu *vcpu) { struct kvm_vcpu_common *common =3D vcpu->common; --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 23F72453A50; Wed, 5 Aug 2026 11:04:07 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927849; cv=none; b=jPZrzO1XkhBNR9Ibu+BesEqp8Lq03h0EZzPFRvdw2/WZVIDUl6W0Eag3CmAUqERlLzLZQ7A1WOk0txFujE20NCgjwNVbcypbT9Sbb3wg8h58fwwwDeDD9dGRokDjNH3GkFbMWv+7QkGbtuEp/SL4TxuvUfsN5FPBH61vk3xCpnc= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927849; c=relaxed/simple; bh=+fYJcgdArIlnXh9ktpBWsb3Jl/kG4rP5Z6JxbGMNUIc=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=IXxPMorya/vVBVGOCuvjB1gGj6ytTtAewiA/FvC3mxXuC6h1q9rZZP3v74OzkUowQPZJrnBW+P2vR5rxtvqDumg+uJHCOBOGWi4j1LtszZD0Q+EASDn38mE/UepSfg7r2Eh901+zcRNqS2IEYefcVPJVu8NMF5sISeoEp/6mQWc= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=E2CTWX4t; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="E2CTWX4t" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 2ACAE20B716C; Wed, 5 Aug 2026 04:03:46 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 2ACAE20B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927826; bh=sq9iQqfm/Xa43a8fE6xVfiPF9fdPsEklWW/dBFeNuZk=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=E2CTWX4tA2cg77Qcr620QOMzDuT1IlUfzNioIoTZLIlS04c+t+jFOEWTwcpmxdxKc k4MRKov2QioKPoFJmayBobFDHElgPbb1zWAmzvdKjMTrHwUb18w6X4uN+ZjoIxzB+y zJ6Au4CnjU3pTVf3cOzEo/wOXD8dlv0UtFjGqNkU= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 28/42] KVM: x86: add VBS VTL call/return and cross-plane set-mem-attrs hypercalls Date: Wed, 5 Aug 2026 04:03:10 -0700 Message-ID: <20260805110324.25067-29-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Add the in-kernel handling for the VBS secure-plane hypercalls so the plane switch happens without bouncing through userspace: - KVM_HC_VBS_VTL_CALL: the normal plane (plane 0) records the calling-area GPA and switches to the secure plane. While the secure plane is still booting the call is parked (vtl_call_pending) and delivered once the plane parks itself; once ready (vtl_plane_ready) the GPA is delivered directly via kvm_vcpu_switch_plane(). - KVM_HC_VBS_VTL_RETURN: the secure plane parks and hands control back to plane 0, marking itself ready and delivering any pending call. - KVM_HC_VBS_SET_MEM_ATTRS: the secure plane applies cross-plane EPT restrictions to a lower plane via kvm_vm_set_mem_attributes() (rejected from plane 0). Track the per-CPU bootstrap state (vtl_plane_ready, vtl_call_pending, vtl_call_ca) in kvm_vcpu_common and assign the new hypercall numbers KVM_HC_VBS_VTL_RETURN (16) and KVM_HC_VBS_SET_MEM_ATTRS (17). Signed-off-by: Sriram Nambakam --- arch/x86/kvm/x86.c | 139 +++++++++++++++++++++++++++++++++- include/linux/kvm_host.h | 17 +++++ include/uapi/linux/kvm_para.h | 2 + 3 files changed, 156 insertions(+), 2 deletions(-) diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index eb82dde62399..3c73ab1dcfe8 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -10588,9 +10588,145 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vc= pu, int cpl, vcpu->arch.complete_userspace_io =3D complete_hypercall; return 0; } + case KVM_HC_VBS_VTL_CALL: +#ifdef CONFIG_VM_PLANES + /* + * Runtime VBS/VTL call from the normal world (plane 0) into the + * secure plane. Serviced in-kernel by switching to the secure + * plane (plane 1) =E2=80=94 no userspace round trip. This is + * arch-neutral: it works for both Intel (VMX) and AMD (SVM), and + * mirrors the SEV-SNP in-kernel VMPL switch. a0 carries the + * guest-physical address of the shared calling area. + * + * Two cases: + * - Secure plane already booted and parked in its dispatch loop + * (vtl_plane_ready): deliver the calling-area GPA directly in + * RAX (its pending VTL return value) and switch to it. + * - Secure plane not booted yet (bootstrap): record the call as + * pending and switch to the secure plane so it boots; it will + * pick up the pending GPA when it reaches its first VTL return. + * + * If there is no secure plane configured at all, fall through to + * the userspace path so QEMU can service the call. + */ + if (vcpu->plane_level =3D=3D 0) { + struct kvm_vcpu_common *common =3D vcpu->common; + struct kvm_vcpu *secure =3D common->vcpus[1]; + + if (secure) { + common->vtl_call_ca =3D a0; + + if (common->vtl_plane_ready) { + /* Parked in vtl_return: deliver now. */ + kvm_rax_write(secure, a0); + common->vtl_call_pending =3D false; + } else { + /* Still booting: deliver on readiness. */ + common->vtl_call_pending =3D true; + } + + if (kvm_vcpu_switch_plane(vcpu, secure) =3D=3D 1) { + ret =3D 0; + goto out; + } + ret =3D -KVM_EINVAL; + goto out; + } + } +#endif /* CONFIG_VM_PLANES */ + goto vtl_userspace_exit; + case KVM_HC_VBS_VTL_RETURN: +#ifdef CONFIG_VM_PLANES + /* + * The secure plane (plane >0) hands control back to plane 0 + * in-kernel. This covers three situations: + * - Bootstrap "ready": the secure plane has just booted and is + * issuing its first VTL return to announce it is parked. + * - Normal completion: it has finished servicing a VTL call; + * the result is already in the shared calling area. + * - A call that arrived while the secure plane was still booting + * is now delivered (vtl_call_pending) by returning its + * calling-area GPA in RAX and keeping the secure plane running. + * a0 is an optional status carried for tracing only. + */ + if (vcpu->plane_level =3D=3D 0) { + ret =3D -KVM_EPERM; + goto out; + } else { + struct kvm_vcpu_common *common =3D vcpu->common; + + common->vtl_plane_ready =3D true; + + if (common->vtl_call_pending) { + /* + * Deliver the call that triggered the secure + * plane's boot: return its calling-area GPA and + * stay in the secure plane to service it. The + * GPA is delivered as this hypercall's return + * value (RAX) via the normal completion path; do + * not write RAX directly here, as the completion + * handler would overwrite it with hypercall.ret. + */ + common->vtl_call_pending =3D false; + ret =3D common->vtl_call_ca; + goto out; + } + + if (kvm_vcpu_switch_plane(vcpu, common->vcpus[0]) =3D=3D 1) { + ret =3D 0; + goto out; + } + } +#endif /* CONFIG_VM_PLANES */ + ret =3D -KVM_EINVAL; + goto out; + case KVM_HC_VBS_SET_MEM_ATTRS: +#if defined(CONFIG_VM_PLANES) && defined(CONFIG_KVM_GENERIC_MEMORY_ATTRIBU= TES) + /* + * The secure plane (plane >0) enforces EPT permissions on the + * normal plane's memory. It cannot issue the host + * KVM_SET_MEMORY_ATTRIBUTES ioctl, so it asks KVM to do it via + * this hypercall. Only a higher-privilege plane may call it. + * + * a0 =3D guest-physical address (page aligned) + * a1 =3D region size in bytes (page aligned) + * a2 =3D access bits to retain for lower planes: + * bit0 read (implicit), bit1 write, bit2 exec + * (matches VBS_MEM_READ/WRITE/EXEC) + */ + if (vcpu->plane_level =3D=3D 0) { + ret =3D -KVM_EPERM; + goto out; + } + + if (!PAGE_ALIGNED(a0) || !PAGE_ALIGNED(a1) || a1 =3D=3D 0 || + a0 + a1 < a0) { + ret =3D -KVM_EINVAL; + goto out; + } else { + unsigned long attrs =3D 0; + gfn_t start =3D a0 >> PAGE_SHIFT; + gfn_t end =3D (a0 + a1) >> PAGE_SHIFT; + + if (!(a2 & BIT(1))) + attrs |=3D KVM_MEMORY_ATTRIBUTE_NO_WRITE; + if (!(a2 & BIT(2))) + attrs |=3D KVM_MEMORY_ATTRIBUTE_NO_EXEC; + + if (kvm_vm_set_mem_attributes(vcpu->kvm, start, end, + attrs)) + ret =3D -KVM_EINVAL; + else + ret =3D 0; + goto out; + } +#else + ret =3D -KVM_ENOSYS; + goto out; +#endif /* CONFIG_VM_PLANES && CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES */ case KVM_HC_VM_PLANES_CONFIG: case KVM_HC_VM_PLANES_ACTIVATE: - case KVM_HC_VBS_VTL_CALL: { + vtl_userspace_exit: ret =3D -KVM_ENOSYS; if (!user_exit_on_hypercall(vcpu->kvm, nr)) break; @@ -10609,7 +10745,6 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vcpu= , int cpl, WARN_ON_ONCE(vcpu->run->hypercall.flags & KVM_EXIT_HYPERCALL_MBZ); vcpu->arch.complete_userspace_io =3D complete_hypercall; return 0; - } default: ret =3D -KVM_ENOSYS; break; diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index c6cf2b6c0076..f14d78fd8cd3 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -386,6 +386,23 @@ struct kvm_vcpu_common { =20 bool plane_switch; =20 +#ifdef CONFIG_VM_PLANES + /* + * VBS/VTL secure-plane bootstrap state (per logical CPU). + * + * @vtl_plane_ready: the secure plane has booted and parked itself in + * its dispatch loop (issued its first VTL return). + * @vtl_call_pending: a normal-plane VTL call has been registered but + * not yet delivered to the secure plane (used while + * the secure plane is still booting). + * @vtl_call_ca: guest-physical address of the pending call's + * shared calling area. + */ + bool vtl_plane_ready; + bool vtl_call_pending; + u64 vtl_call_ca; +#endif + struct kvm_vcpu_arch_common arch; }; =20 diff --git a/include/uapi/linux/kvm_para.h b/include/uapi/linux/kvm_para.h index 1703238952fb..eec4fce6b33a 100644 --- a/include/uapi/linux/kvm_para.h +++ b/include/uapi/linux/kvm_para.h @@ -33,6 +33,8 @@ #define KVM_HC_VM_PLANES_CONFIG 13 #define KVM_HC_VM_PLANES_ACTIVATE 14 #define KVM_HC_VBS_VTL_CALL 15 +#define KVM_HC_VBS_VTL_RETURN 16 +#define KVM_HC_VBS_SET_MEM_ATTRS 17 =20 /* * hypercalls use architecture specific --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id DC468431A23; Wed, 5 Aug 2026 11:04:07 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927849; cv=none; b=mXcwUZstLNFHV1kyeaDdDwMBjsXz325feyEctcuiAo7y9yU8zlhE8w56u2Z0zwv0ukH5INlvB0B+ruhT2XWGONjL6uVpWGI1rXfAzuzA/3W0VHaZNySrTr7Ji4twBHT3DvvVhk8pK1qRrcA6LIObXpmO/VX/lhw7KFDNfercY7g= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927849; c=relaxed/simple; bh=EKXx8RcCnMkNFoQA8erndqmfSfcs71xAzEYOPAn++tc=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=hlbUfTDx5IC9ANEXXOJ2dB5mpynwtkXFrMmwcszrypp3Kz46QddBmx4al90kXQlL8W6hk2CjiVl3tC0GS/Ipbm7jm/hLGIp25iYYpihlfbKnx4isj3096cv77nwxoQTuyTlO1cgojPrGD38kzkU9McGvjIcCM2FsVGrT6S7arA0= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=aWeEDzYA; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="aWeEDzYA" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id CDE9920B716A; Wed, 5 Aug 2026 04:03:46 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com CDE9920B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927826; bh=LCof4AolZjngCYSNX1ZFL4j1gUXtjHh6tmKkZT1PHOw=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=aWeEDzYA7b/eKCiXp9CVR2QoUQ8AyAIWEN8zIZTCTYPe7sFo7aC7CbPTYwK+qa7kv vWBY9ZtR260rTRQvgloQX7O0vM9T6kh9DcKj+Wr/7K4ZQHI/XVXjcfF+BGJn113Wv1 WfhGBxqfIoxMlZ1rSaSOg9Z5ly4YoBaR3oRojBFA= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 29/42] init/vm_planes: set up planes from rootfs_initcall and load ELF payloads Date: Wed, 5 Aug 2026 04:03:11 -0700 Message-ID: <20260805110324.25067-30-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Move plane setup out of start_kernel()/kernel_init_freeable() and into a self-registering rootfs_initcall. Link vm_planes.o after initramfs.o so populate_rootfs() has unpacked the initramfs (which carries the config-vm-planes file and the plane kernels) before arch_init_vm_planes() runs. arch_init_vm_planes() becomes static and no longer needs a declaration in vm_planes.h. Also load the plane kernels as ELF payloads, copying loadable segments into the reserved plane memory and zeroing the BSS, replacing the early_ioremap path with a plain io.h mapping. Signed-off-by: Sriram Nambakam --- include/linux/vm_planes.h | 1 - init/Kconfig | 5 ++- init/Makefile | 5 ++- init/main.c | 4 -- init/vm_planes.c | 88 +++++++++++++++++++++------------------ 5 files changed, 54 insertions(+), 49 deletions(-) diff --git a/include/linux/vm_planes.h b/include/linux/vm_planes.h index 1130557cf5aa..e33fa03d1d4a 100644 --- a/include/linux/vm_planes.h +++ b/include/linux/vm_planes.h @@ -25,7 +25,6 @@ struct vm_plane_config { char cmdline[VM_PLANE_CMDLINE_MAX]; }; =20 -void __init arch_init_vm_planes(void); int __init load_vm_plane_kernels(unsigned int plane_count, struct vm_plane_config *plane_cfg); =20 diff --git a/init/Kconfig b/init/Kconfig index 23d9cca334ba..5d76fe852376 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1720,8 +1720,9 @@ config VM_PLANES Enable hypervisor enabled multi-kernel support. =20 This allows processing the kernel command-line parameter - "enable-vm-planes" and, when requested, calling - arch_init_vm_planes() during start_kernel(). + "enable-vm-planes" and, when requested, setting up the configured + planes from a rootfs_initcall (after the initramfs is populated and + before device drivers and late_initcalls run). =20 The initrd config-vm-planes file is expected to provide per-plane entries for PLANE__KERNEL, PLANE__LOAD_OFFSET, and diff --git a/init/Makefile b/init/Makefile index 113133c8cdd7..f6ac312f1e98 100644 --- a/init/Makefile +++ b/init/Makefile @@ -6,12 +6,15 @@ ccflags-y :=3D -fno-function-sections -fno-data-sections =20 obj-y :=3D main.o version.o mounts.o -obj-y +=3D vm_planes.o ifneq ($(CONFIG_BLK_DEV_INITRD),y) obj-y +=3D noinitramfs.o else obj-$(CONFIG_BLK_DEV_INITRD) +=3D initramfs.o endif +# vm_planes.o must link AFTER initramfs.o so that, at rootfs_initcall leve= l, +# populate_rootfs() (which unpacks the initramfs) runs before +# arch_init_vm_planes() reads the plane config and kernels from the rootfs. +obj-y +=3D vm_planes.o obj-$(CONFIG_GENERIC_CALIBRATE_DELAY) +=3D calibrate.o obj-$(CONFIG_INITRAMFS_TEST) +=3D initramfs_test.o =20 diff --git a/init/main.c b/init/main.c index 1c779f6d60cc..be188e67c556 100644 --- a/init/main.c +++ b/init/main.c @@ -1663,10 +1663,6 @@ static noinline void __init kernel_init_freeable(voi= d) wait_for_initramfs(); console_on_rootfs(); =20 -#ifdef CONFIG_VM_PLANES - arch_init_vm_planes(); -#endif - /* * check if there is an early userspace init. If yes, let it do all * the work diff --git a/init/vm_planes.c b/init/vm_planes.c index 10c7facdb1af..64d5ff19a736 100644 --- a/init/vm_planes.c +++ b/init/vm_planes.c @@ -12,9 +12,9 @@ #include #include #include +#include #include #include -#include =20 #ifdef CONFIG_VM_PLANES static bool __initdata enable_vm_planes_requested; @@ -360,44 +360,29 @@ static int __init vm_planes_get_cfg(unsigned int *pla= ne_count, static int __init copy_to_early_mem(phys_addr_t dest, const void *src, unsigned long size) { - unsigned long slop, clen; - char *p; - - while (size) { - slop =3D offset_in_page(dest); - clen =3D size; - if (clen > PAGE_SIZE - slop) - clen =3D PAGE_SIZE - slop; - p =3D early_memremap(dest & PAGE_MASK, clen + slop); - if (!p) - return -ENOMEM; - memcpy(p + slop, src, clen); - early_memunmap(p, clen + slop); - dest +=3D clen; - src +=3D clen; - size -=3D clen; - } + void *p; + + if (!size) + return 0; + p =3D memremap(dest, size, MEMREMAP_WB); + if (!p) + return -ENOMEM; + memcpy(p, src, size); + memunmap(p); return 0; } =20 static int __init zero_early_mem(phys_addr_t dest, unsigned long size) { - unsigned long slop, clen; - char *p; - - while (size) { - slop =3D offset_in_page(dest); - clen =3D size; - if (clen > PAGE_SIZE - slop) - clen =3D PAGE_SIZE - slop; - p =3D early_memremap(dest & PAGE_MASK, clen + slop); - if (!p) - return -ENOMEM; - memset(p + slop, 0, clen); - early_memunmap(p, clen + slop); - dest +=3D clen; - size -=3D clen; - } + void *p; + + if (!size) + return 0; + p =3D memremap(dest, size, MEMREMAP_WB); + if (!p) + return -ENOMEM; + memset(p, 0, size); + memunmap(p); return 0; } =20 @@ -616,23 +601,41 @@ int __init __weak alloc_vm_planes(unsigned int plane_= count, int __init __weak activate_vm_planes(unsigned int plane_count, struct vm_plane_config *plane_cfg) { return -ENOSYS; } =20 -void __init arch_init_vm_planes(void) +/* + * Set up VM planes during boot. + * + * This must run after the initramfs is populated (it reads the plane conf= ig + * and plane kernels from the rootfs) and, crucially, *before* any consumer + * that issues a plane switch -- in particular the VBS backend init/seal, = and + * before any device driver, module, or userspace can touch a plane. A + * rootfs_initcall satisfies all of these: it runs immediately after + * populate_rootfs() (initramfs ready) and before every device_initcall and + * late_initcall. Because init/ links before security/, this also runs be= fore + * the VBS probe/HEKI rootfs_initcalls, so the secure plane vcpu exists by= the + * time the first VTL call is issued. + */ +static int __init arch_init_vm_planes(void) { unsigned int plane_count =3D VM_PLANES_DEFAULT_COUNT; struct vm_plane_config *plane_cfg; int ret; =20 if (!enable_vm_planes_requested) - return; + return 0; =20 - if (!kvm_para_available()) - return; + /* Ensure any asynchronous initramfs unpacking has completed. */ + wait_for_initramfs(); + + if (!kvm_para_available()) { + pr_info("vm_planes: KVM paravirt unavailable, skipping plane setup\n"); + return 0; + } =20 ret =3D vm_planes_get_cfg(&plane_count, &plane_cfg); if (ret) { pr_warn("vm_planes: failed to parse %s: %d\n", VM_PLANES_CONFIG_FILE, ret); - return; + return 0; } =20 pr_info("vm_planes: enabling %u planes (ids 0..%u)\n", @@ -641,18 +644,21 @@ void __init arch_init_vm_planes(void) ret =3D alloc_vm_planes(plane_count, plane_cfg); if (ret) { pr_err("vm_planes: failed to allocate planes: %d\n", ret); - return; + return 0; } =20 ret =3D load_vm_plane_kernels(plane_count, plane_cfg); if (ret) { pr_err("vm_planes: failed to load plane kernels: %d\n", ret); - return; + return 0; } =20 ret =3D activate_vm_planes(plane_count, plane_cfg); if (ret) pr_err("vm_planes: failed to activate planes: %d\n", ret); + + return 0; } +rootfs_initcall(arch_init_vm_planes); =20 #endif /* CONFIG_VM_PLANES */ --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 572844582C4; Wed, 5 Aug 2026 11:04:08 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927850; cv=none; b=MRxvoCYxA5lMSAb3KO/vO9y/8XYE7WJ3Re6/Bw8H5hrQJffh6jYoU4/P9XcvVi67jAYxmrcC6djTw6IM4t+tZoObKjZxqHrdW+qdiOzjc4zmWfURlMaZr1kMzKSYjitm092TwzGdE7CiH174qnPFWWDwlu8v+sOO+UzZSvCB6RM= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927850; c=relaxed/simple; bh=Fgr/Agm/bR5/iZn/emRL6wgpHMAmeE9jPpISZ1WSMzM=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=dxx6EWhovqJ2flt3AlS/AxTeKM5U81pEGebs/HCJn7zAqzbR/7HAUNjPDjoN9GxFYDW8MPr32UBna4V5Nnos7ZcMg29TSfRbHxoQHLxp2ZEzAFNPPmylClRA4laIv1IRIxdteJOt5V7hhB/6hdFqgKszwWJMRMlboyeAz/Shg00= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=abTvn7hI; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="abTvn7hI" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 8E69F20B7169; Wed, 5 Aug 2026 04:03:47 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 8E69F20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927827; bh=1iNMm6KdlSbQDTIO8LLHIJTEmbz79B7DNWaze903Dfw=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=abTvn7hIQkR3eQoT0aRGituxwLeZ95tOayeGZoUqujN5bVsf2jFviyGcYRU5HQ880 lFazccm0XUCJR7LvpHVy9UzUdjskAfMYEoF7OhqP4jawltegNc7tSay5RABdpfafD3 XPDXyVB74YAQqTSQ+cqinbxBsdQeWwrmbdD4tMys= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 30/42] security/vbs: run backend probe and HEKI seal at rootfs_initcall Date: Wed, 5 Aug 2026 04:03:12 -0700 Message-ID: <20260805110324.25067-31-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Move VBS backend probing (vbs_probe_init) from device_initcall and the HEKI kernel seal (vbs_heki_late_init) from late_initcall to rootfs_initcall, and link probe.o before core.o so the backend is registered before the seal runs. At this level the initramfs is unpacked and the VM planes have been set up (init/ links before security/), but device drivers, modules and userspace have not started yet, so the kernel is sealed before anything that could tamper with it runs. Signed-off-by: Sriram Nambakam --- security/vbs/Makefile | 5 ++++- security/vbs/core.c | 9 ++++++++- security/vbs/probe.c | 9 +++++---- 3 files changed, 17 insertions(+), 6 deletions(-) diff --git a/security/vbs/Makefile b/security/vbs/Makefile index e33052ccde2d..01e831e28ac7 100644 --- a/security/vbs/Makefile +++ b/security/vbs/Makefile @@ -1,6 +1,9 @@ # SPDX-License-Identifier: GPL-2.0-only obj-$(CONFIG_VBS) +=3D vbs.o -vbs-y :=3D core.o probe.o +# probe.o must link before core.o so that, at rootfs_initcall level, the +# backend is registered (vbs_probe_init) before the HEKI seal runs +# (vbs_heki_late_init in core.o). +vbs-y :=3D probe.o core.o =20 vbs-$(CONFIG_VBS_HEKI) +=3D heki.o obj-$(CONFIG_VBS_KVM_PLANES) +=3D kvm_planes.o diff --git a/security/vbs/core.c b/security/vbs/core.c index 16b5329964f9..1167026fc7d1 100644 --- a/security/vbs/core.c +++ b/security/vbs/core.c @@ -195,4 +195,11 @@ static int __init vbs_heki_late_init(void) =20 return 0; } -late_initcall(vbs_heki_late_init); +/* + * Run at rootfs_initcall level (after vbs_probe_init in probe.o, which li= nks + * first) so the kernel is sealed before any device driver, module, or + * userspace runs. The secure plane vcpu already exists by this point bec= ause + * arch_init_vm_planes() (init/, links before security/) ran earlier in the + * same initcall level. + */ +rootfs_initcall(vbs_heki_late_init); diff --git a/security/vbs/probe.c b/security/vbs/probe.c index 292f3663a996..14aa3d59310b 100644 --- a/security/vbs/probe.c +++ b/security/vbs/probe.c @@ -96,8 +96,9 @@ static int __init vbs_probe_init(void) } =20 /* - * Run at device_initcall level: platform detection (CPUID, MSRs, SMCCC) - * is complete by this point, but subsystems that consume VBS (module - * loading, HEKI) have not yet started. + * Run at rootfs_initcall level: platform detection (CPUID, MSRs, SMCCC) + * is complete by this point, the VM planes have been set up (init/ links + * before security/), and subsystems that consume VBS (module loading, HEK= I, + * device drivers, userspace) have not yet started. */ -device_initcall(vbs_probe_init); +rootfs_initcall(vbs_probe_init); --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id DBBCB459ADD; Wed, 5 Aug 2026 11:04:08 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927850; cv=none; b=VbopevkULLBPuh5N3PIOHM94kHZz4Ejc3HH/unBcmYofuqZO+GvWmQnJOM29a39dQNnlQqOwKInvgwa5gbdja2+HD7zrRHDpjAudlAeHOlO4xNctMU+3ptZyp0JH0Nu4x4/+ngXkB3taq1kz8RpgVfzoxdH5b8dkYBjMzOZxe9U= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927850; c=relaxed/simple; bh=ZWZtr/YxItABnDHd3G+/oX76N8nDR9WlNY/bBv8eJ4s=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=KsUxCZaB4pAJ5OcXLralcXmIp2Ziepg6BNw+ftjPtwv62XIDrzGu09r5oFjdz5rGv48fCATl/7SufHqfnUd2HVcVwpOdPrYp114Yj0oMuDdL7NL8jfvF+PYrGwc9b37dYJy0WcsJHhJlulWRzYmNz0etZNihoUbPblC143DgNWs= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=ddNAv6EP; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="ddNAv6EP" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 0B87220B716D; Wed, 5 Aug 2026 04:03:48 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 0B87220B716D DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927828; bh=2o5bvQY6ffcuSOPfk6NCrggFxJBplKNTmYmcX0G06So=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=ddNAv6EPbbXoWPkhl/gcJwz/AywpxwuTGtNIpjZkuwUPY6iz2ImKfH9VChoMWy/NN uIMebT6WVmSKBlgwLw/+7a6/Ng3Ka97NvJihnCGHaHEkRuOPE4NZYy6BP5FEWERItS 5zI973udU9G7HORd/0jCgpMImwiTV5AsV4bC2dpY= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 31/42] security/vbs: pin the VTL call hypercall to CPU0 Date: Wed, 5 Aug 2026 04:03:13 -0700 Message-ID: <20260805110324.25067-32-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable KVM switches planes per logical CPU, and the secure plane is a single in-guest kernel that boots only on CPU0's sibling (common->vcpus[1] of CPU0). A VTL call issued from any other CPU would switch that CPU's non-existent secure sibling and fail. Drive the hypercall through work_on_cpu(0, ...) so the calling area build and the plane switch always land on CPU0 regardless of the caller's CPU. The request/response marshalling moves into a kvm_vtl_call_ctx run on CPU0; the validation (calling-area present, arg size) stays on the caller. Signed-off-by: Sriram Nambakam --- security/vbs/kvm_planes.c | 67 ++++++++++++++++++++++++++++----------- 1 file changed, 48 insertions(+), 19 deletions(-) diff --git a/security/vbs/kvm_planes.c b/security/vbs/kvm_planes.c index 061163a4d303..c41ea7fdb472 100644 --- a/security/vbs/kvm_planes.c +++ b/security/vbs/kvm_planes.c @@ -25,6 +25,7 @@ #include #include #include +#include #include #include #include @@ -59,28 +60,34 @@ static void *kvm_ca_page; /* single calling-area page = */ =20 /* =E2=94=80=E2=94=80 low-level VTL call =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80 */ =20 -static int kvm_planes_vtl_call(enum vbs_call_id id, - const void *arg, size_t arg_size, - void *resp, size_t resp_size) +struct kvm_vtl_call_ctx { + enum vbs_call_id id; + const void *arg; + size_t arg_size; + void *resp; + size_t resp_size; +}; + +/* + * Issue the VTL call hypercall. MUST run on the BSP (CPU0): KVM switches + * planes per logical CPU (the secure sibling is common->vcpus[1] of the + * *calling* CPU), and the secure plane is a single in-guest kernel that + * boots only on CPU0's sibling. Driven via work_on_cpu() so the hypercall + * always lands on CPU0 regardless of the caller's CPU. + */ +static long kvm_planes_vtl_call_on_cpu(void *data) { - struct vbs_kvm_ca *ca; + struct kvm_vtl_call_ctx *ctx =3D data; + struct vbs_kvm_ca *ca =3D kvm_ca_page; long hc_ret; =20 - if (!kvm_ca_page) - return -ENOMEM; - - if (arg_size > VBS_CA_BUF_SIZE) - return -E2BIG; - - ca =3D kvm_ca_page; - /* Build request */ - ca->call_id =3D id; - ca->arg_size =3D arg_size; + ca->call_id =3D ctx->id; + ca->arg_size =3D ctx->arg_size; ca->status =3D 0; ca->resp_size =3D 0; - if (arg_size && arg) - memcpy(ca->buffer, arg, arg_size); + if (ctx->arg_size && ctx->arg) + memcpy(ca->buffer, ctx->arg, ctx->arg_size); ca->call_pending =3D 1; =20 /* Issue hypercall: pass physical address of the calling area */ @@ -97,14 +104,36 @@ static int kvm_planes_vtl_call(enum vbs_call_id id, return ca->status; =20 /* Read response from the same buffer */ - if (resp && resp_size && ca->resp_size) { - size_t copy =3D min_t(size_t, resp_size, ca->resp_size); + if (ctx->resp && ctx->resp_size && ca->resp_size) { + size_t copy =3D min_t(size_t, ctx->resp_size, ca->resp_size); =20 - memcpy(resp, ca->buffer, copy); + memcpy(ctx->resp, ca->buffer, copy); } return 0; } =20 +static int kvm_planes_vtl_call(enum vbs_call_id id, + const void *arg, size_t arg_size, + void *resp, size_t resp_size) +{ + struct kvm_vtl_call_ctx ctx =3D { + .id =3D id, + .arg =3D arg, + .arg_size =3D arg_size, + .resp =3D resp, + .resp_size =3D resp_size, + }; + + if (!kvm_ca_page) + return -ENOMEM; + + if (arg_size > VBS_CA_BUF_SIZE) + return -E2BIG; + + /* Pin the plane switch to CPU0's secure sibling (the only booted one). */ + return work_on_cpu(0, kvm_planes_vtl_call_on_cpu, &ctx); +} + /* =E2=94=80=E2=94=80 memory protection =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ =20 struct vbs_protect_args { --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id BB1DD44D014; Wed, 5 Aug 2026 11:04:09 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927851; cv=none; b=hUxykJaKoEQbir3zbxDv5iCydt5qdlZdJ5ouizSEe3qvUwPpNlEANJMsmzW3KKttq1ZNVIf6AZuYlz+JxouOc4vaLxQKzDU8+oNCC+qSM+6ZgGEpivFu//h4fJWyI8Ujgo5j3KSVEQieeYTVZhah/AKfs/AhuU2eXhV2JZxvf4I= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927851; c=relaxed/simple; bh=G+vUSvo/oC1dWxiI188iu8T0tlEnrBKSqZS4qPUU4mM=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=FWz4xNblZMNixRA+JEdOU1WIt4TnOn8FycqKvUx6RXbhWskcbSU3ClHPFRhHezg91W5K1vYfFVTXEeZE7t7455GdFRMHFdz1KcfkCPIdEgAyINl+J8rq53S/gV6xUGCHCwB5k8I2zZWddaFVxslMhFv489ONe6KQW4LNfQmgLw8= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=awelbRcU; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="awelbRcU" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id A1EB620B716B; Wed, 5 Aug 2026 04:03:48 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com A1EB620B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927828; bh=49pTtJaJB/MDKkTjz4La9OyPCRwD8GJr97cXGLVijQU=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=awelbRcU1/osbLX8fs8tPuAktv26CLD+tenLMZSW2q6rb7pOOfxuxr2lHKNVhxlwp XWvMxjqdijWpziRnlsujNNuwnyD/o5hKLsIqFmJ8w+5sGSOSJghxo+nlMQ774sQqkp NhsQs+M/xkUD1yqmnwSzSXdeV5NIzoF1SQ+OaV9s= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 32/42] security/vbs: add secure-plane monitor backend Date: Wed, 5 Aug 2026 04:03:14 -0700 Message-ID: <20260805110324.25067-33-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Add the secure-plane counterpart of the normal-plane kvm_planes backend. The same kernel image boots as both planes; when started as the secure plane (via the "vbs_secure_plane" command-line option) this monitor takes over, parks via KVM_HC_VBS_VTL_RETURN, and dispatches the VTL calls issued by the normal plane from the shared calling area. Unlike the minimal CONFIG_VBS_PARK stub, this backend is part of the full VBS stack and is meant to grow real handlers (seal, memory protection, attestation). Built with CONFIG_VBS_KVM_PLANES. Signed-off-by: Sriram Nambakam --- security/vbs/Makefile | 1 + security/vbs/secure_monitor.c | 266 ++++++++++++++++++++++++++++++++++ 2 files changed, 267 insertions(+) create mode 100644 security/vbs/secure_monitor.c diff --git a/security/vbs/Makefile b/security/vbs/Makefile index 01e831e28ac7..f24f31727a65 100644 --- a/security/vbs/Makefile +++ b/security/vbs/Makefile @@ -7,6 +7,7 @@ vbs-y :=3D probe.o core.o =20 vbs-$(CONFIG_VBS_HEKI) +=3D heki.o obj-$(CONFIG_VBS_KVM_PLANES) +=3D kvm_planes.o +obj-$(CONFIG_VBS_KVM_PLANES) +=3D secure_monitor.o obj-$(CONFIG_VBS_SEV_SNP) +=3D sev_snp.o obj-$(CONFIG_VBS_TDX) +=3D tdx.o obj-$(CONFIG_VBS_HV_VSM) +=3D hv_vsm.o diff --git a/security/vbs/secure_monitor.c b/security/vbs/secure_monitor.c new file mode 100644 index 000000000000..c1221ad5019b --- /dev/null +++ b/security/vbs/secure_monitor.c @@ -0,0 +1,266 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * VBS secure-plane monitor =E2=80=94 in-guest VTL dispatcher + * + * This is the secure-plane counterpart of the normal-plane kvm_planes + * backend. The SAME kernel image boots as both the normal plane and the + * secure plane; when booted as the secure plane (selected via the + * "vbs_secure_plane" kernel command-line option) this monitor takes over + * and services VTL calls issued by the normal plane. + * + * "Secure plane" is the highest-privilege plane of the VM (conventionally + * plane 1 / VTL1 / VMPL0, but a VM may have up to KVM_MAX_PLANES planes a= nd + * the index is not hard-coded here). "Normal plane" is the requesting, + * lower-privilege plane (conventionally plane 0). + * + * Control flow (all within the normal plane's single KVM_RUN, see + * arch/x86/kvm/x86.c ____kvm_emulate_hypercall): + * + * normal plane KVM secure plane + * ------------ --- ------------ + * fill calling area + * HC_VBS_VTL_CALL(ca_gpa) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=96=B6 switch_plane =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=96=B6 resume in + * (RAX :=3D ca_gpa) secmon_vt= l_return() + * dispatch(cal= l_id) + * write ca->st= atus + * resume after VTL_CALL =E2=97=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80 switch_plane =E2=97=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= HC_VBS_VTL_RETURN(status) + * read ca->status + * + * Because all planes of a VM share the same memslots (struct kvm_plane has + * no memslots of its own; they live in struct kvm), the secure plane sees + * the same guest-physical address space as the normal plane and can read + * the calling area and the GPAs referenced by each request directly. + */ + +#define pr_fmt(fmt) "vbs-secmon: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "heki.h" + +/* + * Shared-memory calling area =E2=80=94 must match struct vbs_kvm_ca in kv= m_planes.c + * (this is the normal-plane <-> secure-plane wire ABI). + * + * [ call_pending | call_id | status | arg_size | resp_size | buffer ] + */ +struct vbs_kvm_ca { + __u8 call_pending; /* 1 while call is in flight */ + __u8 rsvd[3]; + __u32 call_id; /* enum vbs_call_id (set by caller) */ + __s32 status; /* return code (set by responder) */ + __u32 arg_size; /* request payload size */ + __u32 resp_size; /* response payload size */ + __u8 buffer[]; /* request data in, response data out */ +} __packed; + +/* Set from the "vbs_secure_plane" kernel command-line option. */ +static bool secmon_active __ro_after_init; + +static int __init secmon_setup(char *str) +{ + secmon_active =3D true; + return 1; +} +__setup("vbs_secure_plane", secmon_setup); + +/* + * Park the secure plane and hand control back to the normal plane. On the + * next VTL call, KVM resumes us here with the calling-area GPA in the + * hypercall return value (RAX). @status is carried for tracing only; the + * real result is already in the calling area. + */ +static u64 secmon_vtl_return(long status) +{ + return kvm_hypercall1(KVM_HC_VBS_VTL_RETURN, (unsigned long)status); +} + +/* + * Apply EPT permissions on a normal-plane GPA range from the secure plane. + * + * The secure plane cannot issue the host KVM_SET_MEMORY_ATTRIBUTES ioctl, + * so it asks KVM to do it via the KVM_HC_VBS_SET_MEM_ATTRS hypercall, whi= ch + * KVM honours only for a higher-privilege plane. @perms carries the acce= ss + * bits the normal plane should retain (VBS_MEM_*); KVM translates a clear= ed + * write/exec bit into NO_WRITE / NO_EXEC memory attributes. + */ +static int secmon_apply_attrs(u64 gpa, u64 size, u32 perms) +{ + long ret; + + pr_debug("apply_attrs gpa=3D0x%llx size=3D0x%llx perms=3D%c%c%c\n", + gpa, size, + (perms & VBS_MEM_READ) ? 'r' : '-', + (perms & VBS_MEM_WRITE) ? 'w' : '-', + (perms & VBS_MEM_EXEC) ? 'x' : '-'); + + ret =3D kvm_hypercall3(KVM_HC_VBS_SET_MEM_ATTRS, gpa, size, perms); + if (ret) + return (int)ret; + + return 0; +} + +/* =E2=94=80=E2=94=80 per-call handlers =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ + +static int secmon_do_protect_memory(const void *arg, u32 arg_size) +{ + const struct vbs_protect_memory_req *r =3D arg; + + if (arg_size < sizeof(*r)) + return -EINVAL; + + return secmon_apply_attrs(r->gpa, r->size, r->perms); +} + +static int secmon_do_seal_kernel(const void *arg, u32 arg_size) +{ + const struct vbs_seal_kernel_req *r =3D arg; + int ret; + + if (arg_size < sizeof(*r)) + return -EINVAL; + + /* Kernel text: read + execute, no write. */ + ret =3D secmon_apply_attrs(r->text_gpa, r->text_size, + VBS_MEM_READ | VBS_MEM_EXEC); + if (ret) + return ret; + + /* Kernel rodata: read only, no write, no execute. */ + return secmon_apply_attrs(r->rodata_gpa, r->rodata_size, + VBS_MEM_READ); +} + +static int secmon_do_set_module_perms(const void *arg, u32 arg_size) +{ + const struct vbs_set_module_perms_req *hdr =3D arg; + const struct vbs_module_section *sec; + u32 i, n; + + if (arg_size < sizeof(*hdr)) + return -EINVAL; + + n =3D hdr->nr_sections; + if (arg_size < sizeof(*hdr) + n * sizeof(*sec)) + return -EINVAL; + + sec =3D (const struct vbs_module_section *)(hdr + 1); + for (i =3D 0; i < n; i++) { + int ret =3D secmon_apply_attrs(sec[i].gpa, sec[i].size, + sec[i].perms); + if (ret) + return ret; + } + + return 0; +} + +static long secmon_dispatch(u32 call_id, const void *arg, u32 arg_size, + u32 *resp_size) +{ + *resp_size =3D 0; + + switch (call_id) { + case VBS_CALL_INIT: + case VBS_CALL_SHUTDOWN: + return 0; + + case VBS_CALL_PROTECT_MEMORY: + return secmon_do_protect_memory(arg, arg_size); + case VBS_CALL_SEAL_KERNEL: + return secmon_do_seal_kernel(arg, arg_size); + + case VBS_CALL_SET_MODULE_PERMS: + return secmon_do_set_module_perms(arg, arg_size); + + /* + * Module/kexec validation and key management are acknowledged for + * now (mirroring the previous userspace dispatcher); real signature + * verification runs here in a later stage. + */ + case VBS_CALL_VALIDATE_MODULE: + case VBS_CALL_UNLOAD_MODULE: + case VBS_CALL_ADD_KEY: + case VBS_CALL_REVOKE_KEY: + case VBS_CALL_SEND_CERTS: + case VBS_CALL_KEXEC_VALIDATE: + case VBS_CALL_KEXEC_INVALIDATE: + return 0; + + default: + pr_warn_ratelimited("unknown call_id 0x%x\n", call_id); + return -ENOSYS; + } +} + +/* =E2=94=80=E2=94=80 monitor loop =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80 */ + +static int secmon_monitor_fn(void *unused) +{ + long status =3D 0; + + pr_info("secure monitor started\n"); + + for (;;) { + struct vbs_kvm_ca *ca; + u64 ca_gpa; + u32 call_id, arg_size, resp_size =3D 0; + + /* Park; resume with the next request's calling-area GPA. */ + ca_gpa =3D secmon_vtl_return(status); + if (!ca_gpa) { + status =3D -EINVAL; + continue; + } + + ca =3D memremap(ca_gpa, PAGE_SIZE, MEMREMAP_WB); + if (!ca) { + pr_err_ratelimited("failed to map calling area 0x%llx\n", + ca_gpa); + status =3D -EFAULT; + continue; + } + + call_id =3D ca->call_id; + arg_size =3D ca->arg_size; + if (arg_size > PAGE_SIZE - sizeof(*ca)) + arg_size =3D PAGE_SIZE - sizeof(*ca); + + status =3D secmon_dispatch(call_id, ca->buffer, arg_size, + &resp_size); + + ca->status =3D (s32)status; + ca->resp_size =3D resp_size; + + memunmap(ca); + } + + return 0; +} + +static int __init secmon_init(void) +{ + struct task_struct *t; + + if (!secmon_active) + return 0; + + t =3D kthread_run(secmon_monitor_fn, NULL, "vbs-secmon"); + if (IS_ERR(t)) { + pr_err("failed to start secure monitor: %ld\n", PTR_ERR(t)); + return PTR_ERR(t); + } + + return 0; +} +late_initcall(secmon_init); --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 01E654611C9; Wed, 5 Aug 2026 11:04:11 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927853; cv=none; b=uwWnABoJFEqJj9GhCaHUSHtiATjy6f0yY6EmL6TQi70A9lv7heOFigS25S0/6kykHN2G4d83qM6s8XN25iDL+Zk35sjmVJe+P65jlZbQaK/DEMY60evjAjSsLZFUYmY0/iepOG+SzU/MKISEdEKiNokkHY5wsOhl4cfAOEkYYuk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927853; c=relaxed/simple; bh=zvjJV8tV8BE05JUgc1FNynSS0+u5Qt0BznrK47FN39g=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=lqylFYLIF2XsqzDRwDg+DNtVt7vwxmtNZk5iwuOaWK30FAoBu8dV7PpzO15Ae4kuMOxRu/sLUXJqkFFbsEY84c1j6behS744bBLJb3cnOK2Nb2tXbU0mpX3Zyt0u0sBfRTGUpe+Pm/utRakUclRT8JQf80Xkf7Fys2/KnKZOAxY= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=qGrVX46l; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="qGrVX46l" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 9075020B716C; Wed, 5 Aug 2026 04:03:49 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 9075020B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927829; bh=bI49O7Gio+nXUuiVovapNuGTXo7eViF48ve8j7ZvVjM=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=qGrVX46lltvkMIYvrvJjjOesOgiErdHK0KSfU7RG8vgTO7nnPH6pvxNMqqSFBaXJ1 Qwk18H4HLMtLRtzg9BmcS8KAiYjAf+HUPNoTp1Rsh+uU1tlESsvRfcCHip7pF4km7V L2Vv7luez3Ct/vq1MRoz2AqdBPFNtIP3NRUYwWws= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 33/42] drivers/virt: rename VBS park loop to secure_monitor Date: Wed, 5 Aug 2026 04:03:15 -0700 Message-ID: <20260805110324.25067-34-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable Replace the secure-plane park loop (drivers/virt/vbs_park.c, CONFIG_VBS_PAR= K) with drivers/virt/secure_monitor.c (CONFIG_VBS_SECURE_MONITOR), activated by the "secure_monitor" kernel command-line option. Behaviour is unchanged: a late_initcall spawns the "vbs-secmon" kthread which hands control back to t= he normal plane via KVM_HC_VBS_VTL_RETURN and acknowledges VTL calls as no-ops; real per-call handlers are plumbed in incrementally. Also drop the unused security/vbs/secure_monitor.c, which was never wired i= nto the running secure-plane path (it keyed off "vbs_secure_plane" under CONFIG_VBS_KVM_PLANES but was never activated). --- drivers/virt/Kconfig | 16 +- drivers/virt/Makefile | 2 +- drivers/virt/{vbs_park.c =3D> secure_monitor.c} | 77 ++--- security/vbs/Makefile | 1 - security/vbs/secure_monitor.c | 266 ------------------ 5 files changed, 52 insertions(+), 310 deletions(-) rename drivers/virt/{vbs_park.c =3D> secure_monitor.c} (50%) delete mode 100644 security/vbs/secure_monitor.c diff --git a/drivers/virt/Kconfig b/drivers/virt/Kconfig index 88e40eaba1c2..5d964f124afe 100644 --- a/drivers/virt/Kconfig +++ b/drivers/virt/Kconfig @@ -13,20 +13,20 @@ menuconfig VIRT_DRIVERS =20 if VIRT_DRIVERS =20 -config VBS_PARK - bool "KVM VM-planes secure-plane park loop" +config VBS_SECURE_MONITOR + bool "KVM VM-planes secure-plane monitor" depends on X86 && KVM_GUEST help - Minimal in-kernel handler for the secure plane (plane >0) of a KVM - VM-planes guest. When enabled and the "vbs_park" kernel command-line + In-kernel monitor for the secure plane (plane >0) of a KVM VM-planes + guest. When enabled and the "secure_monitor" kernel command-line option is present, a kernel thread hands control back to the normal plane via the KVM_HC_VBS_VTL_RETURN hypercall and then services VTL calls from a shared calling area. =20 - This is independent of the full VBS/HEKI stack (CONFIG_VBS): it - implements only the park/dispatch handshake so that any secure kernel - can act as plane 1. Calls are acknowledged as no-ops. Say N unless - this kernel is used as a VM-planes secure plane. + This is independent of the full VBS/HEKI stack (CONFIG_VBS) so that + any secure kernel can act as plane 1. Per-call handlers are plumbed + in incrementally; until then calls are acknowledged as no-ops. Say N + unless this kernel is used as a VM-planes secure plane. =20 config VMGENID tristate "Virtual Machine Generation ID driver" diff --git a/drivers/virt/Makefile b/drivers/virt/Makefile index fa91899a356d..22d1121ba5bd 100644 --- a/drivers/virt/Makefile +++ b/drivers/virt/Makefile @@ -5,7 +5,7 @@ =20 obj-$(CONFIG_FSL_HV_MANAGER) +=3D fsl_hypervisor.o obj-$(CONFIG_VMGENID) +=3D vmgenid.o -obj-$(CONFIG_VBS_PARK) +=3D vbs_park.o +obj-$(CONFIG_VBS_SECURE_MONITOR) +=3D secure_monitor.o obj-y +=3D vboxguest/ =20 obj-$(CONFIG_NITRO_ENCLAVES) +=3D nitro_enclaves/ diff --git a/drivers/virt/vbs_park.c b/drivers/virt/secure_monitor.c similarity index 50% rename from drivers/virt/vbs_park.c rename to drivers/virt/secure_monitor.c index fabb6beeea7b..2d181c32c439 100644 --- a/drivers/virt/vbs_park.c +++ b/drivers/virt/secure_monitor.c @@ -1,31 +1,41 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * vbs_park - minimal KVM VM-planes secure-plane park loop + * secure_monitor - KVM VM-planes secure-plane monitor * - * This provides only the secure-plane (plane >0) side of the VM-planes - * park/dispatch handshake so that an otherwise ordinary kernel can act as - * plane 1. It is deliberately independent of the full VBS/HEKI stack - * (CONFIG_VBS): it implements no security policy. Its single job is to h= and - * control back to the normal plane (plane 0) via the KVM_HC_VBS_VTL_RETURN - * hypercall and then service VTL calls from the shared calling area. + * This is the secure-plane (plane >0) side of the VM-planes park/dispatch + * handshake. It lets an otherwise ordinary kernel act as the secure plane + * (conventionally plane 1 / VTL1 / VMPL0, though the index is not hard-co= ded) + * without pulling in the full VBS/HEKI stack (CONFIG_VBS). Its single jo= b is + * to hand control back to the normal plane (plane 0) via the + * KVM_HC_VBS_VTL_RETURN hypercall and then service VTL calls from the sha= red + * calling area. * * Control flow (all within plane 0's single KVM_RUN; see * arch/x86/kvm/x86.c __kvm_emulate_hypercall): * - * plane 0 KVM plane 1 (her= e) - * ------- --- ------------= -- + * normal plane KVM secure plane + * ------------ --- ------------ * fill calling area * HC_VBS_VTL_CALL(ca_gpa) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=96=B6 switch_plane =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=96=B6 resume in - * (RAX :=3D ca_gpa) vtl_retur= n() - * handle call_= id + * (RAX :=3D ca_gpa) secmon_vt= l_return() + * dispatch(cal= l_id) * write ca->st= atus * resume after VTL_CALL =E2=97=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80 switch_plane =E2=97=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= HC_VBS_VTL_RETURN * - * Activated by the "vbs_park" kernel command-line option; without it this - * kernel boots normally and never parks. + * Because all planes of a VM share the same memslots (struct kvm_plane ha= s no + * memslots of its own; they live in struct kvm), the secure plane sees the + * same guest-physical address space as the normal plane and can read the + * calling area and the GPAs referenced by each request directly. + * + * For now every VTL call is acknowledged as a no-op so the normal plane c= an + * make progress; the real per-call handlers (self-protection, HEKI memory + * protection, kernel sealing, =E2=80=A6) are plumbed in incrementally. + * + * Activated by the "secure_monitor" kernel command-line option; without it + * this kernel boots normally and never parks. */ =20 -#define pr_fmt(fmt) "vbs-park: " fmt +#define pr_fmt(fmt) "vbs-secmon: " fmt =20 #include #include @@ -44,7 +54,7 @@ * * [ call_pending | call_id | status | arg_size | resp_size | buffer ] */ -struct vtl_ca { +struct vbs_kvm_ca { __u8 call_pending; /* 1 while call is in flight */ __u8 rsvd[3]; __u32 call_id; /* request id (set by caller) */ @@ -54,15 +64,15 @@ struct vtl_ca { __u8 buffer[]; /* request data in, response data out */ } __packed; =20 -/* Set from the "vbs_park" kernel command-line option. */ -static bool vbs_park_active __ro_after_init; +/* Set from the "secure_monitor" kernel command-line option. */ +static bool secmon_active __ro_after_init; =20 -static int __init vbs_park_setup(char *str) +static int __init secmon_setup(char *str) { - vbs_park_active =3D true; + secmon_active =3D true; return 1; } -__setup("vbs_park", vbs_park_setup); +__setup("secure_monitor", secmon_setup); =20 /* * Park the secure plane and hand control back to the normal plane. On the @@ -70,23 +80,23 @@ __setup("vbs_park", vbs_park_setup); * hypercall return value (RAX). @status is carried for tracing only; the * real result is already in the calling area. */ -static u64 vtl_return(long status) +static u64 secmon_vtl_return(long status) { return kvm_hypercall1(KVM_HC_VBS_VTL_RETURN, (unsigned long)status); } =20 -static int vbs_park_fn(void *unused) +static int secmon_monitor_fn(void *unused) { long status =3D 0; =20 - pr_info("secure-plane park loop started\n"); + pr_info("secure monitor started\n"); =20 for (;;) { - struct vtl_ca *ca; + struct vbs_kvm_ca *ca; u64 ca_gpa; =20 /* Park; resume with the next request's calling-area GPA. */ - ca_gpa =3D vtl_return(status); + ca_gpa =3D secmon_vtl_return(status); if (!ca_gpa) { status =3D -EINVAL; continue; @@ -101,10 +111,9 @@ static int vbs_park_fn(void *unused) } =20 /* - * No security policy lives here: acknowledge the call as a - * no-op so the normal plane can make progress. Replace this - * with real handlers (or move plane 1 to a dedicated SVSM) to - * enforce actual VBS semantics. + * No handlers are plumbed in yet: acknowledge the call as a + * no-op so the normal plane can make progress. Real per-call + * dispatch is added incrementally. */ pr_info_ratelimited("VTL call id=3D0x%x arg_size=3D%u (no-op)\n", ca->call_id, ca->arg_size); @@ -118,19 +127,19 @@ static int vbs_park_fn(void *unused) return 0; } =20 -static int __init vbs_park_init(void) +static int __init secmon_init(void) { struct task_struct *t; =20 - if (!vbs_park_active) + if (!secmon_active) return 0; =20 - t =3D kthread_run(vbs_park_fn, NULL, "vbs-park"); + t =3D kthread_run(secmon_monitor_fn, NULL, "vbs-secmon"); if (IS_ERR(t)) { - pr_err("failed to start park loop: %ld\n", PTR_ERR(t)); + pr_err("failed to start secure monitor: %ld\n", PTR_ERR(t)); return PTR_ERR(t); } =20 return 0; } -late_initcall(vbs_park_init); +late_initcall(secmon_init); diff --git a/security/vbs/Makefile b/security/vbs/Makefile index f24f31727a65..01e831e28ac7 100644 --- a/security/vbs/Makefile +++ b/security/vbs/Makefile @@ -7,7 +7,6 @@ vbs-y :=3D probe.o core.o =20 vbs-$(CONFIG_VBS_HEKI) +=3D heki.o obj-$(CONFIG_VBS_KVM_PLANES) +=3D kvm_planes.o -obj-$(CONFIG_VBS_KVM_PLANES) +=3D secure_monitor.o obj-$(CONFIG_VBS_SEV_SNP) +=3D sev_snp.o obj-$(CONFIG_VBS_TDX) +=3D tdx.o obj-$(CONFIG_VBS_HV_VSM) +=3D hv_vsm.o diff --git a/security/vbs/secure_monitor.c b/security/vbs/secure_monitor.c deleted file mode 100644 index c1221ad5019b..000000000000 --- a/security/vbs/secure_monitor.c +++ /dev/null @@ -1,266 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -/* - * VBS secure-plane monitor =E2=80=94 in-guest VTL dispatcher - * - * This is the secure-plane counterpart of the normal-plane kvm_planes - * backend. The SAME kernel image boots as both the normal plane and the - * secure plane; when booted as the secure plane (selected via the - * "vbs_secure_plane" kernel command-line option) this monitor takes over - * and services VTL calls issued by the normal plane. - * - * "Secure plane" is the highest-privilege plane of the VM (conventionally - * plane 1 / VTL1 / VMPL0, but a VM may have up to KVM_MAX_PLANES planes a= nd - * the index is not hard-coded here). "Normal plane" is the requesting, - * lower-privilege plane (conventionally plane 0). - * - * Control flow (all within the normal plane's single KVM_RUN, see - * arch/x86/kvm/x86.c ____kvm_emulate_hypercall): - * - * normal plane KVM secure plane - * ------------ --- ------------ - * fill calling area - * HC_VBS_VTL_CALL(ca_gpa) =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=96=B6 switch_plane =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=96=B6 resume in - * (RAX :=3D ca_gpa) secmon_vt= l_return() - * dispatch(cal= l_id) - * write ca->st= atus - * resume after VTL_CALL =E2=97=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80 switch_plane =E2=97=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= HC_VBS_VTL_RETURN(status) - * read ca->status - * - * Because all planes of a VM share the same memslots (struct kvm_plane has - * no memslots of its own; they live in struct kvm), the secure plane sees - * the same guest-physical address space as the normal plane and can read - * the calling area and the GPAs referenced by each request directly. - */ - -#define pr_fmt(fmt) "vbs-secmon: " fmt - -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#include "heki.h" - -/* - * Shared-memory calling area =E2=80=94 must match struct vbs_kvm_ca in kv= m_planes.c - * (this is the normal-plane <-> secure-plane wire ABI). - * - * [ call_pending | call_id | status | arg_size | resp_size | buffer ] - */ -struct vbs_kvm_ca { - __u8 call_pending; /* 1 while call is in flight */ - __u8 rsvd[3]; - __u32 call_id; /* enum vbs_call_id (set by caller) */ - __s32 status; /* return code (set by responder) */ - __u32 arg_size; /* request payload size */ - __u32 resp_size; /* response payload size */ - __u8 buffer[]; /* request data in, response data out */ -} __packed; - -/* Set from the "vbs_secure_plane" kernel command-line option. */ -static bool secmon_active __ro_after_init; - -static int __init secmon_setup(char *str) -{ - secmon_active =3D true; - return 1; -} -__setup("vbs_secure_plane", secmon_setup); - -/* - * Park the secure plane and hand control back to the normal plane. On the - * next VTL call, KVM resumes us here with the calling-area GPA in the - * hypercall return value (RAX). @status is carried for tracing only; the - * real result is already in the calling area. - */ -static u64 secmon_vtl_return(long status) -{ - return kvm_hypercall1(KVM_HC_VBS_VTL_RETURN, (unsigned long)status); -} - -/* - * Apply EPT permissions on a normal-plane GPA range from the secure plane. - * - * The secure plane cannot issue the host KVM_SET_MEMORY_ATTRIBUTES ioctl, - * so it asks KVM to do it via the KVM_HC_VBS_SET_MEM_ATTRS hypercall, whi= ch - * KVM honours only for a higher-privilege plane. @perms carries the acce= ss - * bits the normal plane should retain (VBS_MEM_*); KVM translates a clear= ed - * write/exec bit into NO_WRITE / NO_EXEC memory attributes. - */ -static int secmon_apply_attrs(u64 gpa, u64 size, u32 perms) -{ - long ret; - - pr_debug("apply_attrs gpa=3D0x%llx size=3D0x%llx perms=3D%c%c%c\n", - gpa, size, - (perms & VBS_MEM_READ) ? 'r' : '-', - (perms & VBS_MEM_WRITE) ? 'w' : '-', - (perms & VBS_MEM_EXEC) ? 'x' : '-'); - - ret =3D kvm_hypercall3(KVM_HC_VBS_SET_MEM_ATTRS, gpa, size, perms); - if (ret) - return (int)ret; - - return 0; -} - -/* =E2=94=80=E2=94=80 per-call handlers =E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80 */ - -static int secmon_do_protect_memory(const void *arg, u32 arg_size) -{ - const struct vbs_protect_memory_req *r =3D arg; - - if (arg_size < sizeof(*r)) - return -EINVAL; - - return secmon_apply_attrs(r->gpa, r->size, r->perms); -} - -static int secmon_do_seal_kernel(const void *arg, u32 arg_size) -{ - const struct vbs_seal_kernel_req *r =3D arg; - int ret; - - if (arg_size < sizeof(*r)) - return -EINVAL; - - /* Kernel text: read + execute, no write. */ - ret =3D secmon_apply_attrs(r->text_gpa, r->text_size, - VBS_MEM_READ | VBS_MEM_EXEC); - if (ret) - return ret; - - /* Kernel rodata: read only, no write, no execute. */ - return secmon_apply_attrs(r->rodata_gpa, r->rodata_size, - VBS_MEM_READ); -} - -static int secmon_do_set_module_perms(const void *arg, u32 arg_size) -{ - const struct vbs_set_module_perms_req *hdr =3D arg; - const struct vbs_module_section *sec; - u32 i, n; - - if (arg_size < sizeof(*hdr)) - return -EINVAL; - - n =3D hdr->nr_sections; - if (arg_size < sizeof(*hdr) + n * sizeof(*sec)) - return -EINVAL; - - sec =3D (const struct vbs_module_section *)(hdr + 1); - for (i =3D 0; i < n; i++) { - int ret =3D secmon_apply_attrs(sec[i].gpa, sec[i].size, - sec[i].perms); - if (ret) - return ret; - } - - return 0; -} - -static long secmon_dispatch(u32 call_id, const void *arg, u32 arg_size, - u32 *resp_size) -{ - *resp_size =3D 0; - - switch (call_id) { - case VBS_CALL_INIT: - case VBS_CALL_SHUTDOWN: - return 0; - - case VBS_CALL_PROTECT_MEMORY: - return secmon_do_protect_memory(arg, arg_size); - case VBS_CALL_SEAL_KERNEL: - return secmon_do_seal_kernel(arg, arg_size); - - case VBS_CALL_SET_MODULE_PERMS: - return secmon_do_set_module_perms(arg, arg_size); - - /* - * Module/kexec validation and key management are acknowledged for - * now (mirroring the previous userspace dispatcher); real signature - * verification runs here in a later stage. - */ - case VBS_CALL_VALIDATE_MODULE: - case VBS_CALL_UNLOAD_MODULE: - case VBS_CALL_ADD_KEY: - case VBS_CALL_REVOKE_KEY: - case VBS_CALL_SEND_CERTS: - case VBS_CALL_KEXEC_VALIDATE: - case VBS_CALL_KEXEC_INVALIDATE: - return 0; - - default: - pr_warn_ratelimited("unknown call_id 0x%x\n", call_id); - return -ENOSYS; - } -} - -/* =E2=94=80=E2=94=80 monitor loop =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94= =80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80= =E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2=94=80=E2= =94=80 */ - -static int secmon_monitor_fn(void *unused) -{ - long status =3D 0; - - pr_info("secure monitor started\n"); - - for (;;) { - struct vbs_kvm_ca *ca; - u64 ca_gpa; - u32 call_id, arg_size, resp_size =3D 0; - - /* Park; resume with the next request's calling-area GPA. */ - ca_gpa =3D secmon_vtl_return(status); - if (!ca_gpa) { - status =3D -EINVAL; - continue; - } - - ca =3D memremap(ca_gpa, PAGE_SIZE, MEMREMAP_WB); - if (!ca) { - pr_err_ratelimited("failed to map calling area 0x%llx\n", - ca_gpa); - status =3D -EFAULT; - continue; - } - - call_id =3D ca->call_id; - arg_size =3D ca->arg_size; - if (arg_size > PAGE_SIZE - sizeof(*ca)) - arg_size =3D PAGE_SIZE - sizeof(*ca); - - status =3D secmon_dispatch(call_id, ca->buffer, arg_size, - &resp_size); - - ca->status =3D (s32)status; - ca->resp_size =3D resp_size; - - memunmap(ca); - } - - return 0; -} - -static int __init secmon_init(void) -{ - struct task_struct *t; - - if (!secmon_active) - return 0; - - t =3D kthread_run(secmon_monitor_fn, NULL, "vbs-secmon"); - if (IS_ERR(t)) { - pr_err("failed to start secure monitor: %ld\n", PTR_ERR(t)); - return PTR_ERR(t); - } - - return 0; -} -late_initcall(secmon_init); --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 240CF468C0B; Wed, 5 Aug 2026 11:04:12 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927854; cv=none; b=GOp60DOYu4Im4ZJTcZxj9QGhThdhVwqhkzBUczgZ+gLwHWq/K0dfXHLIX+4dRM3Z2FzNKgHia89OwzsRVRXF0XOulpWkk9SFmZFXBMZYaRttotFjJl1pEiphHIrU/XeUMkJbOQUeSWecEQ0l6knd5YJHeb3eaHVrAdSRLpMXVIk= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927854; c=relaxed/simple; bh=hp3inTy6vwa9u/UMgIvXdXMwZDjFfkMf6mB3v768YOU=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=tlt+AELtqV6XH62ZfjapG/PY+0UpER9/NRAmUMNOlOpzORfWLR7fo9mUHSK7ytKDkypoQQ4PCUf1GDlfMCTPziR/vL5+ag3x3g/vmiKzFYNWZhAFG9bx3+xPZ39vSdtl9IWcVPRLut+ue7/qe4A8Bn0qW/og4c2TL+dhrhgkR4s= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=bOqp6a9e; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="bOqp6a9e" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id C624520B7169; Wed, 5 Aug 2026 04:03:50 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com C624520B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927830; bh=RQx/+tS3Gguj0oQsRoaGqW2DUL7VawR8cXjvWahtnwE=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=bOqp6a9eBubODjXnHNNo2eaStXBqrL+wwVpv5ToXbO2JjG20yOvPCdt1l6fsPtnf6 vg/4UWdFR4XKfOK99EjjFyNgZRwjzPIu1jpo9fVFLEn7WueYm4YdEiP9CQE27sw0ZZ T4BqH/1kdeh/cGHoF3h7OVTljuzFN+Golll62xcg= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 34/42] x86/realmode: skip the sub-1M trampoline for the VBS secure plane Date: Wed, 5 Aug 2026 04:03:16 -0700 Message-ID: <20260805110324.25067-35-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" The VBS secure plane (plane >0) boots from a single high-memory region carved out of the normal plane's address space and therefore has no sub-1M RAM for the x86 real-mode AP trampoline. reserve_real_mode() followed by init_real_mode() then panics with "Real mode trampoline was not allocated". The secure plane is uniprocessor, enters directly in long mode and never uses the trampoline, so point x86_platform.realmode_reserve/realmode_init at x86_init_noop for it. This mirrors how the Hyper-V VTL (hv_vtl.c) and Xen PV ports disable the trampoline. Gated at compile time on CONFIG_VBS_SECURE_MONITOR (only the secure-plane kernel sets it) and at runtime on the "secure_monitor" early param (the normal plane never passes it), so plane 0 is unaffected. --- arch/x86/include/asm/kvm_host.h | 17 +++++-- arch/x86/kvm/mmu/mmu.c | 36 ++++++++++++++ arch/x86/kvm/mmu/spte.h | 12 +++-- arch/x86/kvm/x86.c | 85 +++++++++++++++++++++++++++++---- arch/x86/realmode/init.c | 26 ++++++++++ include/linux/kvm_host.h | 35 ++++++++++++++ include/uapi/linux/kvm.h | 1 + virt/kvm/kvm_main.c | 6 +++ 8 files changed, 202 insertions(+), 16 deletions(-) diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_hos= t.h index b7d478dcc1a5..bbccb9d3d801 100644 --- a/arch/x86/include/asm/kvm_host.h +++ b/arch/x86/include/asm/kvm_host.h @@ -380,15 +380,24 @@ union kvm_mmu_page_role { */ unsigned cr4_smep:1; =20 - unsigned:3; + /* + * Plane (privilege level) that owns this shadow page. VM + * planes share memslots but must have independent page + * tables so that a higher-privilege plane can restrict a + * lower plane's access (e.g. deny reads of secure-plane + * memory). Tagging the role keeps each plane's roots and + * SPTEs separate. Always 0 when CONFIG_VM_PLANES is off. + */ + unsigned plane:4; =20 /* * This is left at the top of the word so that * kvm_memslots_for_spte_role can extract it with a - * simple shift. While there is room, give it a whole - * byte so it is also faster to load it from memory. + * simple shift. smm is only ever used as a boolean, so it + * is reduced to 7 bits (from a full byte) to make room for + * cr4_smep and the VM-planes plane tag above. */ - unsigned smm:8; + unsigned smm:7; }; }; =20 diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c index 6e41c5df72ed..960e212c5ee3 100644 --- a/arch/x86/kvm/mmu/mmu.c +++ b/arch/x86/kvm/mmu/mmu.c @@ -4708,6 +4708,34 @@ static int kvm_mmu_faultin_pfn(struct kvm_vcpu *vcpu, return -EFAULT; } =20 + /* + * A higher-privilege plane may forbid this plane from accessing a gfn + * (e.g. to hide secure-plane memory from the normal plane). Two cases + * cannot be represented as a present SPTE and must be denied outright, + * exiting to userspace with a memory fault rather than (re)building an + * SPTE the access will immediately re-fault on: + * + * - NO_READ: there is no present-but-unreadable EPT entry, so leave the + * gfn unmapped for this plane. + * + * - NO_WRITE on a write fault: make_spte() strips ACC_WRITE_MASK and + * builds a read-only SPTE, so a guest write would re-fault forever + * (an unresolvable EPT write-violation livelock). Deny it instead so + * the violation is visible and can be mediated (e.g. HEKI text_poke + * is routed through the secure plane rather than written directly). + */ + { + unsigned long plane_attrs =3D + kvm_plane_access_attributes(vcpu->plane, fault->gfn); + + if ((plane_attrs & KVM_MEMORY_ATTRIBUTE_NO_READ) || + (fault->write && + (plane_attrs & KVM_MEMORY_ATTRIBUTE_NO_WRITE))) { + kvm_mmu_prepare_memory_fault_exit(vcpu, fault); + return -EFAULT; + } + } + if (unlikely(!slot)) return kvm_handle_noslot_fault(vcpu, fault, access); =20 @@ -5884,6 +5912,14 @@ kvm_calc_tdp_mmu_root_page_role(struct kvm_vcpu *vcp= u, role.direct =3D true; role.has_4_byte_gpte =3D false; =20 + /* + * Give each VM plane its own TDP root. Planes share memslots but + * need independent page tables so a higher-privilege plane can + * restrict a lower plane's access to a GFN. plane_level is 0 (and + * thus a no-op) on non-plane VMs and when CONFIG_VM_PLANES is off. + */ + role.plane =3D vcpu->plane_level; + /* All TDP pages are supervisor-executable */ role.access =3D ACC_ALL; if (role.cr4_smep && shadow_user_mask) diff --git a/arch/x86/kvm/mmu/spte.h b/arch/x86/kvm/mmu/spte.h index 144f7c5a1040..ed03bcdbf82d 100644 --- a/arch/x86/kvm/mmu/spte.h +++ b/arch/x86/kvm/mmu/spte.h @@ -580,9 +580,13 @@ void __init kvm_mmu_spte_module_init(void); void kvm_mmu_reset_all_pte_masks(void); =20 /* - * Apply memory protection attributes to pte_access. - * If memory attributes have NO_WRITE or NO_EXEC set for a GFN, - * strip the corresponding access bits before building the SPTE. + * Apply cross-plane access restrictions to pte_access when building an SP= TE + * for the faulting plane. A higher-privilege plane may downgrade a lower + * plane's access to a GFN via its per-plane access_attr_array. NO_WRITE = and + * NO_EXEC are enforced here by stripping the corresponding access bits. + * NO_READ cannot be expressed as a present-but-unreadable SPTE on EPT, so= it + * is enforced earlier in the fault handler (kvm_mmu_faultin_pfn) by refus= ing + * to map the page. */ #ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES static inline unsigned int kvm_plane_filter_pte_access(struct kvm_vcpu *vc= pu, @@ -591,7 +595,7 @@ static inline unsigned int kvm_plane_filter_pte_access(= struct kvm_vcpu *vcpu, { unsigned long attrs; =20 - attrs =3D kvm_get_memory_attributes(vcpu->kvm, gfn); + attrs =3D kvm_plane_access_attributes(vcpu->plane, gfn); if (attrs & KVM_MEMORY_ATTRIBUTE_NO_WRITE) pte_access &=3D ~ACC_WRITE_MASK; if (attrs & KVM_MEMORY_ATTRIBUTE_NO_EXEC) diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index 3c73ab1dcfe8..c8c37d569023 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -10494,6 +10494,61 @@ static int complete_hypercall_exit(struct kvm_vcpu= *vcpu) return kvm_skip_emulated_instruction(vcpu); } =20 +#if defined(CONFIG_VM_PLANES) && defined(CONFIG_KVM_GENERIC_MEMORY_ATTRIBU= TES) +/* + * Apply cross-plane access restrictions requested by a higher-privilege p= lane. + * Stores @attrs (NO_READ/NO_WRITE/NO_EXEC) for [@start, @end) in @plane's + * access_attr_array and zaps the range so any pages already mapped in @pl= ane's + * EPT re-fault and pick up the restriction. @attrs =3D=3D 0 clears the + * restriction for the range. + */ +static int kvm_plane_set_access_attrs(struct kvm *kvm, struct kvm_plane *p= lane, + gfn_t start, gfn_t end, unsigned long attrs) +{ + void *entry =3D attrs ? xa_mk_value(attrs) : NULL; + gfn_t gfn; + int r =3D 0; + + mutex_lock(&kvm->slots_lock); + + /* + * Reserve slots up front so the store loop below cannot fail partway + * through and leave a gap (a still-readable page) in the protected + * range. Clearing a restriction (entry =3D=3D NULL) never allocates. + */ + if (entry) { + for (gfn =3D start; gfn < end; gfn++) { + r =3D xa_reserve(&plane->access_attr_array, gfn, + GFP_KERNEL_ACCOUNT); + if (r) + goto out_unlock; + + cond_resched(); + } + } + + for (gfn =3D start; gfn < end; gfn++) { + r =3D xa_err(xa_store(&plane->access_attr_array, gfn, entry, + GFP_KERNEL_ACCOUNT)); + if (KVM_BUG_ON(r, kvm)) + goto out_unlock; + + cond_resched(); + } + + /* + * Re-fault the affected gfns in the plane's EPT so the new restriction + * takes effect on existing mappings. Zapping all roots is harmless; + * other planes simply rebuild identical entries on next access. + */ + kvm_zap_gfn_range(kvm, start, end); + +out_unlock: + mutex_unlock(&kvm->slots_lock); + return r; +} +#endif /* CONFIG_VM_PLANES && CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES */ + int ____kvm_emulate_hypercall(struct kvm_vcpu *vcpu, int cpl, int (*complete_hypercall)(struct kvm_vcpu *)) { @@ -10683,16 +10738,21 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vc= pu, int cpl, case KVM_HC_VBS_SET_MEM_ATTRS: #if defined(CONFIG_VM_PLANES) && defined(CONFIG_KVM_GENERIC_MEMORY_ATTRIBU= TES) /* - * The secure plane (plane >0) enforces EPT permissions on the - * normal plane's memory. It cannot issue the host + * The secure plane (plane >0) enforces EPT permissions on a + * lower plane's memory. It cannot issue the host * KVM_SET_MEMORY_ATTRIBUTES ioctl, so it asks KVM to do it via - * this hypercall. Only a higher-privilege plane may call it. + * this hypercall. Only a higher-privilege plane may call it; + * the restriction is applied to the plane directly below the + * caller. * * a0 =3D guest-physical address (page aligned) * a1 =3D region size in bytes (page aligned) - * a2 =3D access bits to retain for lower planes: - * bit0 read (implicit), bit1 write, bit2 exec - * (matches VBS_MEM_READ/WRITE/EXEC) + * a2 =3D access bits to retain for the lower plane: + * bit0 read, bit1 write, bit2 exec + * (matches VBS_MEM_READ/WRITE/EXEC). A cleared bit adds + * the corresponding NO_READ/NO_WRITE/NO_EXEC restriction; + * a2 =3D 0 hides the range entirely (e.g. secure-plane + * memory that the normal plane must not read). */ if (vcpu->plane_level =3D=3D 0) { ret =3D -KVM_EPERM; @@ -10704,17 +10764,26 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vc= pu, int cpl, ret =3D -KVM_EINVAL; goto out; } else { + struct kvm_plane *target; unsigned long attrs =3D 0; gfn_t start =3D a0 >> PAGE_SHIFT; gfn_t end =3D (a0 + a1) >> PAGE_SHIFT; =20 + target =3D vcpu->kvm->planes[vcpu->plane_level - 1]; + if (!target) { + ret =3D -KVM_EINVAL; + goto out; + } + + if (!(a2 & BIT(0))) + attrs |=3D KVM_MEMORY_ATTRIBUTE_NO_READ; if (!(a2 & BIT(1))) attrs |=3D KVM_MEMORY_ATTRIBUTE_NO_WRITE; if (!(a2 & BIT(2))) attrs |=3D KVM_MEMORY_ATTRIBUTE_NO_EXEC; =20 - if (kvm_vm_set_mem_attributes(vcpu->kvm, start, end, - attrs)) + if (kvm_plane_set_access_attrs(vcpu->kvm, target, start, + end, attrs)) ret =3D -KVM_EINVAL; else ret =3D 0; diff --git a/arch/x86/realmode/init.c b/arch/x86/realmode/init.c index 694d80a5c68e..01855a913b10 100644 --- a/arch/x86/realmode/init.c +++ b/arch/x86/realmode/init.c @@ -11,6 +11,7 @@ #include #include #include +#include =20 struct real_mode_header *real_mode_header; u32 *trampoline_cr4_features; @@ -44,6 +45,31 @@ void load_trampoline_pgtable(void) __flush_tlb_all(); } =20 +#ifdef CONFIG_VBS_SECURE_MONITOR +/* + * A KVM VM-planes secure plane (plane > 0) is entered directly in 64-bit = long + * mode and boots from a single carved-out high-memory region that contain= s no + * RAM below 1 MiB. It runs uniprocessor with no firmware, ACPI sleep, or + * hibernation, so the 16-bit real-mode trampoline can neither be allocated + * (there is no sub-1M memory) nor is it ever used (no AP bringup or wakeu= p). + * + * Disable the real-mode setup the same way Hyper-V VTL and Xen PV do, by + * pointing the x86_platform real-mode hooks at the no-op handler. This is + * installed from an early_param so it takes effect before setup_arch() ca= lls + * x86_platform.realmode_reserve(). Triggered by the "secure_monitor" + * command-line option, the same switch that activates the in-kernel + * secure-plane monitor. + */ +static int __init secure_plane_no_real_mode(char *arg) +{ + x86_platform.realmode_reserve =3D x86_init_noop; + x86_platform.realmode_init =3D x86_init_noop; + pr_info("realmode: secure plane: skipping sub-1M trampoline\n"); + return 0; +} +early_param("secure_monitor", secure_plane_no_real_mode); +#endif /* CONFIG_VBS_SECURE_MONITOR */ + void __init reserve_real_mode(void) { phys_addr_t mem, limit =3D x86_init.resources.realmode_limit; diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index f14d78fd8cd3..05c9edd4a73d 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -895,6 +895,18 @@ struct kvm_plane { /* Per-Plane VCPU array */ struct xarray vcpu_array; =20 +#ifdef CONFIG_VM_PLANES + /* + * Cross-plane access restrictions imposed on THIS plane by a + * higher-privilege plane. Each entry holds NO_READ/NO_WRITE/NO_EXEC + * bits for a gfn and is enforced when building this plane's SPTEs + * (planes have independent EPT roots). Distinct from + * kvm->mem_attr_array, which holds VM-wide PRIVATE/CoCo attributes. + * Protected by kvm->slots_lock for writes, RCU for reads. + */ + struct xarray access_attr_array; +#endif + struct kvm_arch_plane arch; }; =20 @@ -2739,6 +2751,29 @@ static inline bool kvm_mem_is_private(struct kvm *kv= m, gfn_t gfn) } #endif /* CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES */ =20 +#ifdef CONFIG_VM_PLANES +/* + * Cross-plane access restrictions: a higher-privilege plane downgrades a + * lower plane's access (NO_READ/NO_WRITE/NO_EXEC) to a gfn by storing bit= s in + * that lower plane's access_attr_array. Enforced when building the lower + * plane's SPTEs (planes have independent EPT roots). Returns 0 when no + * restriction applies. + */ +static inline unsigned long kvm_plane_access_attributes(struct kvm_plane *= plane, + gfn_t gfn) +{ + if (!plane) + return 0; + return xa_to_value(xa_load(&plane->access_attr_array, gfn)); +} +#else +static inline unsigned long kvm_plane_access_attributes(struct kvm_plane *= plane, + gfn_t gfn) +{ + return 0; +} +#endif /* CONFIG_VM_PLANES */ + #ifdef CONFIG_KVM_GUEST_MEMFD int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot, gfn_t gfn, kvm_pfn_t *pfn, struct page **page, diff --git a/include/uapi/linux/kvm.h b/include/uapi/linux/kvm.h index 348628c7b17e..3118b31d13f6 100644 --- a/include/uapi/linux/kvm.h +++ b/include/uapi/linux/kvm.h @@ -1688,6 +1688,7 @@ struct kvm_memory_attributes { #define KVM_MEMORY_ATTRIBUTE_PRIVATE (1ULL << 3) #define KVM_MEMORY_ATTRIBUTE_NO_WRITE (1ULL << 4) #define KVM_MEMORY_ATTRIBUTE_NO_EXEC (1ULL << 5) +#define KVM_MEMORY_ATTRIBUTE_NO_READ (1ULL << 6) =20 #define KVM_CREATE_GUEST_MEMFD _IOWR(KVMIO, 0xd4, struct kvm_create_guest= _memfd) #define GUEST_MEMFD_FLAG_MMAP (1ULL << 0) diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 553c282500fd..3a1a09f26340 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -1236,6 +1236,9 @@ static struct kvm_plane *kvm_create_plane(struct kvm = *kvm, unsigned plane_level) plane->level =3D plane_level; =20 xa_init(&plane->vcpu_array); +#ifdef CONFIG_VM_PLANES + xa_init(&plane->access_attr_array); +#endif =20 if (kvm_arch_plane_init(kvm, plane, plane_level)) goto out_free_plane; @@ -1254,6 +1257,9 @@ static struct kvm_plane *kvm_create_plane(struct kvm = *kvm, unsigned plane_level) static void kvm_destroy_one_plane(struct kvm_plane *plane) { kvm_arch_plane_destroy(plane); +#ifdef CONFIG_VM_PLANES + xa_destroy(&plane->access_attr_array); +#endif kvm_free_plane(plane); } =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id CDC4546983C; Wed, 5 Aug 2026 11:04:12 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927855; cv=none; b=t8pNCBRarZ3NjMylX0tFqdDN1v8sOo5yeZbSwYJuVE9wplM1QEGFcU2eSo2XpMXmEIzR8+IO/iJhPM/dqxHbxQpYDPmv1azPGYYlVfxcPDbOEyvNekjfW0vIlg9H0ZkkJtzXwOpjR92zV3u60EjEx+Ixhst6VO9FMCyRb9hzRGA= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927855; c=relaxed/simple; bh=pyKS7+F0zwetOCNSEOXtWXwvWirtCIBTLIx9T8AhX8Q=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version:Content-Type; b=QgGfZJEjLj0yzoKjSBhLZ5nT/pzBgZsmlF+rMQj0JXmpKhH4GA3Smf54+PDJpTTSGujH5HIOg9oS6iqpIDuvcWm71cnHwZTZClbkg+LXNO9hTHKnHZHW12feao/1GtkB5EX56028BgPmBudBd1Y4gmVz5O3BOpz+LZvo78ed75o= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=CJlBgF8l; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="CJlBgF8l" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id D0DFB20B716A; Wed, 5 Aug 2026 04:03:51 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com D0DFB20B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927831; bh=dl2+i7STnM4/26bHF7J69Rp6JVzOlj+gj+sfuPNR6s8=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=CJlBgF8lev9RS7nfdVY4Vpj2J4+DZg2iLqrSaQxFGTNGPDGpPTv5UG+0XNnaV2R98 /Z7MGecRVQElK9jfGkRu9rWCHJ6yMqsTwIXCsF46CfFen9IV+6nR12q7zC20y1oOy7 HZRUqPtvmBZMM2CrlZ6I+SWc8gVH/vfdMlLW7Qw0= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 35/42] KVM: x86: deny normal-plane access to secure-plane memory Date: Wed, 5 Aug 2026 04:03:17 -0700 Message-ID: <20260805110324.25067-36-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Type: text/plain; charset="utf-8" Content-Transfer-Encoding: quoted-printable VM planes share one guest physical address space (one set of memslots), so today the normal plane (plane 0) can read the secure plane's RAM. Add per-plane access control so a higher-privilege plane can hide its memory from a lower one: - Encode the plane into union kvm_mmu_page_role (the previously spare 4 bits) so each plane gets its own TDP/EPT root instead of sharing one set of page tables. - Give each struct kvm_plane its own access_attr_array (xarray), independent of kvm->mem_attr_array, to avoid coupling with the private/CoCo memory-attribute machinery. - Add KVM_MEMORY_ATTRIBUTE_NO_READ. NO_READ cannot be expressed as a present-but-unreadable EPT entry on all hardware, so it is enforced in the fault path: kvm_mmu_faultin_pfn() refuses to map a NO_READ gfn (and a write to a NO_WRITE gfn) for the faulting plane and exits with KVM_EXIT_MEMORY_FAULT instead of building an SPTE the access would immediately re-fault on. NO_WRITE/NO_EXEC continue to be stripped in kvm_plane_filter_pte_access(). - KVM_HC_VBS_SET_MEM_ATTRS lets a plane >0 apply NO_READ/NO_WRITE/ NO_EXEC to the plane directly below it (the secure plane cannot issue the host KVM_SET_MEMORY_ATTRIBUTES ioctl). a2 is an allow-mask: bit0 read, bit1 write, bit2 exec; a cleared bit adds the matching restriction, a2 =3D=3D 0 hides the range entirely. drivers/virt/secure_monitor.c uses this to seal the secure plane's own RAM (walk_system_ram_range -> SET_MEM_ATTRS with perms 0) from the normal plane before handing control back, so plane 0 can no longer read plane 1. --- drivers/virt/secure_monitor.c | 79 +++++++++++++++++++++++++++++++++-- 1 file changed, 76 insertions(+), 3 deletions(-) diff --git a/drivers/virt/secure_monitor.c b/drivers/virt/secure_monitor.c index 2d181c32c439..028ae222037a 100644 --- a/drivers/virt/secure_monitor.c +++ b/drivers/virt/secure_monitor.c @@ -25,11 +25,15 @@ * Because all planes of a VM share the same memslots (struct kvm_plane ha= s no * memslots of its own; they live in struct kvm), the secure plane sees the * same guest-physical address space as the normal plane and can read the - * calling area and the GPAs referenced by each request directly. + * calling area and the GPAs referenced by each request directly. This sa= me + * sharing means the secure plane must explicitly hide its own RAM from the + * normal plane: on startup it walks its system RAM and asks KVM (via + * KVM_HC_VBS_SET_MEM_ATTRS) to deny the normal plane read/write/exec acce= ss, + * so plane 0 cannot read secure-plane memory. * * For now every VTL call is acknowledged as a no-op so the normal plane c= an - * make progress; the real per-call handlers (self-protection, HEKI memory - * protection, kernel sealing, =E2=80=A6) are plumbed in incrementally. + * make progress; the remaining per-call handlers (HEKI memory protection, + * kernel sealing, =E2=80=A6) are plumbed in incrementally. * * Activated by the "secure_monitor" kernel command-line option; without it * this kernel boots normally and never parks. @@ -41,10 +45,13 @@ #include #include #include +#include +#include #include #include #include #include +#include #include #include =20 @@ -85,12 +92,78 @@ static u64 secmon_vtl_return(long status) return kvm_hypercall1(KVM_HC_VBS_VTL_RETURN, (unsigned long)status); } =20 +/* + * Apply EPT permissions on a normal-plane GPA range from the secure plane. + * + * The secure plane cannot issue the host KVM_SET_MEMORY_ATTRIBUTES ioctl,= so + * it asks KVM to do it via the KVM_HC_VBS_SET_MEM_ATTRS hypercall, which = KVM + * honours only for a higher-privilege plane (it applies the attributes to= the + * plane directly below the caller). @perms carries the access bits the + * normal plane should retain (VBS_MEM_*); KVM translates a cleared + * read/write/exec bit into NO_READ / NO_WRITE / NO_EXEC. @perms =3D=3D 0= hides + * the range entirely. + */ +static int secmon_apply_attrs(u64 gpa, u64 size, u32 perms) +{ + long ret; + + pr_debug("apply_attrs gpa=3D0x%llx size=3D0x%llx perms=3D%c%c%c\n", + gpa, size, + (perms & VBS_MEM_READ) ? 'r' : '-', + (perms & VBS_MEM_WRITE) ? 'w' : '-', + (perms & VBS_MEM_EXEC) ? 'x' : '-'); + + ret =3D kvm_hypercall3(KVM_HC_VBS_SET_MEM_ATTRS, gpa, size, perms); + if (ret) + return (int)ret; + + return 0; +} + +/* + * Hide one range of this plane's RAM from the normal plane. perms =3D 0 = means + * "retain no access" (no read/write/exec), so the normal plane faults and= is + * denied if it tries to touch secure-plane memory. + */ +static int secmon_hide_range(unsigned long start_pfn, unsigned long nr_pag= es, + void *arg) +{ + unsigned long gpa =3D start_pfn << PAGE_SHIFT; + unsigned long size =3D nr_pages << PAGE_SHIFT; + int r; + + r =3D secmon_apply_attrs(gpa, size, 0); + if (r) + pr_warn("failed to protect RAM [0x%lx+0x%lx]: %d\n", + gpa, size, r); + else + pr_info("protected RAM [0x%lx+0x%lx] from normal plane\n", + gpa, size); + + /* Continue with the remaining ranges even if one fails. */ + return 0; +} + +/* + * Deny the normal plane access to all of the secure plane's own RAM. Runs + * while the normal plane is frozen in the KVM_RUN that switched to us, so + * there is no window during which the memory is both populated and still + * readable by the normal plane. + */ +static void secmon_protect_self(void) +{ + walk_system_ram_range(0, max_pfn, NULL, secmon_hide_range); +} + static int secmon_monitor_fn(void *unused) { long status =3D 0; =20 pr_info("secure monitor started\n"); =20 + /* Seal our memory from the normal plane before handing control back. */ + secmon_protect_self(); + for (;;) { struct vbs_kvm_ca *ca; u64 ca_gpa; --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 3D10B46A5F2; Wed, 5 Aug 2026 11:04:13 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927854; cv=none; b=T3+G3A6OyO70hLfTHjf7cv+zq692Pt/S3Efnwsn3aVBOOIM2hY6+IDRf/25FQc9bp9HAR8V5vd9I3tdTsRgWkjFXxbRlPA4BZE4jm6uhchLc6agh9QNxoN7PKxCpaWP3zrGZ2b7lPGWEo5gKmTzgYzLOgIrXdJJ7hUfYl79Q0r8= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927854; c=relaxed/simple; bh=CMtQOBxsOyZafsRq9Hm/+RIAhu7pzUJUx0wZ70Pn+qA=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=anso/W7X+S+jIrGR7EgxoSdpASDlfXkm5cB1OoaE4283rDLu3ayBe2XHuexuVrDIXUdngde2/ILdZEUiR01ycnJZhXTRwkRmYfe5sMlqcgUIm9ICNdAdNKT01/3PYUXsUTXQ17qoUe4BXkEAXAMI/0s3LdhWbjbzNNtex6gqJzs= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=ptqwnP4n; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="ptqwnP4n" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 703A420B716B; Wed, 5 Aug 2026 04:03:52 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 703A420B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927832; bh=sI2sXVp/yDVu/Cytrm86fNxGZAmR2vZkYQFR/0LYH8s=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=ptqwnP4nE71RqbshWWQTSPUb724PWfo3h4jWJQ1rQmspZmSAZ54+nGmQsODPrqLn3 HGa6y//yfmt3xJoUAeSsCRDCgaxyLgfA3tmd298neLoDt56lGmEuKSV/yJuwn4Hayc ntfx8/TaAJsQHEZ8mPVOOhbSFfoWkGvTS4FOHKps= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 36/42] KVM: plane: handle KVM_CHECK_EXTENSION on the plane fd Date: Wed, 5 Aug 2026 04:03:18 -0700 Message-ID: <20260805110324.25067-37-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" A plane file descriptor did not implement KVM_CHECK_EXTENSION and failed with -ENOTTY, but userspace needs to be able to query capabilities on it. Forward the query to the plane's parent VM, except for KVM_CAP_PLANES which returns 0 because a plane cannot host planes of its own. --- virt/kvm/kvm_main.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 3a1a09f26340..6e4f3f3e6881 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -4946,12 +4946,21 @@ static long kvm_vcpu_compat_ioctl(struct file *filp, } #endif =20 +static int kvm_vm_ioctl_check_extension_generic(struct kvm *kvm, long arg); + static long __kvm_plane_ioctl(struct kvm_plane *plane, unsigned int ioctl,= unsigned long arg) { void __user *argp =3D (void __user *)arg; long r; =20 switch (ioctl) { + case KVM_CHECK_EXTENSION: + /* A plane cannot host planes of its own. */ + if (arg =3D=3D KVM_CAP_PLANES) + r =3D 0; + else + r =3D kvm_vm_ioctl_check_extension_generic(plane->kvm, arg); + break; case KVM_CREATE_VCPU: r =3D kvm_plane_ioctl_create_vcpu(plane, arg); break; --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 3306A46AA7D; Wed, 5 Aug 2026 11:04:14 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927856; cv=none; b=NquqtRLp5o2lHSpKP7iK58AC3TiJMe9EJwK363D8XM7I74SiBPR2tpWOA2QmCpRVERhLlzJ2neiTa00jk0sEo0BHwDc/qnBnLmzBBDSYIlC8W/9OQ74CEFEmazIfrKbT77thT88q5VIoIMu+bBxPZOwmzuOaMxv6EAnE09gzTvw= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927856; c=relaxed/simple; bh=BfpEaAGkDyfbANpfPBIaZd/uzX61Ff66TSNwy44xLiA=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=FdqdRQx1EHWLe19ZSvM9WHxY77bqp6+v3cp+r8cKXsYStrZEzp8KcHLhWsbobel5nHFaoL8Cq7jI0i1GIn+4GPRFRt454nixbdSLxbC9ppIyzrXoytUc05q6pt42AU+OU5+PSExqRDM7mr5C6sMK6oE+VBbfk7KX+9RNLD0j/RQ= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=kO9BUPAv; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="kO9BUPAv" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 08ADA20B716D; Wed, 5 Aug 2026 04:03:52 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 08ADA20B716D DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927833; bh=CZO/FdjjrPxcYw4qgRLhHr2fLzYDF8y1zRIOsjWT26g=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=kO9BUPAvWRCzrsYobVhBUcB2xm2bM2DD4VJpERjHV/eSn5VIeK8bsumdJESYK45rF WkzGQhI4i2uMD2Z9sHvJZR5OdY1VB7ejAsN9Xe+BlMjXYG3rwnQmFYBYdLfZlLRKDe s+icGXDgINZyI4YasxfUFbj3PxI8NUqUtck1jRbE= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 37/42] KVM: selftests: run plane tests with a split IRQ chip Date: Wed, 5 Aug 2026 04:03:19 -0700 Message-ID: <20260805110324.25067-38-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Wire the plane selftests to actually exercise planes instead of skipping: - Add vm_create_barebones_irqchip() and create the test VMs with a split IRQ chip, which planes require, and query KVM_CAP_PLANES on the VM fd (system scope always returns 1). - Switch plane vCPU creation to KVM_CREATE_VCPU with the vCPU id, matching the current plane ABI, and drop the removed KVM_CREATE_VCPU_PLANE, KVM_CAP_PLANES_FPU and req_exit_planes paths from the tests and docs. - Rename x86/plane_test.c to x86/plane_x86_test.c. --- tools/testing/selftests/kvm/Makefile.kvm | 2 +- .../testing/selftests/kvm/include/kvm_util.h | 12 ++ tools/testing/selftests/kvm/lib/kvm_util.c | 8 +- tools/testing/selftests/kvm/plane_test.c | 20 ++-- .../x86/{plane_test.c =3D> plane_x86_test.c} | 109 +++--------------- 5 files changed, 45 insertions(+), 106 deletions(-) rename tools/testing/selftests/kvm/x86/{plane_test.c =3D> plane_x86_test.c= } (58%) diff --git a/tools/testing/selftests/kvm/Makefile.kvm b/tools/testing/selft= ests/kvm/Makefile.kvm index 80933e942ecf..750350e187c8 100644 --- a/tools/testing/selftests/kvm/Makefile.kvm +++ b/tools/testing/selftests/kvm/Makefile.kvm @@ -102,7 +102,7 @@ TEST_GEN_PROGS_x86 +=3D x86/nested_tdp_fault_test TEST_GEN_PROGS_x86 +=3D x86/nested_tsc_adjust_test TEST_GEN_PROGS_x86 +=3D x86/nested_tsc_scaling_test TEST_GEN_PROGS_x86 +=3D x86/nested_vmsave_vmload_test -TEST_GEN_PROGS_x86 +=3D x86/plane_test +TEST_GEN_PROGS_x86 +=3D x86/plane_x86_test TEST_GEN_PROGS_x86 +=3D x86/platform_info_test TEST_GEN_PROGS_x86 +=3D x86/pmu_counters_test TEST_GEN_PROGS_x86 +=3D x86/pmu_event_filter_test diff --git a/tools/testing/selftests/kvm/include/kvm_util.h b/tools/testing= /selftests/kvm/include/kvm_util.h index 2ea2960f1e2d..c4726fc7b065 100644 --- a/tools/testing/selftests/kvm/include/kvm_util.h +++ b/tools/testing/selftests/kvm/include/kvm_util.h @@ -1070,6 +1070,18 @@ static inline struct kvm_vm *vm_create_barebones(voi= d) return ____vm_create(VM_SHAPE_DEFAULT); } =20 +static inline struct kvm_vm *vm_create_barebones_irqchip(bool split) +{ + struct kvm_vm *vm =3D vm_create_barebones(); + + if (split) + vm_enable_cap(vm, KVM_CAP_SPLIT_IRQCHIP, 24); + else + vm_create_irqchip(vm); + + return vm; +} + static inline struct kvm_vm *vm_create_barebones_type(unsigned long type) { const struct vm_shape shape =3D { diff --git a/tools/testing/selftests/kvm/lib/kvm_util.c b/tools/testing/sel= ftests/kvm/lib/kvm_util.c index 43a23634b4f4..ce53230b23d0 100644 --- a/tools/testing/selftests/kvm/lib/kvm_util.c +++ b/tools/testing/selftests/kvm/lib/kvm_util.c @@ -790,10 +790,8 @@ static void vm_vcpu_rm(struct kvm_vm *vm, struct kvm_v= cpu *vcpu) void kvm_vm_release(struct kvm_vm *vmp) { struct kvm_vcpu *vcpu, *tmp_vcpu; - struct kvm_plane_vcpu *plane_vcpu, *tmp_plane_vcpu; - struct kvm_plane *plane, *tmp_plane; =20 - list_for_each_entry_safe(vcpu, tmp, &vmp->vcpus, list) + list_for_each_entry_safe(vcpu, tmp_vcpu, &vmp->vcpus, list) vm_vcpu_rm(vmp, vcpu); =20 kvm_free_fd(vmp->fd); @@ -1366,8 +1364,8 @@ struct kvm_plane_vcpu *__vm_plane_vcpu_add(struct kvm= _vcpu *vcpu, struct kvm_pla plane_vcpu =3D calloc(1, sizeof(*plane_vcpu)); TEST_ASSERT(plane_vcpu !=3D NULL, "Insufficient Memory"); =20 - plane_vcpu->fd =3D __plane_ioctl(plane, KVM_CREATE_VCPU_PLANE, (void *)(u= nsigned long)vcpu->fd); - TEST_ASSERT_VM_VCPU_IOCTL(plane_vcpu->fd >=3D 0, KVM_CREATE_VCPU_PLANE, p= lane_vcpu->fd, plane->vm); + plane_vcpu->fd =3D __plane_ioctl(plane, KVM_CREATE_VCPU, (void *)(unsigne= d long)vcpu->id); + TEST_ASSERT_VM_VCPU_IOCTL(plane_vcpu->fd >=3D 0, KVM_CREATE_VCPU, plane_v= cpu->fd, plane->vm); plane_vcpu->id =3D vcpu->id; plane_vcpu->plane0 =3D vcpu; =20 diff --git a/tools/testing/selftests/kvm/plane_test.c b/tools/testing/selft= ests/kvm/plane_test.c index 9cf3ab76b3cd..fd09d1f78ebe 100644 --- a/tools/testing/selftests/kvm/plane_test.c +++ b/tools/testing/selftests/kvm/plane_test.c @@ -21,7 +21,8 @@ void test_create_plane_errors(int max_planes) struct kvm_vcpu *vcpu; int planefd, plane_vcpufd; =20 - vm =3D vm_create_barebones(); + /* Planes require an in-kernel (split) IRQ chip. */ + vm =3D vm_create_barebones_irqchip(true); vcpu =3D __vm_vcpu_add(vm, 0); =20 planefd =3D __vm_ioctl(vm, KVM_CREATE_PLANE, (void *)(unsigned long)0); @@ -34,9 +35,9 @@ void test_create_plane_errors(int max_planes) "Creating plane %d, expecting EINVAL. ret: %d, errno: %d", max_planes, planefd, errno); =20 - plane_vcpufd =3D __vm_ioctl(vm, KVM_CREATE_VCPU_PLANE, (void *)(unsigned = long)vcpu->fd); - TEST_ASSERT(plane_vcpufd =3D=3D -1 && errno =3D=3D ENOTTY, - "Creating vCPU for plane 0, expecting ENOTTY. ret: %d, errno: %d", + plane_vcpufd =3D __vm_ioctl(vm, KVM_CREATE_VCPU, (void *)(unsigned long)v= cpu->id); + TEST_ASSERT(plane_vcpufd =3D=3D -1 && errno =3D=3D EEXIST, + "Creating existing vCPU for plane 0, expecting EEXIST. ret: %d, errn= o: %d", plane_vcpufd, errno); =20 kvm_vm_free(vm); @@ -50,7 +51,7 @@ void test_create_plane(void) struct kvm_plane *plane; int r; =20 - vm =3D vm_create_barebones(); + vm =3D vm_create_barebones_irqchip(true); vcpu =3D __vm_vcpu_add(vm, 0); =20 plane =3D vm_plane_add(vm, 1); @@ -70,7 +71,7 @@ void test_create_plane(void) =20 __vm_plane_vcpu_add(vcpu, plane); =20 - r =3D __plane_ioctl(plane, KVM_CREATE_VCPU_PLANE, (void *)(unsigned long)= vcpu->fd); + r =3D __plane_ioctl(plane, KVM_CREATE_VCPU, (void *)(unsigned long)vcpu->= id); TEST_ASSERT(r =3D=3D -1 && errno =3D=3D EEXIST, "Creating vCPU again for plane 1. ret: %d, errno: %d", r, errno); @@ -86,7 +87,10 @@ void test_create_plane(void) =20 int main(int argc, char *argv[]) { - int cap_planes =3D kvm_check_cap(KVM_CAP_PLANES); + struct kvm_vm *vm =3D vm_create_barebones_irqchip(true); + int cap_planes =3D vm_check_cap(vm, KVM_CAP_PLANES); + + kvm_vm_free(vm); TEST_REQUIRE(cap_planes); =20 ksft_print_header(); @@ -98,6 +102,8 @@ int main(int argc, char *argv[]) =20 if (cap_planes > 1) test_create_plane(); + else + ksft_test_result_skip("plane creation requires KVM_CAP_PLANES > 1\n"); =20 ksft_finished(); } diff --git a/tools/testing/selftests/kvm/x86/plane_test.c b/tools/testing/s= elftests/kvm/x86/plane_x86_test.c similarity index 58% rename from tools/testing/selftests/kvm/x86/plane_test.c rename to tools/testing/selftests/kvm/x86/plane_x86_test.c index 0fdd8a066723..8f0919371383 100644 --- a/tools/testing/selftests/kvm/x86/plane_test.c +++ b/tools/testing/selftests/kvm/x86/plane_x86_test.c @@ -5,6 +5,7 @@ * Test for x86-specific VM plane functionality */ #include +#include #include #include #include @@ -26,7 +27,7 @@ static void test_plane_regs(void) =20 struct kvm_regs regs0, regs1; =20 - vm =3D vm_create_barebones(); + vm =3D vm_create_barebones_irqchip(true); vcpu =3D __vm_vcpu_add(vm, 0); plane =3D vm_plane_add(vm, 1); plane_vcpu =3D __vm_plane_vcpu_add(vcpu, plane); @@ -62,8 +63,7 @@ static void test_plane_fpu_nonshared(void) =20 struct kvm_xsave xsave0, xsave1; =20 - vm =3D vm_create_barebones(); - TEST_ASSERT_EQ(vm_check_cap(vm, KVM_CAP_PLANES_FPU), false); + vm =3D vm_create_barebones_irqchip(true); =20 vcpu =3D __vm_vcpu_add(vm, 0); vcpu_init_cpuid(vcpu, kvm_get_supported_cpuid()); @@ -93,79 +93,15 @@ static void test_plane_fpu_nonshared(void) ksft_test_result_pass("get/set FPU not shared across planes\n"); } =20 -static void test_plane_fpu_shared(void) -{ - struct kvm_vm *vm; - struct kvm_vcpu *vcpu; - struct kvm_plane *plane; - struct kvm_plane_vcpu *plane_vcpu; - - struct kvm_xsave xsave0, xsave1; - - vm =3D vm_create_barebones(); - vm_enable_cap(vm, KVM_CAP_PLANES_FPU, 1ul); - TEST_ASSERT_EQ(vm_check_cap(vm, KVM_CAP_PLANES_FPU), true); - - vcpu =3D __vm_vcpu_add(vm, 0); - vcpu_init_cpuid(vcpu, kvm_get_supported_cpuid()); - vcpu_set_cpuid(vcpu); - - plane =3D vm_plane_add(vm, 1); - plane_vcpu =3D __vm_plane_vcpu_add(vcpu, plane); - - vcpu_ioctl(vcpu, KVM_GET_XSAVE, &xsave0); - - xsave0.region[XSTATE_BV_OFFSET] |=3D XFEATURE_MASK_FP | XFEATURE_MASK_SSE; - xsave0.region[XMM_OFFSET] =3D 0x12345678; - vcpu_ioctl(vcpu, KVM_SET_XSAVE, &xsave0); - plane_vcpu_ioctl(plane_vcpu, KVM_GET_XSAVE, &xsave1); - TEST_ASSERT_EQ(xsave1.region[XMM_OFFSET], 0x12345678); - - xsave1.region[XSTATE_BV_OFFSET] |=3D XFEATURE_MASK_FP | XFEATURE_MASK_SSE; - xsave1.region[XMM_OFFSET] =3D 0x87654321; - plane_vcpu_ioctl(plane_vcpu, KVM_SET_XSAVE, &xsave1); - vcpu_ioctl(vcpu, KVM_GET_XSAVE, &xsave0); - TEST_ASSERT_EQ(xsave0.region[XMM_OFFSET], 0x87654321); - - ksft_test_result_pass("get/set FPU shared across planes\n"); - - if (!this_cpu_has(X86_FEATURE_PKU)) { - ksft_test_result_skip("get/set PKRU with shared FPU\n"); - goto exit; - } - - xsave0.region[XSTATE_BV_OFFSET] =3D XFEATURE_MASK_PKRU; - xsave0.region[PKRU_OFFSET] =3D 0xffffffff; - vcpu_ioctl(vcpu, KVM_SET_XSAVE, &xsave0); - plane_vcpu_ioctl(plane_vcpu, KVM_GET_XSAVE, &xsave0); - - xsave0.region[XSTATE_BV_OFFSET] =3D XFEATURE_MASK_PKRU; - xsave0.region[PKRU_OFFSET] =3D 0xaaaaaaaa; - vcpu_ioctl(vcpu, KVM_SET_XSAVE, &xsave0); - plane_vcpu_ioctl(plane_vcpu, KVM_GET_XSAVE, &xsave1); - assert(xsave1.region[PKRU_OFFSET] =3D=3D 0xffffffff); - - xsave1.region[XSTATE_BV_OFFSET] =3D XFEATURE_MASK_PKRU; - xsave1.region[PKRU_OFFSET] =3D 0x55555555; - plane_vcpu_ioctl(plane_vcpu, KVM_SET_XSAVE, &xsave1); - vcpu_ioctl(vcpu, KVM_GET_XSAVE, &xsave0); - assert(xsave0.region[PKRU_OFFSET] =3D=3D 0xaaaaaaaa); - - ksft_test_result_pass("get/set PKRU with shared FPU\n"); - -exit: - kvm_vm_free(vm); -} - #define APIC_SPIV 0xF0 #define APIC_IRR 0x200 =20 #define MYVEC 192 =20 -#define MAKE_MSI(cpu, vector) ((struct kvm_msi){ \ - .address_lo =3D APIC_DEFAULT_GPA + (((cpu) & 0xff) << 8), \ - .address_hi =3D (cpu) & ~0xff, \ - .data =3D (vector), \ +#define MAKE_MSI(cpu, vector) ((struct kvm_msi){ \ + .address_lo =3D APIC_DEFAULT_GPA + (((cpu) & 0xff) << 8), \ + .address_hi =3D (cpu) & ~0xff, \ + .data =3D (vector), \ }) =20 static bool has_irr(struct kvm_lapic_state *apic, int vector) @@ -194,7 +130,7 @@ static void test_plane_msi(void) struct kvm_msi msi =3D MAKE_MSI(0, MYVEC); struct kvm_lapic_state lapic0, lapic1; =20 - vm =3D __vm_create(VM_SHAPE_DEFAULT, 1, 0); + vm =3D vm_create_barebones_irqchip(true); =20 vcpu =3D __vm_vcpu_add(vm, 0); vcpu_init_cpuid(vcpu, kvm_get_supported_cpuid()); @@ -215,6 +151,7 @@ static void test_plane_msi(void) do_enable_lapic(&lapic1); plane_vcpu_ioctl(plane_vcpu, KVM_SET_LAPIC, &lapic1); =20 + /* Deliver to plane 1 (via the plane fd); it must land only in plane 1. */ r =3D __plane_ioctl(plane, KVM_SIGNAL_MSI, &msi); TEST_ASSERT(r =3D=3D 1, "Delivering interrupt to plane 1. ret: %d, errno: %d", r, errno); @@ -224,46 +161,32 @@ static void test_plane_msi(void) plane_vcpu_ioctl(plane_vcpu, KVM_GET_LAPIC, &lapic1); TEST_ASSERT(has_irr(&lapic1, MYVEC), "Vector set in plane 1"); =20 - /* req_exit_planes always has priority */ - vcpu->run->req_exit_planes =3D (1 << 1); - vcpu_run(vcpu); - TEST_ASSERT_EQ(vcpu->run->exit_reason, KVM_EXIT_PLANE_EVENT); - TEST_ASSERT_EQ(vcpu->run->plane_event.cause, KVM_PLANE_EVENT_INTERRUPT); - TEST_ASSERT_EQ(vcpu->run->plane_event.pending_event_planes, (1 << 1)); - TEST_ASSERT_EQ(vcpu->run->plane_event.target, (1 << 1)); - + /* Deliver to plane 0 (via the vm fd); it must land in plane 0. */ r =3D __vm_ioctl(vm, KVM_SIGNAL_MSI, &msi); TEST_ASSERT(r =3D=3D 1, "Delivering interrupt to plane 0. ret: %d, errno: %d", r, errno); vcpu_ioctl(vcpu, KVM_GET_LAPIC, &lapic0); TEST_ASSERT(has_irr(&lapic0, MYVEC), "Vector set in plane 0"); =20 - /* req_exit_planes ignores current plane; current plane is cleared */ - vcpu->run->plane =3D 1; - vcpu->run->req_exit_planes =3D (1 << 0) | (1 << 1); - vcpu_run(vcpu); - TEST_ASSERT_EQ(vcpu->run->exit_reason, KVM_EXIT_PLANE_EVENT); - TEST_ASSERT_EQ(vcpu->run->plane_event.cause, KVM_PLANE_EVENT_INTERRUPT); - TEST_ASSERT_EQ(vcpu->run->plane_event.pending_event_planes, (1 << 0)); - TEST_ASSERT_EQ(vcpu->run->plane_event.target, (1 << 0)); - kvm_vm_free(vm); - ksft_test_result_pass("signal MSI for planes\n"); + ksft_test_result_pass("signal MSI routed per plane\n"); } =20 int main(int argc, char *argv[]) { - int cap_planes =3D kvm_check_cap(KVM_CAP_PLANES); + struct kvm_vm *vm =3D vm_create_barebones_irqchip(true); + int cap_planes =3D vm_check_cap(vm, KVM_CAP_PLANES); + + kvm_vm_free(vm); TEST_REQUIRE(cap_planes && cap_planes > 1); =20 ksft_print_header(); - ksft_set_plan(5); + ksft_set_plan(3); =20 pr_info("# KVM_CAP_PLANES: %d\n", cap_planes); =20 test_plane_regs(); test_plane_fpu_nonshared(); - test_plane_fpu_shared(); test_plane_msi(); =20 ksft_finished(); --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id E565746AF1B; Wed, 5 Aug 2026 11:04:14 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927857; cv=none; b=MtMPOGY3aCJ0cBoSdQX+jdN3NBAjHCTNFgMBGy+g5HeeolpiSnnFWpLErLetOciuAkg38ts/J1DfI+oKgbXIH3iec61p0mNv/HYD3nCyk2CmBggAKzc1sYKEARqO2OaSrMgn9BsRlWCqytW3aipQoH/Lor4Kw95tDq+UxU7FKJs= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927857; c=relaxed/simple; bh=V/OmEqubhut8+oe2kbPfOysWHu3Q0MY+ynqiulXQ3Ns=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=c9ts+et5HkwxgqkeOgr/p6/IHQLwor3Y5FAAIDZDi6VR9Mefc9Xfwcf02RNEB1NGNQtLvkQBNpsThGDikKzkW93LSaKfYM7/1XAkZ7t5vy74awjPmyeEnSz2ORkfRHvtQYFthv9/baSLrsfarEk8ceNjKgvYS9I3JF86DXfnbhc= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=L+cQ4xcI; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="L+cQ4xcI" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id EA4D920B716E; Wed, 5 Aug 2026 04:03:53 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com EA4D920B716E DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927834; bh=wj5UF/1hwIz+DHXT+8I+TLhcRxrnRC+oQlVIsaDurzY=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=L+cQ4xcIqOIsMgxsYPi1VSr5W/3EWJpE46+B6mgU2ejV/wJfwkVofLr0ZdOIH+A0+ cU+xU2QDJ9hGivL6kkT/PXYG8671Y53FpmhUqXxzuHZbQ2/JTYOp/OHu+p02aJSu51 Udh1vQnGtusW2ry72bDG1/eFgA7o9tUrL+veRll0= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 38/42] kvm: x86: drop obsolete kvm_cache_regs.h Date: Wed, 5 Aug 2026 04:03:20 -0700 Message-ID: <20260805110324.25067-39-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" This header was removed upstream (its GPR/CR accessors now live in kvm_host.h); the linear cherry-pick spuriously recreated it while resolving a conflict. Delete it to match current upstream and the vm-planes-merged tree. --- arch/x86/kvm/kvm_cache_regs.h | 249 ---------------------------------- 1 file changed, 249 deletions(-) delete mode 100644 arch/x86/kvm/kvm_cache_regs.h diff --git a/arch/x86/kvm/kvm_cache_regs.h b/arch/x86/kvm/kvm_cache_regs.h deleted file mode 100644 index 8ddb01191d6f..000000000000 --- a/arch/x86/kvm/kvm_cache_regs.h +++ /dev/null @@ -1,249 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#ifndef ASM_KVM_CACHE_REGS_H -#define ASM_KVM_CACHE_REGS_H - -#include - -#define KVM_POSSIBLE_CR0_GUEST_BITS (X86_CR0_TS | X86_CR0_WP) -#define KVM_POSSIBLE_CR4_GUEST_BITS \ - (X86_CR4_PVI | X86_CR4_DE | X86_CR4_PCE | X86_CR4_OSFXSR \ - | X86_CR4_OSXMMEXCPT | X86_CR4_PGE | X86_CR4_TSD | X86_CR4_FSGSBASE \ - | X86_CR4_CET) - -#define X86_CR0_PDPTR_BITS (X86_CR0_CD | X86_CR0_NW | X86_CR0_PG) -#define X86_CR4_TLBFLUSH_BITS (X86_CR4_PGE | X86_CR4_PCIDE | X86_CR4_PAE |= X86_CR4_SMEP) -#define X86_CR4_PDPTR_BITS (X86_CR4_PGE | X86_CR4_PSE | X86_CR4_PAE | X= 86_CR4_SMEP) - -static_assert(!(KVM_POSSIBLE_CR0_GUEST_BITS & X86_CR0_PDPTR_BITS)); - -#define BUILD_KVM_GPR_ACCESSORS(lname, uname) \ -static __always_inline unsigned long kvm_##lname##_read(struct kvm_vcpu *v= cpu)\ -{ \ - return vcpu->arch.regs[VCPU_REGS_##uname]; \ -} \ -static __always_inline void kvm_##lname##_write(struct kvm_vcpu *vcpu, = \ - unsigned long val) \ -{ \ - vcpu->arch.regs[VCPU_REGS_##uname] =3D val; \ -} -BUILD_KVM_GPR_ACCESSORS(rax, RAX) -BUILD_KVM_GPR_ACCESSORS(rbx, RBX) -BUILD_KVM_GPR_ACCESSORS(rcx, RCX) -BUILD_KVM_GPR_ACCESSORS(rdx, RDX) -BUILD_KVM_GPR_ACCESSORS(rbp, RBP) -BUILD_KVM_GPR_ACCESSORS(rsi, RSI) -BUILD_KVM_GPR_ACCESSORS(rdi, RDI) -#ifdef CONFIG_X86_64 -BUILD_KVM_GPR_ACCESSORS(r8, R8) -BUILD_KVM_GPR_ACCESSORS(r9, R9) -BUILD_KVM_GPR_ACCESSORS(r10, R10) -BUILD_KVM_GPR_ACCESSORS(r11, R11) -BUILD_KVM_GPR_ACCESSORS(r12, R12) -BUILD_KVM_GPR_ACCESSORS(r13, R13) -BUILD_KVM_GPR_ACCESSORS(r14, R14) -BUILD_KVM_GPR_ACCESSORS(r15, R15) -#endif - -/* - * Using the register cache from interrupt context is generally not allowe= d, as - * caching a register and marking it available/dirty can't be done atomica= lly, - * i.e. accesses from interrupt context may clobber state or read stale da= ta if - * the vCPU task is in the process of updating the cache. The exception i= s if - * KVM is handling a PMI IRQ/NMI VM-Exit, as that bound code sequence does= n't - * touch the cache, it runs after the cache is reset (post VM-Exit), and P= MIs - * need to access several registers that are cacheable. - */ -#define kvm_assert_register_caching_allowed(vcpu) \ - lockdep_assert_once(in_task() || kvm_arch_pmi_in_guest(vcpu)) - -/* - * avail dirty - * 0 0 register in VMCS/VMCB - * 0 1 *INVALID* - * 1 0 register in vcpu->arch - * 1 1 register in vcpu->arch, needs to be stored back - */ -static inline bool kvm_register_is_available(struct kvm_vcpu *vcpu, - enum kvm_reg reg) -{ - kvm_assert_register_caching_allowed(vcpu); - return test_bit(reg, (unsigned long *)&vcpu->arch.regs_avail); -} - -static inline bool kvm_register_is_dirty(struct kvm_vcpu *vcpu, - enum kvm_reg reg) -{ - kvm_assert_register_caching_allowed(vcpu); - return test_bit(reg, (unsigned long *)&vcpu->arch.regs_dirty); -} - -static inline void kvm_register_mark_available(struct kvm_vcpu *vcpu, - enum kvm_reg reg) -{ - kvm_assert_register_caching_allowed(vcpu); - __set_bit(reg, (unsigned long *)&vcpu->arch.regs_avail); -} - -static inline void kvm_register_mark_dirty(struct kvm_vcpu *vcpu, - enum kvm_reg reg) -{ - kvm_assert_register_caching_allowed(vcpu); - __set_bit(reg, (unsigned long *)&vcpu->arch.regs_avail); - __set_bit(reg, (unsigned long *)&vcpu->arch.regs_dirty); -} - -/* - * kvm_register_test_and_mark_available() is a special snowflake that uses= an - * arch bitop directly to avoid the explicit instrumentation that comes wi= th - * the generic bitops. This allows code that cannot be instrumented (noin= str - * functions), e.g. the low level VM-Enter/VM-Exit paths, to cache registe= rs. - */ -static __always_inline bool kvm_register_test_and_mark_available(struct kv= m_vcpu *vcpu, - enum kvm_reg reg) -{ - kvm_assert_register_caching_allowed(vcpu); - return arch___test_and_set_bit(reg, (unsigned long *)&vcpu->arch.regs_ava= il); -} - -/* - * The "raw" register helpers are only for cases where the full 64 bits of= a - * register are read/written irrespective of current vCPU mode. In other = words, - * odds are good you shouldn't be using the raw variants. - */ -static inline unsigned long kvm_register_read_raw(struct kvm_vcpu *vcpu, i= nt reg) -{ - if (WARN_ON_ONCE((unsigned int)reg >=3D NR_VCPU_REGS)) - return 0; - - if (!kvm_register_is_available(vcpu, reg)) - kvm_x86_call(cache_reg)(vcpu, reg); - - return vcpu->arch.regs[reg]; -} - -static inline void kvm_register_write_raw(struct kvm_vcpu *vcpu, int reg, - unsigned long val) -{ - if (WARN_ON_ONCE((unsigned int)reg >=3D NR_VCPU_REGS)) - return; - - vcpu->arch.regs[reg] =3D val; - kvm_register_mark_dirty(vcpu, reg); -} - -static inline unsigned long kvm_rip_read(struct kvm_vcpu *vcpu) -{ - return kvm_register_read_raw(vcpu, VCPU_REGS_RIP); -} - -static inline void kvm_rip_write(struct kvm_vcpu *vcpu, unsigned long val) -{ - kvm_register_write_raw(vcpu, VCPU_REGS_RIP, val); -} - -static inline unsigned long kvm_rsp_read(struct kvm_vcpu *vcpu) -{ - return kvm_register_read_raw(vcpu, VCPU_REGS_RSP); -} - -static inline void kvm_rsp_write(struct kvm_vcpu *vcpu, unsigned long val) -{ - kvm_register_write_raw(vcpu, VCPU_REGS_RSP, val); -} - -static inline u64 kvm_pdptr_read(struct kvm_vcpu *vcpu, int index) -{ - might_sleep(); /* on svm */ - - if (!kvm_register_is_available(vcpu, VCPU_EXREG_PDPTR)) - kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_PDPTR); - - return vcpu->arch.walk_mmu->pdptrs[index]; -} - -static inline void kvm_pdptr_write(struct kvm_vcpu *vcpu, int index, u64 v= alue) -{ - vcpu->arch.walk_mmu->pdptrs[index] =3D value; -} - -static inline ulong kvm_read_cr0_bits(struct kvm_vcpu *vcpu, ulong mask) -{ - ulong tmask =3D mask & KVM_POSSIBLE_CR0_GUEST_BITS; - if ((tmask & vcpu->arch.cr0_guest_owned_bits) && - !kvm_register_is_available(vcpu, VCPU_EXREG_CR0)) - kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_CR0); - return vcpu->arch.cr0 & mask; -} - -static __always_inline bool kvm_is_cr0_bit_set(struct kvm_vcpu *vcpu, - unsigned long cr0_bit) -{ - BUILD_BUG_ON(!is_power_of_2(cr0_bit)); - - return !!kvm_read_cr0_bits(vcpu, cr0_bit); -} - -static inline ulong kvm_read_cr0(struct kvm_vcpu *vcpu) -{ - return kvm_read_cr0_bits(vcpu, ~0UL); -} - -static inline ulong kvm_read_cr4_bits(struct kvm_vcpu *vcpu, ulong mask) -{ - ulong tmask =3D mask & KVM_POSSIBLE_CR4_GUEST_BITS; - if ((tmask & vcpu->arch.cr4_guest_owned_bits) && - !kvm_register_is_available(vcpu, VCPU_EXREG_CR4)) - kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_CR4); - return vcpu->arch.cr4 & mask; -} - -static __always_inline bool kvm_is_cr4_bit_set(struct kvm_vcpu *vcpu, - unsigned long cr4_bit) -{ - BUILD_BUG_ON(!is_power_of_2(cr4_bit)); - - return !!kvm_read_cr4_bits(vcpu, cr4_bit); -} - -static inline ulong kvm_read_cr3(struct kvm_vcpu *vcpu) -{ - if (!kvm_register_is_available(vcpu, VCPU_EXREG_CR3)) - kvm_x86_call(cache_reg)(vcpu, VCPU_EXREG_CR3); - return vcpu->arch.cr3; -} - -static inline ulong kvm_read_cr4(struct kvm_vcpu *vcpu) -{ - return kvm_read_cr4_bits(vcpu, ~0UL); -} - -static inline u64 kvm_read_edx_eax(struct kvm_vcpu *vcpu) -{ - return (kvm_rax_read(vcpu) & -1u) - | ((u64)(kvm_rdx_read(vcpu) & -1u) << 32); -} - -static inline void enter_guest_mode(struct kvm_vcpu *vcpu) -{ - vcpu->arch.hflags |=3D HF_GUEST_MASK; - vcpu->stat.guest_mode =3D 1; -} - -static inline void leave_guest_mode(struct kvm_vcpu *vcpu) -{ - vcpu->arch.hflags &=3D ~HF_GUEST_MASK; - - if (vcpu->arch.load_eoi_exitmap_pending) { - vcpu->arch.load_eoi_exitmap_pending =3D false; - kvm_make_request(KVM_REQ_LOAD_EOI_EXITMAP, vcpu); - } - - vcpu->stat.guest_mode =3D 0; -} - -static inline bool is_guest_mode(struct kvm_vcpu *vcpu) -{ - return vcpu->arch.hflags & HF_GUEST_MASK; -} - -#endif --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 9EF8346C828; Wed, 5 Aug 2026 11:04:15 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927857; cv=none; b=MiiKXeNO/yzcxbVgbyLWPeNDIJLHC+UqNJbZNySWhiZmDRwVywzLPoVseq+hjXKfIBdQsK/t75LPwNA0QhRfV4pqjhBvjeOGZ/q2snizzXbBnnVWRh6ZB8J/GBzMuP2XJyziatCt7JJ/ejWxsg2AU5fZiHZ9i4FOV9mULxfjuT4= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927857; c=relaxed/simple; bh=HjrorX9aEMR3A/jZmW1M5gilRKHOo4eaxQ8e71+nlrk=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=XFe9FSmKDZ9Pe+fGOKmblc5z+VfsxlCXnOobvVmkt7lKjffx8ctU1l+qK/ztbfUrq7Hsug1i8wlqXse3mcSxYkQjZ8OXnLvsg1SEh9IbXWCXwbw6TsYeRQrDmoOcEg/B5Ys0P/ACihuBtkdjOd/0J5SoQKp3iRcdbz93HfP4hgg= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=cDWVByI+; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="cDWVByI+" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id BE50D20B7169; Wed, 5 Aug 2026 04:03:54 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com BE50D20B7169 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927834; bh=plw5PYs+q/L8H//hTTFQmxHz/itglm5JTOg6RfG0a0I=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=cDWVByI+Vstdqh/a+LKd41emN7NWXSn66sG6pK+5PQ42d6BimAIQhTavJj52cTjeI pKtv64YpK2tEwfw20hVCI/DTSadE3SUKFm4jKRcKwssdonqXj7d5BKf1ADCJXhbrKY WBb31Ult3tI12gUpKr85l1lW+6ZysKVePusTQTwI= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 39/42] kvm: arch: finalize plane hooks and kvm_arch_vcpu_create signature Date: Wed, 5 Aug 2026 04:03:21 -0700 Message-ID: <20260805110324.25067-40-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Match the vm-planes-merged tree on non-x86 architectures: revert kvm_arch_vcpu_create() to its plane-less prototype (planes attach the vCPU to plane 0 in generic code) and add the kvm_arch_init_plane/ kvm_arch_free_plane/kvm_arch_sync_events stubs without the stray conflict markers that the integration branch left committed. --- arch/arm64/include/asm/kvm_host.h | 7 +++++++ arch/arm64/kvm/arm.c | 2 +- arch/mips/include/asm/kvm_host.h | 6 ++++++ arch/powerpc/include/asm/kvm_host.h | 6 ++++++ arch/riscv/kvm/vcpu.c | 2 +- arch/s390/include/asm/kvm_host.h | 6 ++++++ 6 files changed, 27 insertions(+), 2 deletions(-) diff --git a/arch/arm64/include/asm/kvm_host.h b/arch/arm64/include/asm/kvm= _host.h index d3807b6535cc..75b4e52c4bac 100644 --- a/arch/arm64/include/asm/kvm_host.h +++ b/arch/arm64/include/asm/kvm_host.h @@ -237,6 +237,9 @@ struct kvm_s2_mmu { struct kvm_arch_memory_slot { }; =20 +struct kvm_arch_plane { +}; + /** * struct kvm_smccc_features: Descriptor of the hypercall services exposed= to the guests * @@ -1441,6 +1444,10 @@ static inline bool kvm_system_needs_idmapped_vectors= (void) return cpus_have_final_cap(ARM64_SPECTRE_V3A); } =20 +static inline void kvm_arch_init_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_free_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_sync_events(struct kvm *kvm) {} + void kvm_init_host_debug_data(void); void kvm_debug_init_vhe(void); void kvm_vcpu_load_debug(struct kvm_vcpu *vcpu); diff --git a/arch/arm64/kvm/arm.c b/arch/arm64/kvm/arm.c index d55be436bf16..c02e19fe2b48 100644 --- a/arch/arm64/kvm/arm.c +++ b/arch/arm64/kvm/arm.c @@ -538,7 +538,7 @@ int kvm_arch_vcpu_precreate(struct kvm *kvm, unsigned i= nt id) return 0; } =20 -int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu, struct kvm_plane *plane) +int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu) { int err; =20 diff --git a/arch/mips/include/asm/kvm_host.h b/arch/mips/include/asm/kvm_h= ost.h index c48bca79207b..8dc465dd9ea8 100644 --- a/arch/mips/include/asm/kvm_host.h +++ b/arch/mips/include/asm/kvm_host.h @@ -147,6 +147,9 @@ struct kvm_vcpu_stat { struct kvm_arch_memory_slot { }; =20 +struct kvm_arch_plane { +}; + #ifdef CONFIG_CPU_LOONGSON64 struct ipi_state { uint32_t status; @@ -903,6 +906,9 @@ extern unsigned long kvm_mips_get_ramsize(struct kvm *k= vm); extern int kvm_vcpu_ioctl_interrupt(struct kvm_vcpu *vcpu, struct kvm_mips_interrupt *irq); =20 +static inline void kvm_arch_init_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_free_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_sync_events(struct kvm *kvm) {} static inline void kvm_arch_free_memslot(struct kvm *kvm, struct kvm_memory_slot *slot) {} static inline void kvm_arch_memslots_updated(struct kvm *kvm, u64 gen) {} diff --git a/arch/powerpc/include/asm/kvm_host.h b/arch/powerpc/include/asm= /kvm_host.h index 47d9900c4f85..5d7036888933 100644 --- a/arch/powerpc/include/asm/kvm_host.h +++ b/arch/powerpc/include/asm/kvm_host.h @@ -256,6 +256,9 @@ struct kvm_arch_memory_slot { #endif /* CONFIG_KVM_BOOK3S_HV_POSSIBLE */ }; =20 +struct kvm_arch_plane { +}; + struct kvm_hpt_info { /* Host virtual (linear mapping) address of guest HPT */ unsigned long virt; @@ -919,6 +922,9 @@ struct kvm_vcpu_arch { #define __KVM_HAVE_ARCH_WQP #define __KVM_HAVE_CREATE_DEVICE =20 +static inline void kvm_arch_init_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_free_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_sync_events(struct kvm *kvm) {} static inline void kvm_arch_memslots_updated(struct kvm *kvm, u64 gen) {} static inline void kvm_arch_flush_shadow_all(struct kvm *kvm) {} static inline void kvm_arch_vcpu_blocking(struct kvm_vcpu *vcpu) {} diff --git a/arch/riscv/kvm/vcpu.c b/arch/riscv/kvm/vcpu.c index f618cd83d19a..4d7ee4059758 100644 --- a/arch/riscv/kvm/vcpu.c +++ b/arch/riscv/kvm/vcpu.c @@ -129,7 +129,7 @@ int kvm_arch_vcpu_precreate(struct kvm *kvm, unsigned i= nt id) return 0; } =20 -int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu, struct kvm_plane *plane) +int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu) { int rc; =20 diff --git a/arch/s390/include/asm/kvm_host.h b/arch/s390/include/asm/kvm_h= ost.h index 15c831304b51..423185a09de6 100644 --- a/arch/s390/include/asm/kvm_host.h +++ b/arch/s390/include/asm/kvm_host.h @@ -476,6 +476,9 @@ struct kvm_vm_stat { struct kvm_arch_memory_slot { }; =20 +struct kvm_arch_plane { +}; + struct s390_map_info { struct list_head list; __u64 guest_addr; @@ -773,6 +776,9 @@ extern int kvm_s390_gisc_unregister(struct kvm *kvm, u3= 2 gisc); =20 bool kvm_s390_is_gpa_in_memslot(struct kvm *kvm, gpa_t gpa); =20 +static inline void kvm_arch_init_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_free_plane(struct kvm_plane *plane) {} +static inline void kvm_arch_sync_events(struct kvm *kvm) {} static inline void kvm_arch_free_memslot(struct kvm *kvm, struct kvm_memory_slot *slot) {} static inline void kvm_arch_memslots_updated(struct kvm *kvm, u64 gen) {} --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 320D046C83D; Wed, 5 Aug 2026 11:04:16 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927858; cv=none; b=fuCyyGcwPgAzcKiaKoYlJZdpKhNKuBoEtl1ksCijxYWwueWO6cOC1/pPcgormkCSzzPsTMwECsuFl4s+Fti4Fh3HDmjIKqfGrZYYD5bUmHwHAdNXOy5zGmHOBiEoK94hFmEWVE3sdDuFtcMBX6hoDOw9+u5v2/5Rzv+DzJGVWzs= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927858; c=relaxed/simple; bh=GAHxE0g1mYsfBXyYnq1OjQb+2BFDikwZ76iBdw1352s=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=mr8JtAPr4t1UeTbvhZzTB4d+5qG9LoearPGp03NeThJzMg72fcAHWysJGv++P6K+w5j4Hy+48vNZXnCoZQ7ZxNS4R7atmxkakVQ9534uFLn2qg8oZugWO9ezAUnklhoqASPNhPZTxo4EDxhkD4b+4xISkMTswXoVo4PNC9vI7W4= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=K8Z83p4x; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="K8Z83p4x" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id 76FD220B716A; Wed, 5 Aug 2026 04:03:55 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 76FD220B716A DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927835; bh=MHHLJUnI1oOnr50BM/bsHyy7S5l3AfW3+3Y7ZXO6LCg=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=K8Z83p4xJKlsVl/kF8Jo2P9VGweUXgwE9Y9Srnefmcu4/DwQTPbpPWJJdxhBH9l3J 3u8mhfl/4Q14nhYM8Jvc7DZK/T+KLJkMJ1q+W8hvaIQGOKialBoiaSkp6vKUSUY9nU 5UlL5u8Gj5iwG2qZK4b0W4mCw+0WfRE2QHChDbDE= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 40/42] kvm: x86: use kvm_vcpu scheduling-state accessors and struct stat fields Date: Wed, 5 Aug 2026 04:03:22 -0700 Message-ID: <20260805110324.25067-41-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Finish the conversion begun by the kvm_vcpu_common rework: read wants_to_run through kvm_vcpu_wants_to_run() and account statistics via the embedded vcpu->stat.* fields (not the removed vcpu->stat-> pointer) in the SVM and nested-VMX paths. --- arch/x86/kvm/svm/svm.c | 4 ++-- arch/x86/kvm/vmx/nested.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/x86/kvm/svm/svm.c b/arch/x86/kvm/svm/svm.c index 73d80f442a49..44abbd629a66 100644 --- a/arch/x86/kvm/svm/svm.c +++ b/arch/x86/kvm/svm/svm.c @@ -2816,7 +2816,7 @@ static bool svm_pat_accesses_gpat(struct kvm_vcpu *vc= pu, bool from_host) * KVM_GET/SET_NESTED_STATE are independent of each other and can * be ordered arbitrarily during save and restore. */ - WARN_ON_ONCE(from_host && vcpu->wants_to_run); + WARN_ON_ONCE(from_host && kvm_vcpu_wants_to_run(vcpu)); return !from_host && is_guest_mode(vcpu) && l2_has_separate_pat(vcpu); } =20 @@ -3240,7 +3240,7 @@ static int interrupt_window_interception(struct kvm_v= cpu *vcpu) kvm_make_request(KVM_REQ_EVENT, vcpu); svm_clear_vintr(to_svm(vcpu)); =20 - ++vcpu->stat->irq_window_exits; + ++vcpu->stat.irq_window_exits; return 1; } =20 diff --git a/arch/x86/kvm/vmx/nested.c b/arch/x86/kvm/vmx/nested.c index ddf6df7bee93..2f504fa2e09a 100644 --- a/arch/x86/kvm/vmx/nested.c +++ b/arch/x86/kvm/vmx/nested.c @@ -619,7 +619,7 @@ static int nested_vmx_check_tpr_shadow_controls(struct = kvm_vcpu *vcpu, * and only perform the check when in KVM_RUN, to avoid a false failure * if userspace hasn't yet configured memslots during state restore. */ - if (warn_on_missed_cc && vcpu->wants_to_run && + if (warn_on_missed_cc && kvm_vcpu_wants_to_run(vcpu) && nested_cpu_has(vmcs12, CPU_BASED_TPR_SHADOW) && !nested_cpu_has_vid(vmcs12) && !nested_cpu_has2(vmcs12, SECONDARY_EXEC_VIRTUALIZE_APIC_ACCESSES) && --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 0A6AA46D09D; Wed, 5 Aug 2026 11:04:17 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927859; cv=none; b=lsaGNv0Trneo6UZm/URs4KrgwPFHkkYeHectjiRtzLIkzTyUaiXZTcTbKGQrrYvBfBRBVk/CX0a/1Qf76ywZw9AfPfN3iafNCtjwCsLpN9PHJ0U992q2jxHqUXRzIKXubQ5dis8wykffc6sQnPrt+o/w9O4gNHK3AEwV6C2TscU= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927859; c=relaxed/simple; bh=b6vgVm6RVK+tWU7Cnq/kwRlgOp/aiCgLOzkYg61p8pk=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=HsaHong+Wlr67N77++hGclpvj/wut06FiWr6e6s4EnRjJH4oADR3XXxftTOsPEafKohkrYFAq0jbiV586wdJIOO6BPex3FqO5/yRV3XfkKdIEuUh+68Ckup0KSYSkRrVHrZb/wuEUNJmMYYlOg6XKunllvR56nL9KWox3Ix5X+M= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=fsomZdNq; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="fsomZdNq" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id E654520B716B; Wed, 5 Aug 2026 04:03:55 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com E654520B716B DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927836; bh=kcbPiVYhmTkNV/i4qwxJbLgGTIdkavUIqaLTCvU7H4Q=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=fsomZdNqRIkAX1VHRyHMmzo6JnxZ2VyrpSWQ9rdp845pbXl/nMBUDnj1Tl6o50GbZ y0J6IDK0k4Ad+36Gw2bvgt72+ef0QR6XPEYTH2D5lhGNXJ6DSXOm2m01NT95lr5mxp uDKiOv4iCKe6ILDgqeIGyKyQ/YNC7DvU4ciNNkiU= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 41/42] kvm: x86: finalize per-plane APIC state and CPUID placement Date: Wed, 5 Aug 2026 04:03:23 -0700 Message-ID: <20260805110324.25067-42-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Adopt the vm-planes-merged design for x86 plane state: keep CPUID and cpu_caps in kvm_vcpu_arch_common, make apic_map and APICv-inhibit tracking VM-scoped again, and drop the superseded intermediate fields (planes_share_fpu, irr_pending_planes, kvm_arch_plane.apicv_inhibit_reasons, kvm_lapic_irq.plane). These changes originated in merge-commit conflict resolutions that a --no-merges linearization could not carry over. --- arch/x86/include/asm/kvm_host.h | 36 +++++----------------- arch/x86/kvm/cpuid.c | 19 +----------- arch/x86/kvm/hyperv.c | 1 - arch/x86/kvm/i8254.c | 4 +-- arch/x86/kvm/lapic.c | 53 +++++++-------------------------- arch/x86/kvm/xen.c | 1 - 6 files changed, 20 insertions(+), 94 deletions(-) diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_hos= t.h index bbccb9d3d801..b1a7e4ca8870 100644 --- a/arch/x86/include/asm/kvm_host.h +++ b/arch/x86/include/asm/kvm_host.h @@ -956,12 +956,6 @@ struct kvm_vcpu_arch { u64 ia32_xss; u64 guest_supported_xss; =20 - /* - * Only valid in plane0. The bitmask of planes that received - * an interrupt, to be checked against req_exit_planes. - */ - atomic_t irr_pending_planes; - struct kvm_pio_request pio; void *pio_data; void *sev_pio_data; @@ -1175,10 +1169,6 @@ struct kvm_arch_memory_slot { unsigned short *gfn_write_track; }; =20 -struct kvm_arch_plane { - unsigned long apicv_inhibit_reasons; -}; - /* * Track the mode of the optimized logical map, as the rules for decoding = the * destination vary per mode. Enabling the optimized logical map requires= all @@ -1397,13 +1387,11 @@ enum kvm_apicv_inhibit { /* * PIT (i8254) 're-inject' mode, relies on EOI intercept, * which AVIC doesn't support for edge triggered interrupts. - * Applied only to plane 0. */ APICV_INHIBIT_REASON_PIT_REINJ, =20 /* - * AVIC is disabled because SEV doesn't support it. Sticky and applied - * only to plane 0. + * AVIC is disabled because SEV doesn't support it. */ APICV_INHIBIT_REASON_SEV, =20 @@ -1483,7 +1471,6 @@ struct kvm_arch { unsigned int indirect_shadow_pages; u8 mmu_valid_gen; u8 vm_type; - bool planes_share_fpu; bool has_private_mem; bool has_protected_state; bool has_protected_eoi; @@ -1805,7 +1792,6 @@ struct kvm_lapic_irq { u16 delivery_mode; u16 dest_mode; bool level; - u8 plane; u16 trig_mode; u32 shorthand; u32 dest_id; @@ -2399,21 +2385,21 @@ gpa_t kvm_mmu_gva_to_gpa_system(struct kvm_vcpu *vc= pu, gva_t gva, bool kvm_apicv_activated(struct kvm *kvm); bool kvm_vcpu_apicv_activated(struct kvm_vcpu *vcpu); void __kvm_vcpu_update_apicv(struct kvm_vcpu *vcpu); -void __kvm_set_or_clear_apicv_inhibit(struct kvm_plane *plane, +void __kvm_set_or_clear_apicv_inhibit(struct kvm *kvm, enum kvm_apicv_inhibit reason, bool set); -void kvm_set_or_clear_apicv_inhibit(struct kvm_plane *plane, +void kvm_set_or_clear_apicv_inhibit(struct kvm *kvm, enum kvm_apicv_inhibit reason, bool set); =20 -static inline void kvm_set_apicv_inhibit(struct kvm_plane *plane, +static inline void kvm_set_apicv_inhibit(struct kvm *kvm, enum kvm_apicv_inhibit reason) { - kvm_set_or_clear_apicv_inhibit(plane, reason, true); + kvm_set_or_clear_apicv_inhibit(kvm, reason, true); } =20 -static inline void kvm_clear_apicv_inhibit(struct kvm_plane *plane, +static inline void kvm_clear_apicv_inhibit(struct kvm *kvm, enum kvm_apicv_inhibit reason) { - kvm_set_or_clear_apicv_inhibit(plane, reason, false); + kvm_set_or_clear_apicv_inhibit(kvm, reason, false); } =20 void kvm_inc_or_dec_irq_window_inhibit(struct kvm *kvm, bool inc); @@ -2503,8 +2489,6 @@ enum { # define kvm_memslots_for_spte_role(kvm, role) __kvm_memslots(kvm, 0) #endif =20 -#define KVM_MAX_VCPU_PLANES 16 - int kvm_cpu_has_injectable_intr(struct kvm_vcpu *v); int kvm_cpu_has_interrupt(struct kvm_vcpu *vcpu); int kvm_cpu_has_extint(struct kvm_vcpu *v); @@ -2539,9 +2523,6 @@ void kvm_make_scan_ioapic_request(struct kvm *kvm); void kvm_make_scan_ioapic_request_mask(struct kvm *kvm, unsigned long *vcpu_bitmap); =20 -void kvm_arch_init_plane(struct kvm_plane *plane); -void kvm_arch_free_plane(struct kvm_plane *plane); - bool kvm_arch_async_page_not_present(struct kvm_vcpu *vcpu, struct kvm_async_pf *work); void kvm_arch_async_page_present(struct kvm_vcpu *vcpu, @@ -2612,7 +2593,4 @@ static inline bool kvm_arch_has_irq_bypass(void) return enable_device_posted_irqs; } =20 -int kvm_arch_nr_vcpu_planes(struct kvm *kvm); -bool kvm_arch_planes_share_fpu(struct kvm *kvm); - #endif /* _ASM_X86_KVM_HOST_H */ diff --git a/arch/x86/kvm/cpuid.c b/arch/x86/kvm/cpuid.c index ce337c6d3bcf..7b8cd379ba9f 100644 --- a/arch/x86/kvm/cpuid.c +++ b/arch/x86/kvm/cpuid.c @@ -555,7 +555,7 @@ static int kvm_set_cpuid(struct kvm_vcpu *vcpu, struct = kvm_cpuid_entry2 *e2, * KVM_SET_CPUID{,2} again. To support this legacy behavior, check * whether the supplied CPUID data is equal to what's already set. */ - if (!kvm_can_set_cpuid_and_feature_msrs(vcpu) || vcpu->has_planes) { + if (!kvm_can_set_cpuid_and_feature_msrs(vcpu)) { r =3D kvm_cpuid_check_equal(vcpu, e2, nent); if (r) goto err; @@ -594,23 +594,6 @@ static int kvm_set_cpuid(struct kvm_vcpu *vcpu, struct= kvm_cpuid_entry2 *e2, return r; } =20 -int kvm_dup_cpuid(struct kvm_vcpu *vcpu, struct kvm_vcpu *source) -{ - if (WARN_ON_ONCE(vcpu->arch.cpuid_entries || vcpu->arch.cpuid_nent)) - return -EEXIST; - - vcpu->arch.cpuid_entries =3D kmemdup(source->arch.cpuid_entries, - source->arch.cpuid_nent * sizeof(struct kvm_cpuid_entry2), - GFP_KERNEL_ACCOUNT); - if (!vcpu->arch.cpuid_entries) - return -ENOMEM; - - memcpy(vcpu->arch.cpu_caps, source->arch.cpu_caps, sizeof(source->arch.cp= u_caps)); - vcpu->arch.cpuid_nent =3D source->arch.cpuid_nent; - - return 0; -} - /* when an old userspace process fills a new kernel module */ int kvm_vcpu_ioctl_set_cpuid(struct kvm_vcpu *vcpu, struct kvm_cpuid *cpuid, diff --git a/arch/x86/kvm/hyperv.c b/arch/x86/kvm/hyperv.c index 8ef09b8125b7..ee6b32d2a5cb 100644 --- a/arch/x86/kvm/hyperv.c +++ b/arch/x86/kvm/hyperv.c @@ -491,7 +491,6 @@ static int synic_set_irq(struct kvm_vcpu_hv_synic *syni= c, u32 sint) irq.delivery_mode =3D APIC_DM_FIXED; irq.vector =3D vector; irq.level =3D 1; - ret =3D kvm_irq_delivery_to_apic(vcpu->plane, vcpu->arch.apic, &irq); trace_kvm_hv_synic_set_irq(vcpu->vcpu_id, sint, irq.vector, ret); return ret; diff --git a/arch/x86/kvm/i8254.c b/arch/x86/kvm/i8254.c index cd47fd88c9f7..bfe590378bd2 100644 --- a/arch/x86/kvm/i8254.c +++ b/arch/x86/kvm/i8254.c @@ -305,13 +305,13 @@ static void kvm_pit_set_reinject(struct kvm_pit *pit,= bool reinject) * So, deactivate APICv when PIT is in reinject mode. */ if (reinject) { - kvm_set_apicv_inhibit(kvm->planes[0], APICV_INHIBIT_REASON_PIT_REINJ); + kvm_set_apicv_inhibit(kvm, APICV_INHIBIT_REASON_PIT_REINJ); /* The initial state is preserved while ps->reinject =3D=3D 0. */ kvm_pit_reset_reinject(pit); kvm_register_irq_ack_notifier(kvm, &ps->irq_ack_notifier); kvm_register_irq_mask_notifier(kvm, 0, &pit->mask_notifier); } else { - kvm_clear_apicv_inhibit(kvm->planes[0], APICV_INHIBIT_REASON_PIT_REINJ); + kvm_clear_apicv_inhibit(kvm, APICV_INHIBIT_REASON_PIT_REINJ); kvm_unregister_irq_ack_notifier(kvm, &ps->irq_ack_notifier); kvm_unregister_irq_mask_notifier(kvm, 0, &pit->mask_notifier); } diff --git a/arch/x86/kvm/lapic.c b/arch/x86/kvm/lapic.c index 4cca1ea6a16e..ff923133a834 100644 --- a/arch/x86/kvm/lapic.c +++ b/arch/x86/kvm/lapic.c @@ -405,7 +405,6 @@ enum { =20 static void kvm_recalculate_apic_map(struct kvm_plane *plane) { - struct kvm_plane *plane =3D kvm->planes[0]; struct kvm_apic_map *new, *old =3D NULL; struct kvm *kvm =3D plane->kvm; struct kvm_vcpu *vcpu; @@ -486,19 +485,19 @@ static void kvm_recalculate_apic_map(struct kvm_plane= *plane) * map also applies to APICv. */ if (!new) - kvm_set_apicv_inhibit(plane, APICV_INHIBIT_REASON_PHYSICAL_ID_ALIASED); + kvm_set_apicv_inhibit(kvm, APICV_INHIBIT_REASON_PHYSICAL_ID_ALIASED); else - kvm_clear_apicv_inhibit(plane, APICV_INHIBIT_REASON_PHYSICAL_ID_ALIASED); + kvm_clear_apicv_inhibit(kvm, APICV_INHIBIT_REASON_PHYSICAL_ID_ALIASED); =20 if (!new || new->logical_mode =3D=3D KVM_APIC_MODE_MAP_DISABLED) - kvm_set_apicv_inhibit(plane, APICV_INHIBIT_REASON_LOGICAL_ID_ALIASED); + kvm_set_apicv_inhibit(kvm, APICV_INHIBIT_REASON_LOGICAL_ID_ALIASED); else - kvm_clear_apicv_inhibit(plane, APICV_INHIBIT_REASON_LOGICAL_ID_ALIASED); + kvm_clear_apicv_inhibit(kvm, APICV_INHIBIT_REASON_LOGICAL_ID_ALIASED); =20 if (xapic_id_mismatch) - kvm_set_apicv_inhibit(plane, APICV_INHIBIT_REASON_APIC_ID_MODIFIED); + kvm_set_apicv_inhibit(kvm, APICV_INHIBIT_REASON_APIC_ID_MODIFIED); else - kvm_clear_apicv_inhibit(plane, APICV_INHIBIT_REASON_APIC_ID_MODIFIED); + kvm_clear_apicv_inhibit(kvm, APICV_INHIBIT_REASON_APIC_ID_MODIFIED); =20 old =3D rcu_dereference_protected(plane->arch.apic_map, lockdep_is_held(&plane->arch.apic_map_lock)); @@ -1396,39 +1395,6 @@ int __kvm_irq_delivery_to_apic(struct kvm_plane *pla= ne, struct kvm_lapic *src, return r; } =20 -static void kvm_lapic_deliver_interrupt(struct kvm_vcpu *vcpu, struct kvm_= lapic *apic, - int delivery_mode, int trig_mode, int vector) -{ - struct kvm_vcpu *plane0_vcpu =3D vcpu->plane0; - struct kvm_plane *running_plane; - u16 req_exit_planes; - - kvm_x86_call(deliver_interrupt)(apic, delivery_mode, trig_mode, vector); - - /* - * test_and_set_bit implies a memory barrier, so IRR is written before - * reading irr_pending_planes below... - */ - if (!test_and_set_bit(vcpu->plane, &plane0_vcpu->arch.irr_pending_planes)= ) { - /* - * ... and also running_plane and req_exit_planes are read after writing - * irr_pending_planes. Both barriers pair with kvm_arch_vcpu_ioctl_run(= ). - */ - smp_mb__after_atomic(); - - running_plane =3D READ_ONCE(plane0_vcpu->running_plane); - if (!running_plane) - return; - - req_exit_planes =3D READ_ONCE(plane0_vcpu->req_exit_planes); - if (!(req_exit_planes & BIT(vcpu->plane))) - return; - - kvm_make_request(KVM_REQ_PLANE_INTERRUPT, - kvm_get_plane_vcpu(running_plane, vcpu->vcpu_id)); - } -} - /* * Add a pending IRQ into lapic. * Return 1 if successfully added and 0 if discarded. @@ -1470,7 +1436,8 @@ static int __apic_accept_irq(struct kvm_lapic *apic, = int delivery_mode, apic_clear_vector(vector, apic->regs + APIC_TMR); } =20 - kvm_lapic_deliver_interrupt(vcpu, apic, delivery_mode, trig_mode, vector= ); + kvm_x86_call(deliver_interrupt)(apic, delivery_mode, + trig_mode, vector); break; =20 case APIC_DM_REMRD: @@ -2087,7 +2054,7 @@ static void apic_timer_expired(struct kvm_lapic *apic= , bool from_timer_fn) if (apic_lvtt_tscdeadline(apic) || ktimer->hv_timer_in_use) ktimer->expired_tscdeadline =3D ktimer->tscdeadline; =20 - if (!from_timer_fn && apic->apicv_active && vcpu->wants_to_run) { + if (!from_timer_fn && apic->apicv_active && kvm_vcpu_wants_to_run(vcpu)) { WARN_ON(kvm_get_running_vcpu() !=3D vcpu); kvm_apic_inject_pending_timer_irqs(apic); return; @@ -2867,7 +2834,7 @@ static void __kvm_apic_set_base(struct kvm_vcpu *vcpu= , u64 value) =20 if ((value & MSR_IA32_APICBASE_ENABLE) && apic->base_address !=3D APIC_DEFAULT_PHYS_BASE) { - kvm_set_apicv_inhibit(vcpu_to_plane(vcpu), + kvm_set_apicv_inhibit(apic->vcpu->kvm, APICV_INHIBIT_REASON_APIC_BASE_MODIFIED); } } diff --git a/arch/x86/kvm/xen.c b/arch/x86/kvm/xen.c index 399406752108..4527f04c6617 100644 --- a/arch/x86/kvm/xen.c +++ b/arch/x86/kvm/xen.c @@ -625,7 +625,6 @@ void kvm_xen_inject_vcpu_vector(struct kvm_vcpu *v) irq.shorthand =3D APIC_DEST_NOSHORT; irq.delivery_mode =3D APIC_DM_FIXED; irq.level =3D 1; - kvm_irq_delivery_to_apic(v->plane, NULL, &irq); } =20 --=20 2.55.0 From nobody Fri Oct 2 04:27:22 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 4B98046D2DC; Wed, 5 Aug 2026 11:04:18 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927861; cv=none; b=JjGAN5Aw4Vdz6xF3OeOhzMi7dcOlXaTxJ1bGxKwfxm9eSzp0iKGV9i6d/cT721+92wsTp0TKP9zj9R7NBnK28U9WaocY5Eb3LbgKKlJ1A9623pcaGklwp0C03ZAgmL3z7KUbUQqxk3ju538appeCKW+7lvZnuhhYsY47rS6nRis= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785927861; c=relaxed/simple; bh=B3Jxs51JdpOLNuoiHj+NQ4cy2Io/khf5jp/0C3Oso7w=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=O2KH9NLw9ZlXOQ5k4cMk65/DFGWTEdHSIhkd92OC3eg1I76kqSVn1lqC+Xg9yYWRYnGXlrdYK/B8x0nfHGiZ0mofTEWkSLFiGZzbt2Q4+YXZvlB2WBh138zwhSlkknBuCm0sUdIAmabjfHa5CiELO2IBLGLvDmoufBf1tLAQDno= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=Exn2+wgE; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="Exn2+wgE" Received: from fedora.hsd1.wa.comcast.net (unknown [52.148.140.42]) by linux.microsoft.com (Postfix) with ESMTPSA id B78F920B716C; Wed, 5 Aug 2026 04:03:56 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com B78F920B716C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1785927836; bh=9DGQDZMGnVdkLT+T7BFikYvxxukUnodYkNvYZab2Vlo=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=Exn2+wgEZEQjnftki+F4cbY08suL61hBripewKHY+1bvdTNJJ8FIcCW+VnTXs3f9V iqpq9YgXEResAzTTxSZ7GFTRwd4hLJiAx+ZG7cT7R48Yig5xxZh51ZUGzrh2ulGC2c joeoC/FcthRR2u1f4Z5N2FRdTiOC3B2h/L2dix1c= From: Sriram Nambakam To: kvm@vger.kernel.org Cc: linux-kernel@vger.kernel.org Subject: [RFC PATCH v1 42/42] kvm: planes: reconcile core plane state, UAPI and hypercall exit Date: Wed, 5 Aug 2026 04:03:24 -0700 Message-ID: <20260805110324.25067-43-snambakam@linux.microsoft.com> X-Mailer: git-send-email 2.55.0 In-Reply-To: <20260805110324.25067-1-snambakam@linux.microsoft.com> References: <20260805110324.25067-1-snambakam@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Align the generic plane core and its userspace ABI with vm-planes-merged: the kvm_vcpu_common/kvm_plane layout and helpers, the KVM_CAP_PLANES / KVM_EXIT_PLANE_EVENT definitions and documentation, and the x86 handling that exits VM-plane and VBS hypercalls to userspace. These deltas came from the integration branch's merge-commit conflict resolutions. --- Documentation/virt/kvm/api.rst | 35 +++++++------- arch/x86/kvm/svm/sev.c | 5 +- arch/x86/kvm/x86.c | 86 ++++++++++------------------------ include/linux/kvm_host.h | 8 ++-- include/uapi/linux/kvm.h | 23 +-------- virt/kvm/kvm_main.c | 86 +++++++++++++++------------------- 6 files changed, 86 insertions(+), 157 deletions(-) diff --git a/Documentation/virt/kvm/api.rst b/Documentation/virt/kvm/api.rst index c6b109fa8945..269be00c8dcf 100644 --- a/Documentation/virt/kvm/api.rst +++ b/Documentation/virt/kvm/api.rst @@ -9101,27 +9101,20 @@ helpful if user space wants to emulate instructions= which are not This capability can be enabled dynamically even if VCPUs were already created and are running. =20 -hpage_2g module parameter is not set to 1, -EINVAL is returned. - -7.47 KVM_CAP_PLANES_FPU ------------------------ - -:Architectures: x86 -:Parameters: arg[0] is 0 if each vCPU plane has a separate FPU, - 1 if the FPU is shared -:Type: vm +7.47 KVM_CAP_S390_HPAGE_2G +-------------------------- =20 -When enabled, such as KVM_SET_XSAVE or KVM_SET_FPU *are* available for -vCPU on all planes, but they will read and write the same data that is pre= sented -to other planes. Note that KVM_GET/SET_XSAVE also allows access to some -registers that are *not* part of FPU state; right now this is just PKRU. -Those are never shared. +:Architectures: s390 +:Parameters: none +:Returns: 0 on success; -EINVAL if hpage_2g module parameter was not set, + cmma is enabled, or the VM has the KVM_VM_S390_UCONTROL + flag set; -EBUSY if vCPUs were already created for the VM. =20 -KVM_CAP_PLANES_FPU is experimental; userspace must *not* assume that -KVM_CAP_PLANES_FPU is present on x86 for *any* VM type and different -VM types may or may not allow enabling KVM_CAP_PLANES_FPU. Like for other -capabilities, KVM_CAP_PLANES_FPU can be queried on the VM file descriptor; -KVM_CHECK_EXTENSION returns 1 if it is possible to enable shared FPU mode. +With this capability the KVM support for memory backing with 2g pages +through hugetlbfs can be enabled for a VM. After the capability is +enabled, cmma can't be enabled anymore and pfmfi and the storage key +interpretation are disabled. If cmma has already been enabled or the +hpage_2g module parameter is not set to 1, -EINVAL is returned. =20 8. Other capabilities. =3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D=3D @@ -9674,6 +9667,10 @@ check for this capability on the VM file descriptor. When called on the system file descriptor, KVM returns the highest value supported on any machine type. =20 +When called on a plane file descriptor, KVM returns 0, because a +plane cannot host planes of its own. Other capabilities are +forwarded to the plane's parent VM. + 8.47 KVM_CAP_S390_VSIE_ESAMODE ------------------------------ =20 diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index b9b0bbb72394..b94de3b8967a 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -4488,7 +4488,7 @@ static void sev_get_apic_ids(struct vcpu_svm *svm) desc->num_entries =3D n; kvm_for_each_vcpu(i, loop_vcpu, kvm) { /*TODO: is this possible? */ - if (i > n) + if (i >=3D n) break; =20 desc->apic_ids[i] =3D loop_vcpu->vcpu_id; @@ -4713,6 +4713,9 @@ static bool is_snp_only_vmgexit(u64 exit_code) case SVM_VMGEXIT_GUEST_REQUEST: case SVM_VMGEXIT_EXT_GUEST_REQUEST: case SVM_VMGEXIT_PSC: + case SVM_VMGEXIT_HVDB_PAGE: + case SVM_VMGEXIT_HV_IPI: + case SVM_VMGEXIT_SNP_RUN_VMPL: return true; default: return false; diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c index c8c37d569023..d80b1caefc70 100644 --- a/arch/x86/kvm/x86.c +++ b/arch/x86/kvm/x86.c @@ -517,31 +517,6 @@ void kvm_free_plane(struct kvm_plane *plane) kvm_x86_call(free_plane)(plane); } =20 -struct kvm_plane *x86_alloc_plane(void) -{ - /* For better type checking, do not return kzalloc() value directly */ - struct kvm_plane *plane =3D kzalloc(sizeof(*plane), GFP_KERNEL_ACCOUNT); - - return plane; -} -EXPORT_SYMBOL_FOR_KVM_INTERNAL(x86_alloc_plane); - -void x86_free_plane(struct kvm_plane *plane) -{ - kfree(plane); -} -EXPORT_SYMBOL_FOR_KVM_INTERNAL(x86_free_plane); - -struct kvm_plane *kvm_alloc_plane(void) -{ - return kvm_x86_call(alloc_plane)(); -} - -void kvm_free_plane(struct kvm_plane *plane) -{ - kvm_x86_call(free_plane)(plane); -} - /* * All feature MSRs except uCode revID, which tracks the currently loaded = uCode * patch, are immutable once the vCPU model is defined. @@ -1026,7 +1001,7 @@ static int complete_emulated_insn_gp(struct kvm_vcpu = *vcpu, int err) void kvm_inject_page_fault(struct kvm_vcpu *vcpu, struct x86_exception *fa= ult, bool from_hardware) { - ++vcpu->stat->pf_guest; + ++vcpu->stat.pf_guest; =20 /* * Async #PF in L2 is always forwarded to L1 as a VM-Exit regardless of @@ -3732,7 +3707,7 @@ static void kvmclock_reset(struct kvm_vcpu *vcpu) =20 static void kvm_vcpu_flush_tlb_all(struct kvm_vcpu *vcpu) { - ++vcpu->stat->tlb_flush; + ++vcpu->stat.tlb_flush; kvm_x86_call(flush_tlb_all)(vcpu); =20 /* Flushing all ASIDs flushes the current ASID... */ @@ -3741,7 +3716,7 @@ static void kvm_vcpu_flush_tlb_all(struct kvm_vcpu *v= cpu) =20 static void kvm_vcpu_flush_tlb_guest(struct kvm_vcpu *vcpu) { - ++vcpu->stat->tlb_flush; + ++vcpu->stat.tlb_flush; =20 if (!tdp_enabled) { /* @@ -3766,7 +3741,7 @@ static void kvm_vcpu_flush_tlb_guest(struct kvm_vcpu = *vcpu) =20 static inline void kvm_vcpu_flush_tlb_current(struct kvm_vcpu *vcpu) { - ++vcpu->stat->tlb_flush; + ++vcpu->stat.tlb_flush; kvm_x86_call(flush_tlb_current)(vcpu); } =20 @@ -5305,11 +5280,11 @@ static void kvm_steal_time_set_preempted(struct kvm= _vcpu *vcpu) * preempted if and only if the VM-Exit was due to a host interrupt. */ if (!vcpu->arch.at_instruction_boundary) { - vcpu->stat->preemption_other++; + vcpu->stat.preemption_other++; return; } =20 - vcpu->stat->preemption_reported++; + vcpu->stat.preemption_reported++; if (!(vcpu->arch.st.msr_val & KVM_MSR_ENABLED)) return; =20 @@ -6845,7 +6820,7 @@ int kvm_vm_ioctl_enable_cap(struct kvm *kvm, r =3D -EEXIST; if (irqchip_in_kernel(kvm) || kvm->has_planes) goto split_irqchip_unlock; - if (kvm->created_vcpus || kvm->has_planes) + if (kvm->created_vcpus) goto split_irqchip_unlock; /* Pairs with irqchip_in_kernel. */ smp_wmb(); @@ -9278,7 +9253,7 @@ static int handle_emulation_failure(struct kvm_vcpu *= vcpu, int emulation_type) { struct kvm *kvm =3D vcpu->kvm; =20 - ++vcpu->stat->insn_emulation_fail; + ++vcpu->stat.insn_emulation_fail; trace_kvm_emulate_insn_failed(vcpu); =20 if (emulation_type & EMULTYPE_VMWARE_GP) { @@ -9510,7 +9485,7 @@ int x86_decode_emulated_instruction(struct kvm_vcpu *= vcpu, int emulation_type, r =3D x86_decode_insn(ctxt, insn, insn_len, emulation_type); =20 trace_kvm_emulate_insn_start(vcpu); - ++vcpu->stat->insn_emulation; + ++vcpu->stat.insn_emulation; =20 return r; } @@ -9685,7 +9660,7 @@ int x86_emulate_instruction(struct kvm_vcpu *vcpu, gp= a_t cr2_or_gpa, } r =3D 0; } else if (vcpu->mmio_needed) { - ++vcpu->stat->mmio_exits; + ++vcpu->stat.mmio_exits; =20 if (!vcpu->mmio_is_write) writeback =3D false; @@ -10452,7 +10427,7 @@ static void kvm_sched_yield(struct kvm_vcpu *vcpu, = unsigned long dest_id) struct kvm_vcpu *target =3D NULL; struct kvm_apic_map *map; =20 - vcpu->stat->directed_yield_attempted++; + vcpu->stat.directed_yield_attempted++; =20 if (single_task_running()) goto no_yield; @@ -10478,7 +10453,7 @@ static void kvm_sched_yield(struct kvm_vcpu *vcpu, = unsigned long dest_id) if (kvm_vcpu_yield_to(target) <=3D 0) goto no_yield; =20 - vcpu->stat->directed_yield_successful++; + vcpu->stat.directed_yield_successful++; =20 no_yield: return; @@ -10555,7 +10530,7 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vcpu= , int cpl, int op_64_bit =3D is_64_bit_hypercall(vcpu); unsigned long ret, nr, a0, a1, a2, a3; =20 - ++vcpu->stat->hypercalls; + ++vcpu->stat.hypercalls; =20 if (op_64_bit) { nr =3D kvm_rax_read_raw(vcpu); @@ -10673,7 +10648,7 @@ int ____kvm_emulate_hypercall(struct kvm_vcpu *vcpu= , int cpl, =20 if (common->vtl_plane_ready) { /* Parked in vtl_return: deliver now. */ - kvm_rax_write(secure, a0); + kvm_rax_write_raw(secure, a0); common->vtl_call_pending =3D false; } else { /* Still booting: deliver on readiness. */ @@ -11388,7 +11363,7 @@ void kvm_inc_or_dec_irq_window_inhibit(struct kvm *= kvm, bool inc) */ guard(rwsem_write)(&kvm->arch.apicv_update_lock); if (atomic_add_return(add, &kvm->arch.apicv_nr_irq_window_req) =3D=3D inc) - __kvm_set_or_clear_apicv_inhibit(kvm->planes[0], APICV_INHIBIT_REASON_IR= QWIN, inc); + __kvm_set_or_clear_apicv_inhibit(kvm, APICV_INHIBIT_REASON_IRQWIN, inc); } EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_inc_or_dec_irq_window_inhibit); =20 @@ -11644,22 +11619,9 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu) goto out; } =20 - if (kvm_check_plane0_events(vcpu)) { - kvm_vcpu_set_plane_runnable(vcpu->common->vcpus[0]); - - kvm_make_request(KVM_REQ_EVENT, vcpu); - kvm_make_request(KVM_REQ_PLANE_RESCHED, vcpu); - } - - if (kvm_check_request(KVM_REQ_PLANE_RESCHED, vcpu)) { - vcpu->common->plane_switch =3D true; - r =3D 0; - goto out; - } - if (kvm_check_request(KVM_REQ_EVENT, vcpu) || req_int_win || kvm_xen_has_interrupt(vcpu)) { - ++vcpu->stat->req_event; + ++vcpu->stat.req_event; r =3D kvm_apic_accept_events(vcpu); if (r < 0) { r =3D 0; @@ -11815,7 +11777,7 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu) run_flags =3D 0; =20 /* Note, VM-Exits that go down the "slow" path are accounted below. */ - ++vcpu->stat->exits; + ++vcpu->stat.exits; } =20 kvm_load_host_pkru(vcpu); @@ -11881,11 +11843,11 @@ static int vcpu_enter_guest(struct kvm_vcpu *vcpu) * VM-Exit on SVM and any ticks that occur between VM-Exit and now. * An instruction is required after local_irq_enable() to fully unblock * interrupts on processors that implement an interrupt shadow, the - * stat->exits increment will do nicely. + * stat.exits increment will do nicely. */ kvm_before_interrupt(vcpu, KVM_HANDLING_IRQ); local_irq_enable(); - ++vcpu->stat->exits; + ++vcpu->stat.exits; local_irq_disable(); kvm_after_interrupt(vcpu); =20 @@ -12103,7 +12065,7 @@ static int vcpu_run(struct kvm_vcpu *vcpu) kvm_vcpu_ready_for_interrupt_injection(vcpu)) { r =3D 0; vcpu->run->exit_reason =3D KVM_EXIT_IRQ_WINDOW_OPEN; - ++vcpu->stat->request_irq_exits; + ++vcpu->stat.request_irq_exits; break; } =20 @@ -12128,7 +12090,7 @@ static int __kvm_emulate_halt(struct kvm_vcpu *vcpu= , int state, int reason) * managed by userspace, in which case userspace is responsible for * handling wake events. */ - ++vcpu->stat->halt_exits; + ++vcpu->stat.halt_exits; if (lapic_in_kernel(vcpu)) { if (kvm_vcpu_has_events(vcpu) || vcpu->arch.pv.pv_unhalted) state =3D KVM_MP_STATE_RUNNABLE; @@ -12300,7 +12262,7 @@ static void kvm_put_guest_fpu(struct kvm_vcpu *vcpu) return; =20 fpu_swap_kvm_fpstate(&vcpu->arch.guest_fpu, false); - ++vcpu->stat->fpu_reload; + ++vcpu->stat.fpu_reload; trace_kvm_fpu(0); } =20 @@ -12387,7 +12349,7 @@ static int __kvm_arch_vcpu_ioctl_run(struct kvm_vcp= u *vcpu) if (signal_pending(current)) { r =3D -EINTR; kvm_run->exit_reason =3D KVM_EXIT_INTR; - ++vcpu->stat->signal_exits; + ++vcpu->stat.signal_exits; } goto out; } @@ -13180,7 +13142,7 @@ int kvm_arch_vcpu_precreate(struct kvm *kvm, unsign= ed int id) return 0; } =20 -int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu, struct kvm_plane *plane) +int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu) { int r; =20 diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h index 05c9edd4a73d..bee8eaea05bc 100644 --- a/include/linux/kvm_host.h +++ b/include/linux/kvm_host.h @@ -453,8 +453,7 @@ struct kvm_vcpu { #endif =20 struct kvm_vcpu_arch arch; - struct kvm_vcpu_stat *stat; - struct kvm_vcpu_stat __stat; + struct kvm_vcpu_stat stat; char stats_id[KVM_STATS_NAME_SIZE]; =20 /* @@ -1012,7 +1011,6 @@ struct kvm { bool dirty_ring_with_bitmap; bool vm_bugged; bool vm_dead; - bool has_planes; =20 #ifdef CONFIG_HAVE_KVM_PM_NOTIFIER struct notifier_block pm_notifier; @@ -1801,7 +1799,7 @@ int kvm_arch_vcpu_ioctl_run(struct kvm_vcpu *vcpu); void kvm_arch_vcpu_load(struct kvm_vcpu *vcpu, int cpu); void kvm_arch_vcpu_put(struct kvm_vcpu *vcpu); int kvm_arch_vcpu_precreate(struct kvm *kvm, unsigned int id); -int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu, struct kvm_plane *plane); +int kvm_arch_vcpu_create(struct kvm_vcpu *vcpu); void kvm_arch_vcpu_postcreate(struct kvm_vcpu *vcpu); void kvm_arch_vcpu_destroy(struct kvm_vcpu *vcpu); =20 @@ -2664,7 +2662,7 @@ static inline int kvm_arch_vcpu_run_pid_change(struct= kvm_vcpu *vcpu) static inline void kvm_handle_signal_exit(struct kvm_vcpu *vcpu) { vcpu->run->exit_reason =3D KVM_EXIT_INTR; - vcpu->stat->signal_exits++; + vcpu->stat.signal_exits++; } =20 static inline int kvm_xfer_to_guest_mode_handle_work(struct kvm_vcpu *vcpu) diff --git a/include/uapi/linux/kvm.h b/include/uapi/linux/kvm.h index 3118b31d13f6..fa2799c4dddb 100644 --- a/include/uapi/linux/kvm.h +++ b/include/uapi/linux/kvm.h @@ -140,16 +140,6 @@ struct kvm_xen_exit { } u; }; =20 -struct kvm_plane_event_exit { -#define KVM_PLANE_EVENT_INTERRUPT 1 - __u16 cause; - __u16 pending_event_planes; - __u16 target; - __u16 padding; - __u32 flags; - __u64 extra[8]; -}; - struct kvm_exit_snp_req_certs { __u64 gpa; __u64 npages; @@ -243,13 +233,7 @@ struct kvm_run { /* in */ __u8 request_interrupt_window; __u8 HINT_UNSAFE_IN_KVM(immediate_exit); - - /* in/out */ - __u8 plane; - __u16 suspended_planes; - - /* in */ - __u16 req_exit_planes; + __u8 padding1[6]; =20 /* out */ __u32 exit_reason; @@ -486,8 +470,6 @@ struct kvm_run { __u64 gpa; __u64 size; } memory_fault; - /* KVM_EXIT_PLANE_EVENT */ - struct kvm_plane_event_exit plane_event; /* KVM_EXIT_TDX */ struct { __u64 flags; @@ -1709,7 +1691,4 @@ struct kvm_pre_fault_memory { __u64 padding[5]; }; =20 -#define KVM_CREATE_PLANE _IO(KVMIO, 0xd6) -#define KVM_CREATE_VCPU_PLANE _IO(KVMIO, 0xd7) - #endif /* __LINUX_KVM_H */ diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c index 6e4f3f3e6881..6b2d272797d3 100644 --- a/virt/kvm/kvm_main.c +++ b/virt/kvm/kvm_main.c @@ -440,7 +440,7 @@ void *kvm_mmu_memory_cache_alloc(struct kvm_mmu_memory_= cache *mc) =20 static int kvm_vcpu_init_common(struct kvm_vcpu *vcpu, struct kvm *kvm, un= signed long id) { - struct kvm_vcpu_common *common =3D kzalloc(sizeof(*common), GFP_KERNEL_AC= COUNT); + struct kvm_vcpu_common *common __free(kfree) =3D kzalloc(sizeof(*common),= GFP_KERNEL_ACCOUNT); struct page *page; int r; =20 @@ -503,10 +503,7 @@ static int kvm_vcpu_init_common(struct kvm_vcpu *vcpu,= struct kvm *kvm, unsigned if (r) goto out_free_dirty_ring; =20 - vcpu->common =3D common; - - kvm_vcpu_set_in_spin_loop(vcpu, false); - kvm_vcpu_set_dy_eligible(vcpu, false); + vcpu->common =3D no_free_ptr(common); =20 kvm_vcpu_set_in_spin_loop(vcpu, false); kvm_vcpu_set_dy_eligible(vcpu, false); @@ -522,8 +519,6 @@ static int kvm_vcpu_init_common(struct kvm_vcpu *vcpu, = struct kvm *kvm, unsigned kvm->created_vcpus--; mutex_unlock(&kvm->lock); =20 - kfree(common); - return r; } =20 @@ -1243,7 +1238,6 @@ static struct kvm_plane *kvm_create_plane(struct kvm = *kvm, unsigned plane_level) if (kvm_arch_plane_init(kvm, plane, plane_level)) goto out_free_plane; =20 - kvm->planes[plane_level] =3D plane; =20 return plane; @@ -1490,6 +1484,7 @@ static void kvm_destroy_vm(struct kvm *kvm) #ifdef CONFIG_KVM_GENERIC_MEMORY_ATTRIBUTES xa_destroy(&kvm->mem_attr_array); #endif + kvm_destroy_planes(kvm); kvm_arch_free_vm(kvm); kvm_destroy_planes(kvm); preempt_notifier_dec(); @@ -4383,6 +4378,7 @@ static int kvm_plane_ioctl_create_vcpu(struct kvm_pla= ne *plane, unsigned long id { struct kvm *kvm =3D plane->kvm; struct kvm_vcpu *vcpu; + struct kvm_vcpu *prev_current_vcpu; int r; =20 mutex_lock(&kvm->lock); @@ -4427,7 +4423,25 @@ static int kvm_plane_ioctl_create_vcpu(struct kvm_pl= ane *plane, unsigned long id =20 kvm_vcpu_init(vcpu, kvm, id); =20 - r =3D kvm_arch_vcpu_create(vcpu, plane); + /* + * For planes above plane-0 the vCPU shares plane-0's kvm_vcpu_common, + * including ->current_vcpu and the preempt notifier consulted by + * kvm_sched_in()/kvm_sched_out(). kvm_arch_vcpu_create() (and + * kvm_arch_vcpu_postcreate() below) load this vCPU's VMCS via + * vcpu_load() but do not update ->current_vcpu, which still points at + * plane-0's vCPU. The arch create path performs GFP_KERNEL + * allocations, so the creating task can sleep and be rescheduled while + * this vCPU's VMCS is loaded; the shared notifier would then + * save/restore plane-0's vCPU and desync the per-CPU loaded_vmcs + * tracking from the hardware-current VMCS, wedging VMX (host hard + * lockup). Mirror the run loop's invariant (see + * kvm_vcpu_select_plane()): make ->current_vcpu the vCPU whose VMCS is + * loaded for the duration, then restore it. + */ + prev_current_vcpu =3D vcpu->common->current_vcpu; + vcpu->common->current_vcpu =3D vcpu; + r =3D kvm_arch_vcpu_create(vcpu); + vcpu->common->current_vcpu =3D prev_current_vcpu; if (r) goto vcpu_free_common; =20 @@ -4456,14 +4470,18 @@ static int kvm_plane_ioctl_create_vcpu(struct kvm_p= lane *plane, unsigned long id kvm_vcpu_unlock(vcpu); =20 mutex_unlock(&kvm->lock); + /* Same VMCS/current_vcpu invariant as above (vcpu_load in postcreate). */ + prev_current_vcpu =3D vcpu->common->current_vcpu; + vcpu->common->current_vcpu =3D vcpu; kvm_arch_vcpu_postcreate(vcpu); + vcpu->common->current_vcpu =3D prev_current_vcpu; kvm_create_vcpu_debugfs(vcpu); return r; =20 kvm_put_xa_erase: kvm_vcpu_unlock(vcpu); kvm_put_kvm_no_destroy(kvm); - xa_erase(&kvm->planes[0]->vcpu_array, vcpu->vcpu_idx); + xa_erase(&plane->vcpu_array, vcpu->vcpu_idx); unlock_vcpu_destroy: mutex_unlock(&kvm->lock); kvm_arch_vcpu_destroy(vcpu); @@ -4614,38 +4632,16 @@ static int kvm_wait_for_vcpu_online(struct kvm_vcpu= *vcpu) static inline bool kvm_is_vcpu_plane_ioctl(unsigned ioctl) { switch (ioctl) { - case KVM_GET_DEBUGREGS: - case KVM_SET_DEBUGREGS: case KVM_GET_FPU: case KVM_SET_FPU: - case KVM_GET_LAPIC: - case KVM_SET_LAPIC: - case KVM_GET_MSRS: - case KVM_SET_MSRS: - case KVM_GET_NESTED_STATE: - case KVM_SET_NESTED_STATE: - case KVM_GET_ONE_REG: - case KVM_SET_ONE_REG: case KVM_GET_REGS: case KVM_SET_REGS: case KVM_GET_SREGS: case KVM_SET_SREGS: - case KVM_GET_SREGS2: - case KVM_SET_SREGS2: - case KVM_GET_VCPU_EVENTS: - case KVM_SET_VCPU_EVENTS: - case KVM_GET_XCRS: - case KVM_SET_XCRS: - case KVM_GET_XSAVE: - case KVM_GET_XSAVE2: - case KVM_SET_XSAVE: - - case KVM_GET_REG_LIST: case KVM_TRANSLATE: return true; - default: - return false; + return kvm_arch_is_vcpu_plane_ioctl(ioctl); } } =20 @@ -4950,7 +4946,6 @@ static int kvm_vm_ioctl_check_extension_generic(struc= t kvm *kvm, long arg); =20 static long __kvm_plane_ioctl(struct kvm_plane *plane, unsigned int ioctl,= unsigned long arg) { - void __user *argp =3D (void __user *)arg; long r; =20 switch (ioctl) { @@ -4966,38 +4961,35 @@ static long __kvm_plane_ioctl(struct kvm_plane *pla= ne, unsigned int ioctl, unsig break; #ifdef CONFIG_HAVE_KVM_MSI case KVM_SIGNAL_MSI: { + void __user *argp =3D (void __user *)arg; struct kvm_msi msi; =20 - r =3D -EFAULT; if (copy_from_user(&msi, argp, sizeof(msi))) - goto out; + return -EFAULT; r =3D kvm_send_userspace_msi(plane->kvm, &msi, plane->level); break; } #endif #ifdef CONFIG_HAVE_KVM_IRQ_ROUTING case KVM_SET_GSI_ROUTING: { + void __user *argp =3D (void __user *)arg; struct kvm_irq_routing routing; struct kvm_irq_routing __user *urouting; struct kvm_irq_routing_entry *entries =3D NULL; =20 - r =3D -EFAULT; if (copy_from_user(&routing, argp, sizeof(routing))) - goto out; - r =3D -EINVAL; - if (!kvm_arch_can_set_irq_routing(plane->kvm)) - goto out; - if (routing.nr > KVM_MAX_IRQ_ROUTES) - goto out; - if (routing.flags) - goto out; + return -EFAULT; + if (!kvm_arch_can_set_irq_routing(plane->kvm) || + routing.nr > KVM_MAX_IRQ_ROUTES || + routing.flags) + return -EINVAL; if (routing.nr) { urouting =3D argp; entries =3D vmemdup_array_user(urouting->entries, routing.nr, sizeof(*entries)); if (IS_ERR(entries)) { r =3D PTR_ERR(entries); - goto out; + return r; } } r =3D kvm_set_irq_routing(plane->kvm, entries, routing.nr, @@ -5010,7 +5002,6 @@ static long __kvm_plane_ioctl(struct kvm_plane *plane= , unsigned int ioctl, unsig r =3D -ENOTTY; } =20 -out: return r; } =20 @@ -5590,7 +5581,6 @@ static int kvm_vm_ioctl_create_plane(struct kvm *kvm,= unsigned id) goto put_kvm; } =20 - kvm->planes[id] =3D plane; kvm->has_planes =3D true; fd_install(fd, file); return fd; --=20 2.55.0