From nobody Tue Sep 29 06:08:30 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id D51F545DF67; Tue, 11 Aug 2026 16:05:25 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786464327; cv=none; b=PW+LOWuUG++wAer2rKhzNJKc5eZ0NZBrJgI9HTX6EvdSPIHpEQfEylWJPbEEKwfJVz4Cr2iv915r+nyyjYOACMtfxHeLydCFqVWsevs7XFepaWhT9EESQhnG1QcuhYyLotwk9o13GOMNLsKhKXzxgi8faz9oFjODmpTX0mt3CZI= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786464327; c=relaxed/simple; bh=nYR4qaoV/dnmBKrZsYQKwoI5X9D4QY4yuPg0S5+KK8Q=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=GM/gA1odAjWFlAhZyeKrydFzDejhWT7+cNMnM+Ge+AneUNESGuh4Co8/bV0ymwyft37VVCG0hsw3DkCBxjAWlOGu+U7M1j8ix3WrF4U0EVLs3GzIIm9A0sUNJRy4AsWF2jx92ghg+GPbdhAxTCn+tbpsNfKCN8eyr5TCQrpOcUQ= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=N0h37BlD; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="N0h37BlD" Received: from linuxonhyperv3.guj3yctzbm1etfxqx2vob5hsef.xx.internal.cloudapp.net (linux.microsoft.com [13.77.154.182]) by linux.microsoft.com (Postfix) with ESMTPSA id 7F2CF20B7168; Tue, 11 Aug 2026 09:05:01 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 7F2CF20B7168 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1786464301; bh=GQiP7rNN/KQRNfzSV1VO0hhypWpn/hjzLyCBuN8NBD4=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=N0h37BlDAZEhKkT2OjLghAKByynCnTV/7LcGdDxWSJASUEdAF2xx5CDe53lY1CBYS 7M0UvI8XDXkNDFuiqtc9tiZfn0J1yRyB5aTx4hiIZOHiLSMzqdWnF3YMi92kVKKf5X D5iSkn41VQABkIZQmDYZSHqT7TYpnCLFCWDgEy4w= From: Kameron Carr To: decui@microsoft.com, haiyangz@microsoft.com, kys@microsoft.com, longli@microsoft.com, wei.liu@kernel.org, mhklinux@outlook.com Cc: andrew+netdev@lunn.ch, davem@davemloft.net, edumazet@google.com, kuba@kernel.org, pabeni@redhat.com, linux-hyperv@vger.kernel.org, linux-kernel@vger.kernel.org, netdev@vger.kernel.org Subject: [PATCH v4 1/3] Drivers: hv: vmbus: add vmbus_establish_gpadl_caller_decrypted() Date: Tue, 11 Aug 2026 09:04:45 -0700 Message-ID: <20260811160447.2529876-2-kameroncarr@linux.microsoft.com> X-Mailer: git-send-email 2.43.7 In-Reply-To: <20260811160447.2529876-1-kameroncarr@linux.microsoft.com> References: <20260811160447.2529876-1-kameroncarr@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Add a new vmbus_establish_gpadl_caller_decrypted() for callers that want to decrypt their own buffers. Add a new hv_gpadl_type, HV_GPADL_BUFFER_DECRYPTED, to communicate the decryption status of the buffer. No functional change for existing callers. Signed-off-by: Kameron Carr Reviewed-by: Michael Kelley --- drivers/hv/channel.c | 27 +++++++++++++++++++++++++-- include/linux/hyperv.h | 8 +++++++- 2 files changed, 32 insertions(+), 3 deletions(-) diff --git a/drivers/hv/channel.c b/drivers/hv/channel.c index 6821f225..4782f50 100644 --- a/drivers/hv/channel.c +++ b/drivers/hv/channel.c @@ -40,6 +40,7 @@ static inline u32 hv_gpadl_size(enum hv_gpadl_type type, = u32 size) { switch (type) { case HV_GPADL_BUFFER: + case HV_GPADL_BUFFER_DECRYPTED: return size; case HV_GPADL_RING: /* The size of a ringbuffer must be page-aligned */ @@ -100,6 +101,7 @@ static inline u64 hv_gpadl_hvpfn(enum hv_gpadl_type typ= e, void *kbuffer, =20 switch (type) { case HV_GPADL_BUFFER: + case HV_GPADL_BUFFER_DECRYPTED: break; case HV_GPADL_RING: if (i =3D=3D 0) @@ -460,7 +462,8 @@ static int __vmbus_establish_gpadl(struct vmbus_channel= *channel, } =20 gpadl->decrypted =3D !((channel->co_external_memory && type =3D=3D HV_GPA= DL_BUFFER) || - (channel->co_ring_buffer && type =3D=3D HV_GPADL_RING)); + (channel->co_ring_buffer && type =3D=3D HV_GPADL_RING) || + (type =3D=3D HV_GPADL_BUFFER_DECRYPTED)); if (gpadl->decrypted) { /* * The "decrypted" flag being true assumes that set_memory_decrypted() s= ucceeds. @@ -575,7 +578,7 @@ static int __vmbus_establish_gpadl(struct vmbus_channel= *channel, * @channel: a channel * @kbuffer: from kmalloc or vmalloc * @size: page-size multiple - * @gpadl_handle: some funky thing + * @gpadl: output gpadl */ int vmbus_establish_gpadl(struct vmbus_channel *channel, void *kbuffer, u32 size, struct vmbus_gpadl *gpadl) @@ -585,6 +588,26 @@ int vmbus_establish_gpadl(struct vmbus_channel *channe= l, void *kbuffer, } EXPORT_SYMBOL_GPL(vmbus_establish_gpadl); =20 +/* + * vmbus_establish_gpadl_caller_decrypted - Establish a GPADL for a buffer + * that has already been decrypted by the caller. + * + * @channel: a channel + * @kbuffer: from kmalloc or vmalloc; must already be decrypted by the cal= ler + * @size: page-size multiple + * @gpadl: output gpadl + * + * The caller is responsible for re-encrypting the buffer before freeing i= t. + */ +int vmbus_establish_gpadl_caller_decrypted(struct vmbus_channel *channel, + void *kbuffer, u32 size, + struct vmbus_gpadl *gpadl) +{ + return __vmbus_establish_gpadl(channel, HV_GPADL_BUFFER_DECRYPTED, + kbuffer, size, 0U, gpadl); +} +EXPORT_SYMBOL_GPL(vmbus_establish_gpadl_caller_decrypted); + /** * request_arr_init - Allocates memory for the requestor array. Each slot * keeps track of the next available slot in the array. Initially, each diff --git a/include/linux/hyperv.h b/include/linux/hyperv.h index 964f1be..1146add 100644 --- a/include/linux/hyperv.h +++ b/include/linux/hyperv.h @@ -70,7 +70,8 @@ */ enum hv_gpadl_type { HV_GPADL_BUFFER, - HV_GPADL_RING + HV_GPADL_RING, + HV_GPADL_BUFFER_DECRYPTED }; =20 /* Single-page buffer */ @@ -1205,6 +1206,11 @@ extern int vmbus_establish_gpadl(struct vmbus_channe= l *channel, u32 size, struct vmbus_gpadl *gpadl); =20 +extern int vmbus_establish_gpadl_caller_decrypted(struct vmbus_channel *ch= annel, + void *kbuffer, + u32 size, + struct vmbus_gpadl *gpadl); + extern int vmbus_teardown_gpadl(struct vmbus_channel *channel, struct vmbus_gpadl *gpadl); =20 --=20 2.45.4 From nobody Tue Sep 29 06:08:30 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id 746D545DF76; Tue, 11 Aug 2026 16:05:26 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786464328; cv=none; b=pQ6Ej/9rIcRUiVlJspiX2rcdGNLavuvfTlN8Qak7zz5J4wqFWxCGbB3OhjOm/5MUFKmhsMsoErQyTJtJ08KIkd0ofY5kGCogsuGj6HZwkbzKefTkSH3y1pJpYrzkslD1k9yA51ixM/ajFZci004M3tBQy58QIzXMrX0xGIhhYp0= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786464328; c=relaxed/simple; bh=oivuZ+iRBqyW+ICsgpfAD4wob/CWkTn2VkGdI+l2xL4=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=Efxm82F6Qid6fZBeF7RE1lOIoSZzaUroFxW0joxvoJ80ZTez37Sdh7Fb9p5O1atjEkG1o8Mus+mC9MRtAYZf5DHpQisJ9/q8H8dpbJxHOfhsqV3pAxQ446Mz30PZ8WQm1xA0b4Srq6PrkfOvb8/2xwScQAhg/hed9MenlFY8qQM= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=DLKpRcGL; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="DLKpRcGL" Received: from linuxonhyperv3.guj3yctzbm1etfxqx2vob5hsef.xx.internal.cloudapp.net (linux.microsoft.com [13.77.154.182]) by linux.microsoft.com (Postfix) with ESMTPSA id E028020B710C; Tue, 11 Aug 2026 09:05:01 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com E028020B710C DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1786464301; bh=HvUsQzIz2U6SyzAMN2miC4OpA6mCq+idX8hwcOPG0Ps=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=DLKpRcGLimVhjaHbVmuEm5rl3qtTQ+IrHpCLAtmekjo90n2mNhNa44S17pj6jrQm4 S1YByjE+86MjE89CT4PZXdRTqWEcs6ScJiGpbK6xIt2AqCwKC3FNCpPfdo200WXM+F /abmlepJX9z5qnMed4tfWAHPiUkTcCx22RJlg7dU= From: Kameron Carr To: decui@microsoft.com, haiyangz@microsoft.com, kys@microsoft.com, longli@microsoft.com, wei.liu@kernel.org, mhklinux@outlook.com Cc: andrew+netdev@lunn.ch, davem@davemloft.net, edumazet@google.com, kuba@kernel.org, pabeni@redhat.com, linux-hyperv@vger.kernel.org, linux-kernel@vger.kernel.org, netdev@vger.kernel.org Subject: [PATCH v4 2/3] Drivers: hv: vmbus: Add vmbus_alloc_buffer()/vmbus_free_buffer() for CoCo VMs Date: Tue, 11 Aug 2026 09:04:46 -0700 Message-ID: <20260811160447.2529876-3-kameroncarr@linux.microsoft.com> X-Mailer: git-send-email 2.43.7 In-Reply-To: <20260811160447.2529876-1-kameroncarr@linux.microsoft.com> References: <20260811160447.2529876-1-kameroncarr@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" On CoCo VMs without confidential VMBus, the netvsc send and receive buffers must be made host-visible by decrypting them. These buffers are vmalloc'ed, but set_memory_decrypted()/encrypted() do not work on vmalloc'ed memory. This use case is (so far) unique to netvsc, so solve it locally rather than changing the set_memory() or allocation APIs. Add vmbus_alloc_buffer()/vmbus_free_buffer() to the VMBus core. When the guest's isolation model requires it, allocate the buffer as a list of physically-contiguous chunks via alloc_pages_node(), starting at MAX_PAGE_ORDER and falling back to smaller orders so the allocation still succeeds under memory fragmentation. Each chunk is decrypted in place via set_memory_decrypted() on its direct-map address, and the chunks are then stitched into a single virtually-contiguous range with vmap(). Buffers that do not need decryption keep using vzalloc(). To free the buffer, vmbus_free_buffer() calls vunmap() on the range then re-encrypts and frees each chunk individually; any chunk that fails re-encryption is leaked to prevent accidentally freeing decrypted memory. This approach minimizes scattering of decrypted 4 KiB pages through the kernel direct map and the resulting shattering of large page mappings. Signed-off-by: Kameron Carr Reviewed-by: Michael Kelley --- drivers/hv/channel.c | 155 +++++++++++++++++++++++++++++++++++++++++ include/linux/hyperv.h | 7 ++ 2 files changed, 162 insertions(+) diff --git a/drivers/hv/channel.c b/drivers/hv/channel.c index 4782f50..f437061 100644 --- a/drivers/hv/channel.c +++ b/drivers/hv/channel.c @@ -13,11 +13,13 @@ #include #include #include +#include #include #include #include #include #include +#include #include #include #include @@ -608,6 +610,159 @@ int vmbus_establish_gpadl_caller_decrypted(struct vmb= us_channel *channel, } EXPORT_SYMBOL_GPL(vmbus_establish_gpadl_caller_decrypted); =20 +/** + * vmbus_free_buffer - release a buffer allocated by vmbus_alloc_buffer(). + * + * @addr: buffer address, or NULL if none was allocated (e.g. cleanup from= a + * failed allocation) + * @chunks: chunks array from vmbus_alloc_buffer(), or NULL + * @chunk_cnt: number of entries in @chunks + * + * When @chunks is NULL the buffer is a plain vzalloc() allocation. + * + * Otherwise tear down the vmap, and for each chunk re-encrypt and free + * the underlying pages. Any chunk that cannot be re-encrypted is leaked. + */ +void vmbus_free_buffer(void *addr, struct page **chunks, u32 chunk_cnt) +{ + u32 i; + + if (!chunks) { + vfree(addr); + return; + } + + vunmap(addr); + + for (i =3D 0; i < chunk_cnt; i++) { + unsigned long vaddr =3D + (unsigned long)page_address(chunks[i]); + unsigned int order =3D folio_order(page_folio(chunks[i])); + + if (set_memory_encrypted(vaddr, 1U << order)) + continue; + __free_pages(chunks[i], order); + } + + kvfree(chunks); +} +EXPORT_SYMBOL_GPL(vmbus_free_buffer); + +/** + * vmbus_alloc_buffer - allocate a host-visible, virtually-contiguous buff= er. + * + * @channel: the channel the buffer will be attached to + * @size: requested buffer size in bytes (will be rounded up to PAGE_SIZE) + * @chunks_out: on success, set to the array of underlying chunks, or NULL= when + * the buffer was allocated with vzalloc() + * @chunk_cnt_out: on success, set to the number of chunks + * + * Buffers not requiring decryption are allocated with vzalloc(). + * + * Buffers requiring decryption are allocated as a series of + * physically-contiguous chunks, starting at MAX_PAGE_ORDER and falling ba= ck to + * smaller orders on allocation failure. Each chunk is transitioned to + * host-visible via set_memory_decrypted() on its direct-map address, then= all + * chunks are combined into a virtually-contiguous range via vmap(). + * + * Return: the buffer's virtual address, or NULL on failure. + */ +void *vmbus_alloc_buffer(struct vmbus_channel *channel, + u32 size, + struct page ***chunks_out, + u32 *chunk_cnt_out) +{ + unsigned long nr_pages =3D PFN_UP(size); + unsigned long remaining =3D nr_pages; + unsigned long page_idx =3D 0; + struct page **chunks =3D NULL; + struct page **pages =3D NULL; + int order =3D MAX_PAGE_ORDER; + u32 chunk_cnt =3D 0; + void *addr; + u32 i; + int ret; + + *chunks_out =3D NULL; + *chunk_cnt_out =3D 0; + + if (!nr_pages) + return NULL; + + /* If the buffer does not need to be decrypted, just use vzalloc() */ + if (!hv_is_isolation_supported() || channel->co_external_memory) + return vzalloc(nr_pages << PAGE_SHIFT); + + /* Worst case: every chunk is a single page. */ + chunks =3D kvmalloc_array(nr_pages, sizeof(*chunks), + GFP_KERNEL | __GFP_ZERO); + if (!chunks) + goto err; + + pages =3D kvmalloc_array(nr_pages, sizeof(*pages), GFP_KERNEL); + if (!pages) + goto err; + + while (remaining) { + struct page *page; + gfp_t gfp; + + order =3D min(order, ilog2(remaining)); + + /* + * Use __GFP_NORETRY | __GFP_NOWARN to avoid OOM-killing, + * but try harder at order 0 since that is the final + * fallback. + * __GFP_COMP stores order information in the page folio. + */ + gfp =3D GFP_KERNEL | __GFP_ZERO; + if (order) + gfp |=3D __GFP_COMP | __GFP_NORETRY | __GFP_NOWARN; + + page =3D alloc_pages_node(cpu_to_node(channel->target_cpu), + gfp, order); + if (!page) { + if (!order--) + goto err; + continue; + } + + ret =3D set_memory_decrypted((unsigned long)page_address(page), + 1U << order); + if (ret) { + /* + * set_memory_decrypted() failed; the page state is + * unknown so it must be leaked rather than freed. + */ + goto err; + } + + chunks[chunk_cnt++] =3D page; + + for (i =3D 0; i < (1U << order); i++) + pages[page_idx++] =3D page + i; + + remaining -=3D 1U << order; + } + + addr =3D vmap(pages, nr_pages, VM_MAP, pgprot_decrypted(PAGE_KERNEL)); + if (!addr) + goto err; + + memset(addr, 0, nr_pages << PAGE_SHIFT); + + kvfree(pages); + *chunks_out =3D chunks; + *chunk_cnt_out =3D chunk_cnt; + return addr; + +err: + kvfree(pages); + vmbus_free_buffer(NULL, chunks, chunk_cnt); + return NULL; +} +EXPORT_SYMBOL_GPL(vmbus_alloc_buffer); + /** * request_arr_init - Allocates memory for the requestor array. Each slot * keeps track of the next available slot in the array. Initially, each diff --git a/include/linux/hyperv.h b/include/linux/hyperv.h index 1146add..f843ee0 100644 --- a/include/linux/hyperv.h +++ b/include/linux/hyperv.h @@ -1214,6 +1214,13 @@ extern int vmbus_establish_gpadl_caller_decrypted(st= ruct vmbus_channel *channel, extern int vmbus_teardown_gpadl(struct vmbus_channel *channel, struct vmbus_gpadl *gpadl); =20 +extern void *vmbus_alloc_buffer(struct vmbus_channel *channel, + u32 size, + struct page ***chunks_out, + u32 *chunk_cnt_out); + +extern void vmbus_free_buffer(void *addr, struct page **chunks, u32 chunk_= cnt); + void vmbus_reset_channel_cb(struct vmbus_channel *channel); =20 extern int vmbus_recvpacket(struct vmbus_channel *channel, --=20 2.45.4 From nobody Tue Sep 29 06:08:30 2026 Received: from linux.microsoft.com (linux.microsoft.com [13.77.154.182]) by smtp.subspace.kernel.org (Postfix) with ESMTP id EC82844CAE4; Tue, 11 Aug 2026 16:05:26 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=13.77.154.182 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786464328; cv=none; b=pOjiTZiz0+R0/ky7CAsnOZPY80TrowobMi53LAOcXCYJ6Z1xOQ7mw3ZaiWxKRIdnoMBv+eFDOd6mOgmuPtQUSSpAULbclWXZRaeUO9AQY8nOnhaXJFguOqnmigj1ErYbK7vbggNBYjSBas8SV3Du0vCcqyMpWdB/YiZVZ0VkEJY= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786464328; c=relaxed/simple; bh=OXf88dKTPUAzNxj2bKBiJ65mnv3ZpoqbsLuIhJvcg9g=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=GCk96hn1O2E8+odIrOL5h+jPnP0mmst4/eACc69iyStvOaVvUKGqqxdkbXpLY8CptFKMwatBBK0Dl8rhieTXLyqulxa+jeCHsZABFfurdvh0AWdfdkpjToIivKXzEV32LmxFvnEcxZ9gwC0QNsXZh4Pt6Zhzkz4J4+wzC4n3NHk= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com; spf=pass smtp.mailfrom=linux.microsoft.com; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b=kfkyBE7j; arc=none smtp.client-ip=13.77.154.182 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=linux.microsoft.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (1024-bit key) header.d=linux.microsoft.com header.i=@linux.microsoft.com header.b="kfkyBE7j" Received: from linuxonhyperv3.guj3yctzbm1etfxqx2vob5hsef.xx.internal.cloudapp.net (linux.microsoft.com [13.77.154.182]) by linux.microsoft.com (Postfix) with ESMTPSA id 7328E20B7128; Tue, 11 Aug 2026 09:05:02 -0700 (PDT) DKIM-Filter: OpenDKIM Filter v2.11.0 linux.microsoft.com 7328E20B7128 DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/relaxed; d=linux.microsoft.com; s=default; t=1786464302; bh=uoV5YR+wFUSNeuF2Bi8eWDiJ2EtLDk1eoIezLs6d6QI=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=kfkyBE7jdKKL4CazVeRRZ2w+AlEFV9AfHi9F3KEKNhewE55zEolt1FDPGk4Ne0qB+ GrWATMHiRu1fKFUae9PCiTPWx7jNNrSNaawgg1E/NcEv/DaKNPfKkI9dMyvnUseaYf M3yBa+ssu4n1vykzhcSq9CxD9iRbbecWqc4wWO1c= From: Kameron Carr To: decui@microsoft.com, haiyangz@microsoft.com, kys@microsoft.com, longli@microsoft.com, wei.liu@kernel.org, mhklinux@outlook.com Cc: andrew+netdev@lunn.ch, davem@davemloft.net, edumazet@google.com, kuba@kernel.org, pabeni@redhat.com, linux-hyperv@vger.kernel.org, linux-kernel@vger.kernel.org, netdev@vger.kernel.org Subject: [PATCH v4 3/3] hv_netvsc: Allocate send/receive buffers using vmbus_alloc_buffer() Date: Tue, 11 Aug 2026 09:04:47 -0700 Message-ID: <20260811160447.2529876-4-kameroncarr@linux.microsoft.com> X-Mailer: git-send-email 2.43.7 In-Reply-To: <20260811160447.2529876-1-kameroncarr@linux.microsoft.com> References: <20260811160447.2529876-1-kameroncarr@linux.microsoft.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" On CoCo VMs without confidential VMBus, the netvsc send and receive buffers must be made host-visible by decrypting them. These buffers are vmalloc'ed, but set_memory_decrypted()/encrypted() do not work on vmalloc'ed memory. This use case is (so far) unique to netvsc, so solve it locally rather than changing the set_memory() or allocation APIs. Use vmbus_alloc_buffer() to allocate the send and receive buffers, which will make them host-visible. Store the list of memory chunks in the netvsc_device struct so they can be individually freed later. Use vmbus_establish_gpadl_caller_decrypted() so there is no attempt to decrypt the virtual address. Appropriately free the buffers with vmbus_free_buffer(). Because vunmap() and set_memory_encrypted() must run in process context, replace the rcu_head/call_rcu() pair used to defer free_netvsc_device() with rcu_work/queue_rcu_work(). This also fixes a small race condition where the buffers may be accessed while being re-encrypted by moving the re-encryption after the RCU grace period. Signed-off-by: Kameron Carr Reviewed-by: Michael Kelley --- drivers/net/hyperv/hyperv_net.h | 8 ++- drivers/net/hyperv/netvsc.c | 103 ++++++++++++++++++++++---------- drivers/net/hyperv/netvsc_drv.c | 6 ++ 3 files changed, 83 insertions(+), 34 deletions(-) diff --git a/drivers/net/hyperv/hyperv_net.h b/drivers/net/hyperv/hyperv_ne= t.h index 7397c69..4841367 100644 --- a/drivers/net/hyperv/hyperv_net.h +++ b/drivers/net/hyperv/hyperv_net.h @@ -220,6 +220,8 @@ struct net_device_context; =20 extern u32 netvsc_ring_bytes; =20 +int netvsc_workqueue_init(void); +void netvsc_workqueue_destroy(void); struct netvsc_device *netvsc_device_add(struct hv_device *device, const struct netvsc_device_info *info); int netvsc_alloc_recv_comp_ring(struct netvsc_device *net_device, u32 q_id= x); @@ -1158,6 +1160,8 @@ struct netvsc_device { /* Receive buffer allocated by us but manages by NetVSP */ void *recv_buf; u32 recv_buf_size; /* allocated bytes */ + struct page **recv_buf_chunks; + u32 recv_buf_chunk_cnt; struct vmbus_gpadl recv_buf_gpadl_handle; u32 recv_section_cnt; u32 recv_section_size; @@ -1166,6 +1170,8 @@ struct netvsc_device { /* Send buffer allocated by us */ void *send_buf; u32 send_buf_size; + struct page **send_buf_chunks; + u32 send_buf_chunk_cnt; struct vmbus_gpadl send_buf_gpadl_handle; u32 send_section_cnt; u32 send_section_size; @@ -1193,7 +1199,7 @@ struct netvsc_device { =20 struct netvsc_channel chan_table[VRSS_CHANNEL_MAX]; =20 - struct rcu_head rcu; + struct rcu_work rwork; }; =20 /* NdisInitialize message */ diff --git a/drivers/net/hyperv/netvsc.c b/drivers/net/hyperv/netvsc.c index 59e9534..c59f2a4 100644 --- a/drivers/net/hyperv/netvsc.c +++ b/drivers/net/hyperv/netvsc.c @@ -28,6 +28,8 @@ #include "hyperv_net.h" #include "netvsc_trace.h" =20 +static struct workqueue_struct *netvsc_wq; + /* * Switch the data path from the synthetic interface to the VF * interface. @@ -125,6 +127,47 @@ static void netvsc_subchan_work(struct work_struct *w) rtnl_unlock(); } =20 +static void __free_netvsc_device(struct netvsc_device *nvdev) +{ + int i; + + kfree(nvdev->extension); + + vmbus_free_buffer(nvdev->recv_buf, nvdev->recv_buf_chunks, + nvdev->recv_buf_chunk_cnt); + vmbus_free_buffer(nvdev->send_buf, nvdev->send_buf_chunks, + nvdev->send_buf_chunk_cnt); + bitmap_free(nvdev->send_section_map); + + for (i =3D 0; i < VRSS_CHANNEL_MAX; i++) { + xdp_rxq_info_unreg(&nvdev->chan_table[i].xdp_rxq); + kfree(nvdev->chan_table[i].recv_buf); + vfree(nvdev->chan_table[i].mrc.slots); + } + + kfree(nvdev); +} + +static void free_netvsc_device(struct work_struct *w) +{ + struct rcu_work *rwork =3D to_rcu_work(w); + + __free_netvsc_device(container_of(rwork, struct netvsc_device, rwork)); +} + +int netvsc_workqueue_init(void) +{ + netvsc_wq =3D alloc_workqueue("hv_netvsc", WQ_UNBOUND, 0); + + return netvsc_wq ? 0 : -ENOMEM; +} + +void netvsc_workqueue_destroy(void) +{ + rcu_barrier(); + destroy_workqueue(netvsc_wq); +} + static struct netvsc_device *alloc_net_device(void) { struct netvsc_device *net_device; @@ -143,36 +186,18 @@ static struct netvsc_device *alloc_net_device(void) init_completion(&net_device->channel_init_wait); init_waitqueue_head(&net_device->subchan_open); INIT_WORK(&net_device->subchan_work, netvsc_subchan_work); + INIT_RCU_WORK(&net_device->rwork, free_netvsc_device); =20 return net_device; } =20 -static void free_netvsc_device(struct rcu_head *head) -{ - struct netvsc_device *nvdev - =3D container_of(head, struct netvsc_device, rcu); - int i; - - kfree(nvdev->extension); - - if (!nvdev->recv_buf_gpadl_handle.decrypted) - vfree(nvdev->recv_buf); - if (!nvdev->send_buf_gpadl_handle.decrypted) - vfree(nvdev->send_buf); - bitmap_free(nvdev->send_section_map); - - for (i =3D 0; i < VRSS_CHANNEL_MAX; i++) { - xdp_rxq_info_unreg(&nvdev->chan_table[i].xdp_rxq); - kfree(nvdev->chan_table[i].recv_buf); - vfree(nvdev->chan_table[i].mrc.slots); - } - - kfree(nvdev); -} - static void free_netvsc_device_rcu(struct netvsc_device *nvdev) { - call_rcu(&nvdev->rcu, free_netvsc_device); + /* + * Defer the actual free to process context: vunmap() and + * set_memory_encrypted() cannot run from RCU softirq context. + */ + queue_rcu_work(netvsc_wq, &nvdev->rwork); } =20 static void netvsc_revoke_recv_buf(struct hv_device *device, @@ -351,7 +376,10 @@ static int netvsc_init_buf(struct hv_device *device, buf_size =3D min_t(unsigned int, buf_size, NETVSC_RECEIVE_BUFFER_SIZE_LEGACY); =20 - net_device->recv_buf =3D vzalloc(buf_size); + net_device->recv_buf =3D + vmbus_alloc_buffer(device->channel, buf_size, + &net_device->recv_buf_chunks, + &net_device->recv_buf_chunk_cnt); if (!net_device->recv_buf) { netdev_err(ndev, "unable to allocate receive buffer of size %u\n", @@ -367,9 +395,10 @@ static int netvsc_init_buf(struct hv_device *device, * channel. Note: This call uses the vmbus connection rather * than the channel to establish the gpadl handle. */ - ret =3D vmbus_establish_gpadl(device->channel, net_device->recv_buf, - buf_size, - &net_device->recv_buf_gpadl_handle); + ret =3D vmbus_establish_gpadl_caller_decrypted(device->channel, + net_device->recv_buf, + buf_size, + &net_device->recv_buf_gpadl_handle); if (ret !=3D 0) { netdev_err(ndev, "unable to establish receive buffer's gpadl\n"); @@ -457,7 +486,10 @@ static int netvsc_init_buf(struct hv_device *device, buf_size =3D device_info->send_sections * device_info->send_section_size; buf_size =3D round_up(buf_size, PAGE_SIZE); =20 - net_device->send_buf =3D vzalloc(buf_size); + net_device->send_buf =3D + vmbus_alloc_buffer(device->channel, buf_size, + &net_device->send_buf_chunks, + &net_device->send_buf_chunk_cnt); if (!net_device->send_buf) { netdev_err(ndev, "unable to allocate send buffer of size %u\n", buf_size); @@ -470,9 +502,10 @@ static int netvsc_init_buf(struct hv_device *device, * channel. Note: This call uses the vmbus connection rather * than the channel to establish the gpadl handle. */ - ret =3D vmbus_establish_gpadl(device->channel, net_device->send_buf, - buf_size, - &net_device->send_buf_gpadl_handle); + ret =3D vmbus_establish_gpadl_caller_decrypted(device->channel, + net_device->send_buf, + buf_size, + &net_device->send_buf_gpadl_handle); if (ret !=3D 0) { netdev_err(ndev, "unable to establish send buffer's gpadl\n"); @@ -1863,7 +1896,11 @@ struct netvsc_device *netvsc_device_add(struct hv_de= vice *device, netif_napi_del(&net_device->chan_table[0].napi); =20 cleanup2: - free_netvsc_device(&net_device->rcu); + /* + * net_device was never published, so we don't need to wait for an + * RCU grace period -- call the free routine synchronously. + */ + __free_netvsc_device(net_device); =20 return ERR_PTR(ret); } diff --git a/drivers/net/hyperv/netvsc_drv.c b/drivers/net/hyperv/netvsc_dr= v.c index ee5ab5c..1d43c73 100644 --- a/drivers/net/hyperv/netvsc_drv.c +++ b/drivers/net/hyperv/netvsc_drv.c @@ -2867,12 +2867,17 @@ static void __exit netvsc_drv_exit(void) { unregister_netdevice_notifier(&netvsc_netdev_notifier); vmbus_driver_unregister(&netvsc_drv); + netvsc_workqueue_destroy(); } =20 static int __init netvsc_drv_init(void) { int ret; =20 + ret =3D netvsc_workqueue_init(); + if (ret) + return ret; + if (ring_size < RING_SIZE_MIN) { ring_size =3D RING_SIZE_MIN; pr_info("Increased ring_size to %u (min allowed)\n", @@ -2890,6 +2895,7 @@ static int __init netvsc_drv_init(void) =20 err_vmbus_reg: unregister_netdevice_notifier(&netvsc_netdev_notifier); + netvsc_workqueue_destroy(); return ret; } =20 --=20 2.45.4