From nobody Tue Sep 29 08:26:17 2026 Received: from va-1-114.ptr.blmpb.com (va-1-114.ptr.blmpb.com [209.127.230.114]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 17C8B36F8F0 for ; Mon, 10 Aug 2026 12:22:08 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.230.114 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364531; cv=none; b=aWHAs2/tZYZ5eSCOURn8c+QrtkoRLeGZVpXFni56hogNHj3731p/ST1czCb0WdOiBgAsgP5NLg5T7n/h0KHtFpdJii+xh6kTBsiYIHLDDvfIqwMJKif/dNLS8gqmkFk+b8Zcr3NdgXNalMH8wE8P7xT96kZBkmR1i5wZNIR2nsI= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364531; c=relaxed/simple; bh=CSS9gwks0lv/fJG4J+/DCdPNhjgQ2fB/8r3a7lOVaFM=; h=References:Content-Type:To:Date:In-Reply-To:Subject:Message-Id:Cc: Mime-Version:From; b=BsVdgn0o1xs6zRGfz57+Nw4k3nb505vPbMFF/kVmiHDX6f0AMQmwh0+4QLHjj3zpHygGRwM6IYwPKiaE4juH+pcnPo0f/BcsRehsAteBBKa2nsjqwsDdcE3cjitytlJyzrRlASAUuXlvxfAZvXFZH/eOTRSItCwtk+vZNTpWZ/8= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=frEo5qcW; arc=none smtp.client-ip=209.127.230.114 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="frEo5qcW" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364523; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=6LDfmUcelxmT9nJB/bHRwgrk2ZzY+G3Dy0t3Kxw/TPQ=; b=frEo5qcWDYAkeE5zop7vhFKmQ/ytTyiQE/r+DxfGySnuncBnp94UDWzDhCG2wcJhUA5aJ6 8NQVn1Ac+ZXgIz13HiQ8fu94x6y78OvfIXZ2MNtaTq+uSA7sP99C7syMikaeGhlRsOLqjk dSQJBW30ANxMW6Mjb8U8ehsbGOu3bTnaYbnQvziVLTAaWb5iJtnyCkUaUYijjfqKHZSQYV E/7G1JM41O6TcxuuWqEwsKPbkJHhdbtYtfjMmwrHRNR+gw5AEZPD4vc2OxVxi077fUREek yp3rna9iHYXoHXwZYRAYPjsb1ENgpXFGngYjbE2hwsOU0bef5/M6Nj38rh44Lg== X-Mailer: git-send-email 2.45.2 References: <20260810122057.30447-1-lizhe.67@bytedance.com> To: , , , , , , , , , , , Date: Mon, 10 Aug 2026 20:20:50 +0800 In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> Subject: [PATCH v10 1/8] mm: fix stale ZONE_DEVICE refcount comment Message-Id: <20260810122057.30447-2-lizhe.67@bytedance.com> X-Lms-Return-Path: Cc: , , , , , Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 Content-Transfer-Encoding: quoted-printable From: "Li Zhe" X-Original-From: Li Zhe Content-Type: text/plain; charset="utf-8" The comment in __init_zone_device_page() still uses the old MEMORY_TYPE_* names and implies that FS_DAX pages regain a refcount of 1 in the free path. That no longer matches the code. Update the comment to describe the current policy correctly: MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free path, while the remaining ZONE_DEVICE types start from 0 here and raise the count again when the allocator or driver hands the page out. No functional change intended. Signed-off-by: Li Zhe Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Alistair Popple Reviewed-by: Muchun Song Reviewed-by: Mike Rapoport (Microsoft) --- mm/mm_init.c | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 0f64909e8d20..95808ab5cfdb 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1030,13 +1030,9 @@ static void __ref __init_zone_device_page(struct pag= e *page, unsigned long pfn, page->zone_device_data =3D NULL; =20 /* - * ZONE_DEVICE pages other than MEMORY_TYPE_GENERIC are released - * directly to the driver page allocator which will set the page count - * to 1 when allocating the page. - * - * MEMORY_TYPE_GENERIC and MEMORY_TYPE_FS_DAX pages automatically have - * their refcount reset to one whenever they are freed (ie. after - * their refcount drops to 0). + * MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free + * path. The remaining ZONE_DEVICE types start from 0 here and raise + * the count again when the allocator or driver hands the page out. */ switch (pgmap->type) { case MEMORY_DEVICE_FS_DAX: --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-1-113.ptr.blmpb.com (va-1-113.ptr.blmpb.com [209.127.230.113]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 9988A3B7759 for ; Mon, 10 Aug 2026 12:22:35 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.230.113 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364557; cv=none; b=tG40Sx3gxdxYHS0H0I4BH3bvxKxJ83+v9aL/Bpq4sWl61euZlR1/eCVwEAhRjUw///eqbiw4Ty8lyam35s+PYde5jvfQyXW0t+IXMhGhbhLRI0RNFqqaleq6BfzBPqSa+e3LsMxOP+7uujwOyZteg8wn/dq/+j5GMCM8a4KnG/Y= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364557; c=relaxed/simple; bh=DevAn1r9WS6pPdFWehOAQ0KmRVHSptKuRZyUe2bUpqI=; h=From:Date:Message-Id:In-Reply-To:Content-Type:Mime-Version:Cc: Subject:References:To; b=ctp9/sjnIPfj7TxjJcSWf4nXc1zImycHZy/sd+Y5kMXZ7qd7NMXSz8HnNLk78Cjdzw5qDUGpqhETS5LXYqzNe+arLlyaP9jw+Aoo0I+odS5QpNf4ZM4WoXMEsiaLtggaG2XakLhb7ZrUlkz0y+6C7Iz4PYYbvXWJTUrSw9qQVXA= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=EQNj1wu1; arc=none smtp.client-ip=209.127.230.113 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="EQNj1wu1" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364550; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=z2FZl2/YzaMjC0aIAZHsTKk9OBnNm9ulSHwBzV0fwO8=; b=EQNj1wu1MdksK5Xx+IUlx3uiyjJAVBhE30PDdio3MhJrGmHu6UgPca+v3miVaK/IqngmAC bugRdNGiyUfFHAjjzTjGh5Qg4pyssRBSKfi/74WxMpgU5uiKUHk9f0NJP1jjOG7ukBC/nk WR1dSiddlc94zmyBUI5hXLs3EHFyetz93baO2iIB4PeZKtjKV1dDxE45ZrXMw8y+LqNsyK v3gMXjCcYKMPpZiswGnzDoMIBz3rmIKudxZfiQaAY0V+oKK6f0gUdBfk44Udg3+MrO7m7X RsqFUgKSP7KFUNCER2fATeeBWWnEwin5VAE+u3vDm8eIE4xR2EhBDm33paHb5A== From: "Li Zhe" Date: Mon, 10 Aug 2026 20:20:51 +0800 Message-Id: <20260810122057.30447-3-lizhe.67@bytedance.com> Content-Transfer-Encoding: quoted-printable X-Original-From: Li Zhe In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 Cc: , , , , , Subject: [PATCH v10 2/8] mm: factor zone-device page init helpers out of __init_zone_device_page X-Lms-Return-Path: X-Mailer: git-send-email 2.45.2 References: <20260810122057.30447-1-lizhe.67@bytedance.com> To: , , , , , , , , , , , Content-Type: text/plain; charset="utf-8" memmap_init_zone_device() currently mixes refcount policy and core ZONE_DEVICE page setup in a single helper. Factor the refcount-reset predicate into pagemap_requires_refcount_reset(), move the common page initialization into __zone_device_page_init(), and wrap the existing slow path in zone_device_page_init_slow(). This keeps the slow-path behaviour unchanged and gives later patches reusable helper boundaries. No functional change intended. Signed-off-by: Li Zhe Reviewed-by: Mike Rapoport (Microsoft) --- mm/mm_init.c | 56 ++++++++++++++++++++++++++++++++++------------------ 1 file changed, 37 insertions(+), 19 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 95808ab5cfdb..a70acb7431a6 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1005,11 +1005,37 @@ static void __init memmap_init(void) } =20 #ifdef CONFIG_ZONE_DEVICE -static void __ref __init_zone_device_page(struct page *page, unsigned long= pfn, +/* + * Return true when memmap_init_zone_device() must initialize the page + * refcount to 0. MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in + * the free path, while the remaining ZONE_DEVICE types start from 0 here + * and raise the count again when the allocator or driver hands the page + * out. + */ +static inline bool pagemap_requires_refcount_reset(const struct dev_pagema= p *pgmap) +{ + /* + * MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free + * path. The remaining ZONE_DEVICE types start from 0 here and raise + * the count again when the allocator or driver hands the page out. + */ + switch (pgmap->type) { + case MEMORY_DEVICE_FS_DAX: + case MEMORY_DEVICE_PRIVATE: + case MEMORY_DEVICE_COHERENT: + case MEMORY_DEVICE_PCI_P2PDMA: + return true; + case MEMORY_DEVICE_GENERIC: + return false; + } + + return false; +} + +static void __ref __zone_device_page_init(struct page *page, unsigned long= pfn, unsigned long zone_idx, int nid, struct dev_pagemap *pgmap) { - __init_single_page(page, pfn, zone_idx, nid); =20 /* @@ -1028,23 +1054,15 @@ static void __ref __init_zone_device_page(struct pa= ge *page, unsigned long pfn, */ page_folio(page)->pgmap =3D pgmap; page->zone_device_data =3D NULL; +} =20 - /* - * MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free - * path. The remaining ZONE_DEVICE types start from 0 here and raise - * the count again when the allocator or driver hands the page out. - */ - switch (pgmap->type) { - case MEMORY_DEVICE_FS_DAX: - case MEMORY_DEVICE_PRIVATE: - case MEMORY_DEVICE_COHERENT: - case MEMORY_DEVICE_PCI_P2PDMA: +static void __ref zone_device_page_init_slow(struct page *page, + unsigned long pfn, unsigned long zone_idx, int nid, + struct dev_pagemap *pgmap) +{ + __zone_device_page_init(page, pfn, zone_idx, nid, pgmap); + if (pagemap_requires_refcount_reset(pgmap)) set_page_count(page, 0); - break; - - case MEMORY_DEVICE_GENERIC: - break; - } } =20 /* @@ -1090,7 +1108,7 @@ static void __ref memmap_init_compound(struct page *h= ead, for (pfn =3D head_pfn + 1; pfn < end_pfn; pfn++) { struct page *page =3D pfn_to_page(pfn); =20 - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + zone_device_page_init_slow(page, pfn, zone_idx, nid, pgmap); prep_compound_tail(page, head, order); set_page_count(page, 0); } @@ -1126,7 +1144,7 @@ void __ref memmap_init_zone_device(struct zone *zone, for (pfn =3D start_pfn; pfn < end_pfn; pfn +=3D pfns_per_compound) { struct page *page =3D pfn_to_page(pfn); =20 - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + zone_device_page_init_slow(page, pfn, zone_idx, nid, pgmap); =20 if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-1-115.ptr.blmpb.com (va-1-115.ptr.blmpb.com [209.127.230.115]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 41B673783C4 for ; Mon, 10 Aug 2026 12:23:01 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.230.115 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364582; cv=none; b=VjmO08Cv1KOrzL7Y1m+kn5BG3jiYelI9YpZmn6zFDfJ8Ci4V6hvD4tUsyhk92zBK1TseUxTPP7kZ4k/sE3RQM3GIiHyMhLgEl3bLlj9sHEBjVOVxGzFt72FFD3RqAo4rCjRwhmM75BFe5u2cEhasXnRLt95FXboeaWnH/mxVpSI= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364582; c=relaxed/simple; bh=BIVDw/5zLYeH4lTLhdwm8OUozWu2BrvT0BiGbt//W2c=; h=Date:Mime-Version:Content-Type:From:Subject:Message-Id: In-Reply-To:Cc:References:To; b=Pk5bx8FDtWb1BCah1nKFEoFW2KXquJFyV3MHb4F7a0ZPSXkI2Oq3NAcTMIGQbJg63htB8YSKQw1if2aA3w7OGmjV68ox7PCY5c3ixrxzpfXN9zVawP6YPLq07ZZa7dEzLI4ZiEPWtFT7e+tNiRZY0i3cpvrg74i1BN2bSMtj8wo= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=KhvC5IC9; arc=none smtp.client-ip=209.127.230.115 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="KhvC5IC9" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364575; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=HX3xWYPWMGyEZ3/iAdp9cHSDCjeSCNmVjV7Tq4qpRvQ=; b=KhvC5IC9t3DsefOPecnt0xCQqz0rUpvTC1Et1etmBREtF0c0F5WYlKcsqv8goPEZRZFo+1 3ritwWNvhnAnXErPOYNI7vIu50Kdz64QdF9gEJVz8+PTB71xkLNqAwLROiuMU11vvGHteb 1hv1z3D5pitNHV4ZZjOSpLZhYCiuuFDqTWADwf93Z5EmVjH4OGGu/l7ZUA/Pw326D5vfqq ez9sGHl9HjUeV6mghyIMCS+WpKFV/EfL9ImBqhk1jFz7OWJ+N+cv29ce9u4FyW+JG4CtJZ aphPbf7dkchqGPbCa/aBEZpRWcLE6W10FVefzOPNaIYDDESo+zmnfWmR4wro8g== Date: Mon, 10 Aug 2026 20:20:52 +0800 Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 X-Lms-Return-Path: X-Mailer: git-send-email 2.45.2 From: "Li Zhe" Subject: [PATCH v10 3/8] mm: add a set_page_section_from_pfn() helper Message-Id: <20260810122057.30447-4-lizhe.67@bytedance.com> In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> Content-Transfer-Encoding: quoted-printable X-Original-From: Li Zhe Cc: , , , , , References: <20260810122057.30447-1-lizhe.67@bytedance.com> To: , , , , , , , , , , , Content-Type: text/plain; charset="utf-8" Callers that want to update section bits from a PFN currently need to open-code: set_page_section(page, pfn_to_section_nr(pfn)); and guard that sequence with #ifdef SECTION_IN_PAGE_FLAGS. Add set_page_section_from_pfn() to wrap that update in one place. When section bits are stored in page flags, the helper derives the section number from the PFN and updates the page flags. Otherwise keep it as a no-op so callers can use one helper without open-coding SECTION_IN_PAGE_FLAGS. Convert set_page_links() to use the new helper so later ZONE_DEVICE fast-path patches can also update section bits without open-coding SECTION_IN_PAGE_FLAGS at each callsite. This keeps the PFN-to-section translation local to the configurations that actually store section bits in struct page flags, and avoids exposing that detail to generic callers. No functional change intended. Signed-off-by: Li Zhe Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Muchun Song Reviewed-by: Balbir Singh --- include/linux/mm.h | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 485df9c2dbdd..43343bfce493 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2541,11 +2541,22 @@ static inline void set_page_section(struct page *pa= ge, unsigned long section) page->flags.f |=3D (section & SECTIONS_MASK) << SECTIONS_PGSHIFT; } =20 +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ + set_page_section(page, pfn_to_section_nr(pfn)); +} + static inline unsigned long memdesc_section(memdesc_flags_t mdf) { return (mdf.f >> SECTIONS_PGSHIFT) & SECTIONS_MASK; } #else /* !SECTION_IN_PAGE_FLAGS */ +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ +} + static inline unsigned long memdesc_section(memdesc_flags_t mdf) { return 0; @@ -2768,9 +2779,7 @@ static inline void set_page_links(struct page *page, = enum zone_type zone, { set_page_zone(page, zone); set_page_node(page, node); -#ifdef SECTION_IN_PAGE_FLAGS - set_page_section(page, pfn_to_section_nr(pfn)); -#endif + set_page_section_from_pfn(page, pfn); } =20 /** --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-1-111.ptr.blmpb.com (va-1-111.ptr.blmpb.com [209.127.230.111]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 38C963B583B for ; Mon, 10 Aug 2026 12:23:34 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.230.111 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364616; cv=none; b=oRfOex/rkWr5RJvHuQnW1CX/etsSLIZDkWZUUFjCsQEqZb2whnch2vu67puqhJUYe9ZLJ3ZSy6AS0OexQoFsVpNS6qHRCg11HarHQrkd/tlCS3zIPNbsMJgqHVr1ixqjrQziy6MAlK3cA+Ilvao3cB5cWXCt2SNnPp72JM4urFw= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364616; c=relaxed/simple; bh=d8Qz30+jm+1jK9zOJX3uxZNIPgKolYW1sbUNz1N9uIc=; h=Mime-Version:References:From:Cc:Date:Message-Id:Content-Type:To: Subject:In-Reply-To; b=WrW+92to5YZ+JQdqlGlXeTDffTuIzvOf2SRgdxjvboi4iH9BdNRzPDNVlK5D04oaqxGdj3dhQU8vvle226xUYreZB5ZU2IXGYXl+Ma5W+nMLuL3Aa1K59pUVkVCMT95/ltbgGnZYh5RlecCYGjn3PEz3ZmVhW8Aud4eeJHwkMRo= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=NEW2BiAv; arc=none smtp.client-ip=209.127.230.111 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="NEW2BiAv" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364603; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=s126gl7tRFhaJiWX2giecuLSdCnS+/Lfp4QcjvwoLqg=; b=NEW2BiAvcaf9WiFuc+Sdy1VBVa6WLD7vxKEl1iam7o0dQf3m4icujr7UkXEEqhDYL7dqUj AZjElPV8OfaphJUDMJLo1eeC2Kh+TW/DBavR6fJEzQM4u1h/yTKXSlutlbDDQ4zUV8YxSs oQJMYyySjDlNaWRAQKnajgr/uBcwsAeq732OOIYHE1ad7Stdvtv3xm/8SFhMgW8Uyfcdwl NKBLOd8OJlLW5W5d1+6Fo/5vJ0OtdGKIeDfI6TOfOUk2O6eb8DNGZq7kRHpLqvmuhrZ+mJ pkuYaOeq81JhcvA2Ue3Zm4zIwUVk3LhbzSpqfOAZ26XNZs0DrrLRVW83f4Uyqg== Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 References: <20260810122057.30447-1-lizhe.67@bytedance.com> Content-Transfer-Encoding: quoted-printable X-Lms-Return-Path: From: "Li Zhe" X-Original-From: Li Zhe Cc: , , , , , Date: Mon, 10 Aug 2026 20:20:53 +0800 Message-Id: <20260810122057.30447-5-lizhe.67@bytedance.com> To: , , , , , , , , , , , Subject: [PATCH v10 4/8] mm: add a template-based fast path for zone-device page init In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> X-Mailer: git-send-email 2.45.2 Content-Type: text/plain; charset="utf-8" memmap_init_zone_device() repeats nearly identical head-page initialization for each PFN. Prepare one reusable ZONE_DEVICE head-page template through the existing slow path, refresh the PFN-dependent fields in that template before each copy, and memcpy it into each destination page. Use the template path unconditionally. The page_ref_set tracepoint is primarily a debugging aid, while this code is still initializing struct pages before they are handed out. From the perspective of users of those pages, the initialization-time refcount transitions are not part of the observable page lifetime. This means page_ref_set will no longer observe every initialization-time refcount assignment for copied ZONE_DEVICE head pages. The impact is controlled because the final initialized struct page state is unchanged, and keeping a separate non-template path only for this local tracepoint observability would add complexity to the common path. This patch accelerates head-page initialization. The pfns_per_compound =3D=3D 1 case gets the full benefit here, compound tails are handled in the next patch. Tested in a VM with a 100 GB fsdax namespace device configured with map=3Ddev on Intel Ice Lake server. This test exercises the nd_pmem rebind path (pfns_per_compound =3D=3D 1). Test procedure: Rebind the nd_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.2-rc1): Average of rebinds for nd_pmem driver: 244.28 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 215.55 ms This reduces the average memmap initialization time measured during rebind from 244.28 ms to 215.55 ms, or about 11%. Signed-off-by: Li Zhe --- mm/mm_init.c | 47 ++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 46 insertions(+), 1 deletion(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index a70acb7431a6..56a36a71ba89 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1065,6 +1065,35 @@ static void __ref zone_device_page_init_slow(struct = page *page, set_page_count(page, 0); } =20 +/* + * 'template' is a reusable page prototype rather than a strictly immutable + * object. Most ZONE_DEVICE fields stay constant across the pages covered = by + * the current template, but section bits and page->virtual may still depe= nd + * on the PFN. Refresh those PFN-dependent fields in the template before + * copying it into @page. + */ +static inline void zone_device_page_update_template(struct page *template, + unsigned long pfn) +{ + set_page_section_from_pfn(template, pfn); +#ifdef WANT_PAGE_VIRTUAL + if (!is_highmem_idx(ZONE_DEVICE)) + set_page_address(template, __va(pfn << PAGE_SHIFT)); +#endif +} + +static void zone_device_page_init_from_template(struct page *page, + unsigned long pfn, struct page *template) +{ + /* + * 'template' carries the invariant portion of a ZONE_DEVICE struct + * page. Update the PFN-dependent fields in place before copying it + * to the destination page. + */ + zone_device_page_update_template(template, pfn); + memcpy(page, template, sizeof(*page)); +} + /* * With compound page geometry and when struct pages are stored in ram most * tail pages are reused. Consequently, the amount of unique struct pages = to @@ -1127,6 +1156,7 @@ void __ref memmap_init_zone_device(struct zone *zone, unsigned long zone_idx =3D zone_idx(zone); unsigned long start =3D jiffies; int nid =3D pgdat->node_id; + struct page template; =20 if (WARN_ON_ONCE(!pgmap || zone_idx !=3D ZONE_DEVICE)) return; @@ -1144,7 +1174,22 @@ void __ref memmap_init_zone_device(struct zone *zone, for (pfn =3D start_pfn; pfn < end_pfn; pfn +=3D pfns_per_compound) { struct page *page =3D pfn_to_page(pfn); =20 - zone_device_page_init_slow(page, pfn, zone_idx, nid, pgmap); + if (pfn =3D=3D start_pfn) { + /* + * Seed the reusable head-page template from the + * first real struct page. This initializes the + * first page through the existing slow path and + * then reuses that final state as the template + * for subsequent pages. + */ + zone_device_page_init_slow(page, pfn, zone_idx, + nid, pgmap); + /* init template page */ + memcpy(&template, page, sizeof(*page)); + } else { + zone_device_page_init_from_template(page, pfn, + &template); + } =20 if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-1-111.ptr.blmpb.com (va-1-111.ptr.blmpb.com [209.127.230.111]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id CA69F3C1D54 for ; Mon, 10 Aug 2026 12:23:50 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.230.111 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364632; cv=none; b=sk4bQqitCGdy1dZT6Q5v1mhcoZJqU6WhktfPfVbDYiqh3RIcDa2ezlIdBQm3XV9q0La+Tmp6nPuzXGizYx2A40W34ycPdSWDQqz8neBcu6r1pKMWJ2bi8vB+baDHN51kJl0dvb3FAeRKHcqyfBW4pfaI/89SGmUWVCs2bvSJ3IA= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364632; c=relaxed/simple; bh=HvOqY7OOXPJystOTPihT2elBgkM3rK57/rQRCgQ7j4E=; h=To:Cc:Date:References:Message-Id:Mime-Version:Subject:From: In-Reply-To:Content-Type; b=I//bJXhvHxdUUSKkhNRaHpGvFMYetNqI1hFnDnps0fJ74O30fV8JfF/mClVDZVARet14Lnm3nPrVXqlnKICSobNd/upQ9BIua/UDzm0DubaLvnvUn57v2BnJ0701H7okjCUTtzLgRA21NZqbRHkRihQJRmaLTLQtyTLeZd+Myg8= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=CrZCPDbt; arc=none smtp.client-ip=209.127.230.111 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="CrZCPDbt" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364625; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=cizGxiXgfyBnRadLUW5Uvz95HeeiSdYZ4khU0tGvPq8=; b=CrZCPDbtXQ5wPBqieg3Di3K2IGP5atdi0QVgOKh1pcFv8Siv34p1dXBXwrxJz0sveG8M47 tUkHcau8CeC5D5P1VTZ4kzW54Ars6kY8MHXDNQ6xeCdyWxOg3P7XRKjhO1davnhqSl5upf 2wknkYrzsy+QYdDXXAtbkq5H295fAwO63PYOpu1S5VO8IaUIGJDL9f/ZnMKPJO38r8UPq6 b1IyVtMWtI5QyDV7IcYuLNksFoGFbuwFtUCKI6y1w/au8cMyR55LrxSfHeNLeUOl27MfVW 0CH02R60B06hIBVkUn3v7QWUVmKVFuKIzCqPwUuqIryr3r1mHFv5GH4GwBwLXQ== To: , , , , , , , , , , , Cc: , , , , , Date: Mon, 10 Aug 2026 20:20:54 +0800 X-Mailer: git-send-email 2.45.2 References: <20260810122057.30447-1-lizhe.67@bytedance.com> Message-Id: <20260810122057.30447-6-lizhe.67@bytedance.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 Subject: [PATCH v10 5/8] mm: extend the template fast path to zone-device compound tails X-Lms-Return-Path: From: "Li Zhe" In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> Content-Transfer-Encoding: quoted-printable X-Original-From: Li Zhe Content-Type: text/plain; charset="utf-8" The template fast path from the previous patch only accelerates head pages. Compound tails in memmap_init_compound() still go through the zone_device_page_init_slow() one by one. Build separate head and tail templates and reuse one prepared tail template across the tail pages in a compound range. Head pages preserve the existing refcount policy, while compound tails always start with a refcount of 0 after prep_compound_tail(). This extends the template-copy fast path to pfns_per_compound > 1 without changing the existing zone_device_page_init_slow() helper. Tail-page PFN-dependent fields are refreshed in the reusable tail template before each copy. Do not keep a separate non-template fallback for compound tails either. These pages are still under memmap initialization, and the initialization-time refcount updates are not part of the observable lifetime of pages handed out later. The impact is controlled for the same reason as for head pages. The first tail page still seeds the reusable tail template through the normal tail initialization sequence, and the copied tail pages have the same final initialized state except for the PFN-dependent fields refreshed before each copy. Tested in a VM with a 100 GB devdax namespace (align=3D2097152) on Intel Ice Lake server. This test exercises the dax_pmem rebind path and measures memmap initialization latency. Test procedure: Unbind and rebind the dax_pmem driver 30 times, collect memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.2-rc1): Average of rebinds for dax_pmem driver: 273.31 ms With this patch and its prerequisites applied: Average of rebinds for dax_pmem driver: 244.37 ms This reduces the average memmap initialization time measured during rebind from 273.31 ms to 244.37 ms, or about 10.6%. Signed-off-by: Li Zhe --- mm/mm_init.c | 31 ++++++++++++++++++++++++++++--- 1 file changed, 28 insertions(+), 3 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 56a36a71ba89..9691fa2a060d 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1065,6 +1065,16 @@ static void __ref zone_device_page_init_slow(struct = page *page, set_page_count(page, 0); } =20 +static inline void zone_device_tail_page_init(struct page *page, + unsigned long pfn, unsigned long zone_idx, int nid, + struct dev_pagemap *pgmap, const struct page *head, + unsigned int order) +{ + zone_device_page_init_slow(page, pfn, zone_idx, nid, pgmap); + prep_compound_tail(page, head, order); + set_page_count(page, 0); +} + /* * 'template' is a reusable page prototype rather than a strictly immutable * object. Most ZONE_DEVICE fields stay constant across the pages covered = by @@ -1126,6 +1136,7 @@ static void __ref memmap_init_compound(struct page *h= ead, { unsigned long pfn, end_pfn =3D head_pfn + nr_pages; unsigned int order =3D pgmap->vmemmap_shift; + struct page template; =20 /* * We have to initialize the pages, including setting up page links. @@ -1134,12 +1145,26 @@ static void __ref memmap_init_compound(struct page = *head, * the pages in the same go. */ __SetPageHead(head); + for (pfn =3D head_pfn + 1; pfn < end_pfn; pfn++) { struct page *page =3D pfn_to_page(pfn); =20 - zone_device_page_init_slow(page, pfn, zone_idx, nid, pgmap); - prep_compound_tail(page, head, order); - set_page_count(page, 0); + if (pfn =3D=3D head_pfn + 1) { + /* + * All tails of the same compound page share the + * state established by prep_compound_tail(). Reuse + * one tail template for the whole range and + * refresh only the PFN-dependent fields in that + * template before each copy. + */ + zone_device_tail_page_init(page, pfn, zone_idx, nid, + pgmap, head, order); + /* init template page */ + memcpy(&template, page, sizeof(*page)); + } else { + zone_device_page_init_from_template(page, pfn, + &template); + } } prep_compound_head(head, order); } --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-1-113.ptr.blmpb.com (va-1-113.ptr.blmpb.com [209.127.230.113]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 064713C1D7C for ; Mon, 10 Aug 2026 12:24:18 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.230.113 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364660; cv=none; b=E8Z774bqnAjU5mas9DAenUccTq6nRIgRmYXu1JwZFBTcEATAHrQx0ScvpI+B6lStvF5URLPUTuymfLz17z0X8u3LWA/X3PSmxMFAJY8f9p6jSoztS43Y+qxBKI0FIwlUy+Ku9LEwWLQi8InRtIBJAHqe4FPuimU0O7HGsWVua9w= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364660; c=relaxed/simple; bh=EfJYEoeIj+mCjp87mw4swqkDiGUflYopMu+3ZYEHKXk=; h=To:Message-Id:Mime-Version:Content-Type:In-Reply-To:Cc:Subject: Date:References:From; b=WOGdQPnOWkiYIoN+kriAkN1D0xIGso77vZ1MeyxFCAs6z9eFUExFWbBh1RSHdFVMBHkrauh+XcN/U4+shP0TA4QwHi4Qd3WCZOORP7K8RpniNw25cX92J8cejfaf2Xlwz6sIW9OyqYl9jCeHESmStHw7sZB3XmiQFyCuwlW18Cw= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=ASgSwMih; arc=none smtp.client-ip=209.127.230.113 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="ASgSwMih" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364653; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=vLcTykUrPpLwId0/keP/ANeuOzuTtpvXBhogtTbcEIQ=; b=ASgSwMihzIWUeCxi5uX8/oVqtNEpF9OANN0v5um83LWOXccOe/rN12Kxvt1QlZQEQ9I1NZ MJ+FCcLP5p2j7n6TrQjbHAdz4km/pmeS6XUmBuqjjqFikZwQ9MaeAWnmuriQW+i3EDe4G1 RowrgxetzsBgCxXFh3gWxlVaFAErV7TCWSI3KBm+xDxa3/06KDUbn5/zIqZGN4R9AFsTZv gGKgMBa2cU9bTJp4MzVKShu7RiI85L9mQzI5a6U5wb11DN8bdVwFI/yzcpdfqCRUHTDjBM xh0sgD9eo+kx/wGGXYaU2B+TxDmQzpkcq16MQBZ+DgtQczgDB3E0O3Zko3ee7Q== To: , , , , , , , , , , , Message-Id: <20260810122057.30447-7-lizhe.67@bytedance.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 X-Mailer: git-send-email 2.45.2 In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> Cc: , , , , , Subject: [PATCH v10 6/8] string: introduce memcpy_nontemporal() Date: Mon, 10 Aug 2026 20:20:55 +0800 X-Lms-Return-Path: References: <20260810122057.30447-1-lizhe.67@bytedance.com> From: "Li Zhe" X-Original-From: Li Zhe Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" Introduce memcpy_nontemporal() for write-once copy sites that want a named non-temporal copy primitive. On x86_64, override the helper in arch/x86/include/asm/string_64.h using the usual self-macro pattern, next to the existing memcpy_flushcache() backend that memcpy_nontemporal() wraps. include/linux/string.h provides the generic memcpy_nontemporal() fallback as #define memcpy_nontemporal(dst, src, len) \ ((void)memcpy(dst, src, len)) instead of an inline wrapper, so architectures without a specialized backend keep the usual memcpy() FORTIFY coverage when the compiler can still see object sizes at the original call site. It also makes the memcpy_nontemporal() API uniformly void, matching memcpy_flushcache() and the x86 backend, so callers cannot accidentally depend on a return value on fallback architectures. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering at that handoff point. The immediate user is the ZONE_DEVICE template-copy path. It populates struct page descriptors in a write-once pattern, so a regular cached memcpy() can incur avoidable write-allocate traffic and cache pollution for data with little near-term reuse. Signed-off-by: Li Zhe --- arch/x86/include/asm/string_64.h | 12 ++++++++++++ include/linux/string.h | 13 +++++++++++++ 2 files changed, 25 insertions(+) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string= _64.h index 4635616863f5..21ae515ae35a 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -100,6 +100,18 @@ static __always_inline void memcpy_flushcache(void *ds= t, const void *src, size_t } __memcpy_flushcache(dst, src, cnt); } + +#define memcpy_nontemporal memcpy_nontemporal +/* + * Reuse the existing x86 flushcache backend as the non-temporal copy + * primitive. + */ +static __always_inline void memcpy_nontemporal(void *dst, const void *src, + size_t cnt) +{ + memcpy_flushcache(dst, src, cnt); +} + #endif =20 #endif /* __KERNEL__ */ diff --git a/include/linux/string.h b/include/linux/string.h index 5702daca4326..6cb5cdd01158 100644 --- a/include/linux/string.h +++ b/include/linux/string.h @@ -278,6 +278,19 @@ static inline void memcpy_flushcache(void *dst, const = void *src, size_t cnt) } #endif =20 +#ifndef memcpy_nontemporal +/* + * memcpy_nontemporal() requests a non-temporal copy when the + * architecture has a suitable backend. Architectures without a + * specialized backend fall back to memcpy(). Keep this as a + * function-like macro so the compiler can still see the original + * memcpy() call site and preserve the usual FORTIFY coverage when + * object sizes remain visible there, while keeping the API void. + */ +#define memcpy_nontemporal(dst, src, len) \ + ((void)memcpy(dst, src, len)) +#endif + void *memchr_inv(const void *s, int c, size_t n); char *strreplace(char *str, char old, char new); =20 --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-2-114.ptr.blmpb.com (va-2-114.ptr.blmpb.com [209.127.231.114]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id A4E8438D for ; Mon, 10 Aug 2026 12:24:52 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.231.114 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364694; cv=none; b=lKBGSirtzDkPOAsbIrZXOhAdxGS+1Fcl653tjux24C9AL4TKhYgqpyzhiwDo6PrdGoMpKbvTsPQDAWCVfYgOi6akyTKvywYmj58VRZVvb0U6OFxQqNFtM7rRPZ8YTWW7qdEqjxus8Zbiibib1xcJ1BjD+SgcB/NMUatexKxUIt0= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364694; c=relaxed/simple; bh=DIf+MYJmJLsTk693QI8cPlBfvGaOnDkShI8XUbuFBoE=; h=Cc:Mime-Version:In-Reply-To:To:From:Subject:Message-Id: Content-Type:Date:References; b=jvJf6itqS58Pxhwy5IJLUiqA4qS1YaAVhYEu2RsPSHpthS0IFCK+mMHQ11FcFlekPeDCea2ngoqMsyIlUBw8hqJm/uNfjIzwTpdfouZqfSnsd1eZzPFB1CjGBkmsftOuaXA+ciDOz2RDWgFr0R2wtWU/ghepmN3pRhZEXQwTick= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=K+Qhhhlf; arc=none smtp.client-ip=209.127.231.114 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="K+Qhhhlf" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364687; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=noN9zN1awgD10Dy4RdMcJEDTTzNu8MQNGVp8sfEJ5vA=; b=K+QhhhlfuYiknpL6DX3R0+Z9hcTSBZnQKE+/p4StdapT3x8CImvj/c0lFOJuiQfBj/xb6R 0yBp3mUSWMflIEX9vzCZxYtFsmWxi0neIuIt1m3EKGZbAZKFr0xbxHfBmDK6Hk6qKaIZMg h5T8lzjlhgXqARby/lKlce8ZTEe1N5o8QpKuets+TAMRMHM6LDcne/e7fZ/4GmQDrf7dbK 7bKrn5iW62IwIAIIZLSOL2lRGIfn3xKu9fXHMHmhgLykRuN2E+zMrqMJuGYcNdp3WyJOzk BacxqBljCJVKaOjsc2WpKDtIYShPz++hbja+sikxyo7LOrS4e6HT7EkHjJhLNQ== Cc: , , , , , Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 X-Original-From: Li Zhe X-Lms-Return-Path: In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> To: , , , , , , , , , , , From: "Li Zhe" Subject: [PATCH v10 7/8] mm: use memcpy_nontemporal() in zone-device template copies X-Mailer: git-send-email 2.45.2 Message-Id: <20260810122057.30447-8-lizhe.67@bytedance.com> Content-Transfer-Encoding: quoted-printable Date: Mon, 10 Aug 2026 20:20:56 +0800 References: <20260810122057.30447-1-lizhe.67@bytedance.com> Content-Type: text/plain; charset="utf-8" The template fast path currently uses memcpy() for the actual struct page copy. Switch zone_device_page_init_from_template() to memcpy_nontemporal(). ZONE_DEVICE memmap initialization is largely write-once: each struct page is populated once, and most destination cachelines are not expected to be reused immediately afterwards. On x86, a regular cached memcpy() can therefore incur write-allocate traffic by pulling destination cachelines into the cache before writeback, and can populate the cache with data that has little near-term reuse. Using memcpy_nontemporal() lets this path request nontemporal stores for that copy pattern, which can reduce cache pollution and avoid part of the associated write-allocate overhead, while architectures without a specialized backend still fall back to memcpy(). Do not add a KASAN/KMSAN-specific fallback around this call site. As Muchun pointed out, special KASAN handling for memcpy_flushcache() or memcpy_nontemporal(), if needed, belongs in the low-level helper rather than in this ZONE_DEVICE caller. No separate drain is added here. memcpy_nontemporal() is used only as the copy primitive while memmap_init_zone_device() is still initializing the struct page array. The ordinary stores that follow in this path, such as compound-page setup, are part of the same CPU's initialization sequence; they are not used as a publication store that tells another CPU or device to consume data written by the non-temporal copy. Therefore this call site does not need a helper-level drain for correctness. Callers that use memcpy_nontemporal() as part of a producer-consumer or device-visible handoff must add the required ordering themselves. Tested in a VM with a 100 GB fsdax namespace device configured with map=3Ddev and a 100 GB devdax namespace (align=3D2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.2-rc1): Average of rebinds for nd_pmem driver: 244.28 ms Average of rebinds for dax_pmem driver: 273.31 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 150.83 ms Average of rebinds for dax_pmem driver: 153.55 ms This reduces the average memmap initialization time measured during rebind by about 38.3% for nd_pmem and 43.8% for dax_pmem. Signed-off-by: Li Zhe --- mm/mm_init.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 9691fa2a060d..bb2007806a28 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1101,7 +1101,7 @@ static void zone_device_page_init_from_template(struc= t page *page, * to the destination page. */ zone_device_page_update_template(template, pfn); - memcpy(page, template, sizeof(*page)); + memcpy_nontemporal(page, template, sizeof(*page)); } =20 /* --=20 2.20.1 From nobody Tue Sep 29 08:26:17 2026 Received: from va-2-114.ptr.blmpb.com (va-2-114.ptr.blmpb.com [209.127.231.114]) (using TLSv1.2 with cipher ECDHE-RSA-AES128-GCM-SHA256 (128/128 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 1039F3C343C for ; Mon, 10 Aug 2026 12:25:16 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=209.127.231.114 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364718; cv=none; b=ByK3CLcF71bEg0V4fmjmHK1E88bQi9PZdPNtMB/4+c6Iza4ByKD4FOE9D/u6lIVWun3IaAXsV3bJaasYjbEvLu5uCQIRS7GvHE0axqVZQWflTsZtyNGWCIBUjJb9upTg+0XdqxYiVMeuauQCRe71CQ0BPm2EcOFdv2PfsEANFQU= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1786364718; c=relaxed/simple; bh=7Sbay6owjwp61KH6sE/CioV+gRC8N8OjwI7PcMzzr6M=; h=Cc:From:To:Message-Id:In-Reply-To:Subject:Date:Mime-Version: References:Content-Type; b=pPNe2WzgRCCAFwcD8OwZrMX8rDMc4RvIHTQW4l7JU/stamiUwKvOOYu5TGuPYu/C0NaifGMzCfrU0Q6bHwyz+QM5yE3SDY6xm9AwXnjnRYCrTcswKiJe2KvMmnm0gPOOO5HzEXOaAO6i9KAA+vvIZBo0FJewFY4Ol599eTCcldg= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com; spf=pass smtp.mailfrom=bytedance.com; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b=RB3opE5L; arc=none smtp.client-ip=209.127.231.114 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=quarantine dis=none) header.from=bytedance.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=bytedance.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=bytedance.com header.i=@bytedance.com header.b="RB3opE5L" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; s=2212171451; d=bytedance.com; t=1786364711; h=from:subject: mime-version:from:date:message-id:subject:to:cc:reply-to:content-type: mime-version:in-reply-to:message-id; bh=TZuj0HlrM29RM25pBmz0Kgx3giTvhF2mK+AyOkzDqng=; b=RB3opE5LVCgkGE5rfHDkFT1Tfq+6NbTyih9BYBYGaPHlL59rpvKi/WxrX/7pluVH19X6a6 kfxyru4nPf9HKjT/U3NI6n6XdHdIcX5KcZcv6rtMw83BgJdxsSEv7ZWArjTPg6CgvoNkPn MKKd3FLTVoZ22AESQi0sg4BpYhjYa9bzolWzFWUUmTq6r0ebZZ3AWRdbUnAjyWKQDtWbU1 M9FAYokMUFydSyZvy7AiUYMXTToOW9tyMODipuvWabQ85+nzDQ17egsbGWBJPH5VeZ7g8Z ODApinJa6WrdmrxR9tF5a2b35KJ0OSq4FsCNPdYt+6P8j5eSW41Tufs6UBvNew== Cc: , , , , , From: "Li Zhe" To: , , , , , , , , , , , Message-Id: <20260810122057.30447-9-lizhe.67@bytedance.com> In-Reply-To: <20260810122057.30447-1-lizhe.67@bytedance.com> Subject: [PATCH v10 8/8] x86/string: extend memcpy_flushcache() fixed-size fastpaths Date: Mon, 10 Aug 2026 20:20:57 +0800 Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: Mime-Version: 1.0 Content-Transfer-Encoding: quoted-printable X-Lms-Return-Path: X-Original-From: Li Zhe X-Mailer: git-send-email 2.45.2 References: <20260810122057.30447-1-lizhe.67@bytedance.com> Content-Type: text/plain; charset="utf-8" The x86 memcpy_nontemporal() helper maps to memcpy_flushcache(), and the ZONE_DEVICE template-copy path uses it to copy one struct page at a time. The relevant copy size is sizeof(struct page). On x86_64, the base struct page layout is 64 bytes. Adding either the KMSAN metadata pointers or an out-of-flags last_cpupid field can make it 80 bytes after alignment, and enabling both can make it 96 bytes. memcpy_flushcache() currently only has inline fixed-size cases for 4, 8, and 16 bytes. As a result, these constant-sized struct page copies fall through to __memcpy_flushcache() even though the compiler knows the copy size at the call site. Add fixed-size MOVNTI cases up to 96 bytes so the ZONE_DEVICE template-copy path can keep these struct page copies in the inline memcpy_flushcache() path. This matters for ZONE_DEVICE memmap initialization because the copy happens once per initialized struct page. For a 100 GB fsdax namespace with map=3Ddev, this is about 25 million struct page copies during nd_pmem binding or rebinding. Tested in a VM with a 100 GB fsdax namespace device configured with map=3Ddev and a 100 GB devdax namespace (align=3D2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). With memcpy_nontemporal() used by the ZONE_DEVICE template-copy path: Average of rebinds for nd_pmem driver: 150.83 ms Average of rebinds for dax_pmem driver: 153.55 ms With this x86 fixed-size fastpath patch applied: Average of rebinds for nd_pmem driver: 96.79 ms Average of rebinds for dax_pmem driver: 119.04 ms This further reduces the average memmap initialization time measured during rebind by about 35.8% for nd_pmem and 22.5% for dax_pmem. Suggested-by: Borislav Petkov Signed-off-by: Li Zhe Acked-by: Borislav Petkov (AMD) --- arch/x86/include/asm/string_64.h | 71 +++++++++++++++++++++++++------- 1 file changed, 56 insertions(+), 15 deletions(-) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string= _64.h index 21ae515ae35a..831d3dda3b38 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -82,23 +82,64 @@ int strcmp(const char *cs, const char *ct); #ifdef CONFIG_ARCH_HAS_UACCESS_FLUSHCACHE #define __HAVE_ARCH_MEMCPY_FLUSHCACHE 1 void __memcpy_flushcache(void *dst, const void *src, size_t cnt); -static __always_inline void memcpy_flushcache(void *dst, const void *src, = size_t cnt) + +static __always_inline void movnti_4(void *dst, const void *src) +{ + asm volatile("movntil %1, %0" + : "=3Dm"(*(u32 *)dst) + : "r"(*(const u32 *)src) + : "memory"); +} + +static __always_inline void movnti_8(void *dst, const void *src) +{ + asm volatile("movntiq %1, %0" + : "=3Dm"(*(u64 *)dst) + : "r"(*(const u64 *)src) + : "memory"); +} + +static __always_inline void movnti_16(void *dst, const void *src) +{ + movnti_8(dst, src); + movnti_8(dst + 8, src + 8); +} + +static __always_inline void movnti_32(void *dst, const void *src) +{ + movnti_16(dst, src); + movnti_16(dst + 16, src + 16); +} + +static __always_inline void movnti_64(void *dst, const void *src) +{ + movnti_32(dst, src); + movnti_32(dst + 32, src + 32); +} + +static __always_inline void memcpy_flushcache(void *dst, const void *src, + size_t cnt) { - if (__builtin_constant_p(cnt)) { - switch (cnt) { - case 4: - asm ("movntil %1, %0" : "=3Dm"(*(u32 *)dst) : "r"(*(u32 *)src)); - return; - case 8: - asm ("movntiq %1, %0" : "=3Dm"(*(u64 *)dst) : "r"(*(u64 *)src)); - return; - case 16: - asm ("movntiq %1, %0" : "=3Dm"(*(u64 *)dst) : "r"(*(u64 *)src)); - asm ("movntiq %1, %0" : "=3Dm"(*(u64 *)(dst + 8)) : "r"(*(u64 *)(src += 8))); - return; - } + if (!__builtin_constant_p(cnt)) + return __memcpy_flushcache(dst, src, cnt); + + /* + * The relevant fixed-size copies here are the x86_64 struct page sizes: + * 64, 80, and 96 bytes. Keep 32-byte and 48-byte copies inline as well + * instead of sending those nearby fixed-size cases back to + * __memcpy_flushcache(). + */ + switch (cnt) { + case 4: movnti_4(dst, src); break; + case 8: movnti_8(dst, src); break; + case 16: movnti_16(dst, src); break; + case 32: movnti_32(dst, src); break; + case 48: movnti_32(dst, src); movnti_16(dst + 32, src + 32); break; + case 64: movnti_64(dst, src); break; + case 80: movnti_64(dst, src); movnti_16(dst + 64, src + 64); break; + case 96: movnti_64(dst, src); movnti_32(dst + 64, src + 64); break; + default: __memcpy_flushcache(dst, src, cnt); break; } - __memcpy_flushcache(dst, src, cnt); } =20 #define memcpy_nontemporal memcpy_nontemporal --=20 2.20.1