From nobody Mon Sep 28 19:23:39 2026 Received: from mgamail.intel.com (mgamail.intel.com [198.175.65.20]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 9208943B4B5 for ; Tue, 18 Aug 2026 08:53:25 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=198.175.65.20 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1787043207; cv=none; b=uM7FfpA9xXbBHuv2N74la4/A7zM3nCfq7dbNZxptxCQM6Zbz4ROqzG82GWY6cPidsXv37F0NZru/7g2hctZsCxU4DyAPtw/SjCtZcjujmQ2vx0ZAAKmYgC1z9o2uQHGO8UCwuG2tvbYXgj7Z/4vNIauTkERTzB8lo6NXmTiV5UY= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1787043207; c=relaxed/simple; bh=49j4fxstgTRL5iumzWsULPLrhk82aBqfzo4vGvGaiyg=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=NDvxtUMV3hmhL1y3SZ3XTXvQQe2AbsZ3zBkGIGGx8Yo/FcGkZu/e8quQSjRI59wddn5wZjHHVljokMJbNi5IXj90d56tIt9BSb4eNDQ4ahHEl6wAENkZ6GsMzvBcR8ZrH9iskNh0IjEWxPrhJYTk51AIVTrlOxXL8GLxOq4P224= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=intel.com; spf=pass smtp.mailfrom=intel.com; dkim=pass (2048-bit key) header.d=intel.com header.i=@intel.com header.b=FQOm7/4W; arc=none smtp.client-ip=198.175.65.20 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=intel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=intel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=intel.com header.i=@intel.com header.b="FQOm7/4W" DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/simple; d=intel.com; i=@intel.com; q=dns/txt; s=Intel; t=1787043205; x=1818579205; h=from:to:cc:subject:date:message-id:in-reply-to: references:mime-version:content-transfer-encoding; bh=49j4fxstgTRL5iumzWsULPLrhk82aBqfzo4vGvGaiyg=; b=FQOm7/4WtF7OtFG5hlhhD6YGOWonPoKC/4ur/Lp3lOsVv7uLud1v56NM sdrrVDf9fUCKLUom5+TSlqe7Iv/2Fv71cwFcdcJU9Yrfqyi2xbN+bMywF qiQ8hBGbcSB61VtkqecXqnlR97uTBTbR6SUhs+2AD0kd68rv2mwiaFSp6 dt+8/heXapKFUyNVAr+Hiy8G2rhqoic9kxdOQZ1aSoO4sNEJNJGiYHnQM 4FPcM8AP2yLhAnI8oQnLLIXZGxMrtUxgH5U43oFrHRZ8ooZCi6JluJCkI ay9OTb4e/W+6hRyT7pYu766hV1raBzMPwqlkB0Zr/k+g1KtZ9VpcK9li5 Q==; X-CSE-ConnectionGUID: ap4oNmgwRSacj0fgs+v/vw== X-CSE-MsgGUID: HElocHNHQPikPL0Tqwz6wA== X-IronPort-AV: E=McAfee;i="6800,10657,11878"; a="87292221" X-IronPort-AV: E=Sophos;i="6.25,230,1779174000"; d="scan'208";a="87292221" Received: from fmviesa008.fm.intel.com ([10.60.135.148]) by orvoesa112.jf.intel.com with ESMTP/TLS/ECDHE-RSA-AES256-GCM-SHA384; 18 Aug 2026 01:53:25 -0700 X-CSE-ConnectionGUID: DTsgAjWrQmG36uO08YDwbA== X-CSE-MsgGUID: F31xNNF9Sxq373JezwsqPA== X-ExtLoop1: 1 X-IronPort-AV: E=Sophos;i="6.25,230,1779174000"; d="scan'208";a="262539338" Received: from spr10.sh.intel.com (HELO localhost) ([10.239.23.75]) by fmviesa008.fm.intel.com with ESMTP; 18 Aug 2026 01:53:21 -0700 From: Yuan Liu To: David Hildenbrand , Oscar Salvador , Mike Rapoport , Wei Yang Cc: linux-mm@kvack.org, Nanhai Zou , Chen Zhang , Yuan Liu , Jason Zeng , Chen Yu , Pan Deng , Tianyou Li , linux-kernel@vger.kernel.org Subject: [PATCH v7 1/2] mm/memory_hotplug: make shrink_zone_span() more robust Date: Tue, 18 Aug 2026 04:57:01 -0400 Message-ID: <20260818085702.3395529-2-yuan1.liu@intel.com> X-Mailer: git-send-email 2.47.3 In-Reply-To: <20260818085702.3395529-1-yuan1.liu@intel.com> References: <20260818085702.3395529-1-yuan1.liu@intel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" From: "David Hildenbrand (Arm)" Let's make shrink_zone_span() more robust by checking in find_smallest_section_pfn() / find_biggest_section_pfn() that the start and end PFNs of the subsection are within the zone. While at it, clean up the function by factoring the core check out into subsection_overlaps_zone(). There likely is no need to check the nid first. We require SPARSEMEM_VMEMMAP_ENABLE, where pfn_to_page() is cheap, and pfn_to_nid() on CONFIG_NUMA would call pfn_to_page() either way. So let's just drop that for now. Signed-off-by: David Hildenbrand (Arm) Tested-by: Yuan Liu Signed-off-by: Yuan Liu --- mm/memory_hotplug.c | 59 ++++++++++++++++++--------------------------- 1 file changed, 24 insertions(+), 35 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index 7ac19fab2263..cd82e79f0782 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -422,49 +422,39 @@ int __add_pages(int nid, unsigned long pfn, unsigned = long nr_pages, return err; } =20 -/* find the smallest valid pfn in the range [start_pfn, end_pfn) */ -static unsigned long find_smallest_section_pfn(int nid, struct zone *zone, - unsigned long start_pfn, - unsigned long end_pfn) +static bool subsection_overlaps_zone(unsigned long pfn, struct zone *zone) { - for (; start_pfn < end_pfn; start_pfn +=3D PAGES_PER_SUBSECTION) { - if (unlikely(!pfn_to_online_page(start_pfn))) - continue; + const unsigned long start_pfn =3D ALIGN_DOWN(pfn, PAGES_PER_SUBSECTION); + const unsigned long end_pfn =3D start_pfn + PAGES_PER_SUBSECTION - 1; =20 - if (unlikely(pfn_to_nid(start_pfn) !=3D nid)) - continue; + /* All pages in a subsection are either online or offline. */ + if (unlikely(!pfn_to_online_page(start_pfn))) + return false; =20 - if (zone !=3D page_zone(pfn_to_page(start_pfn))) - continue; + /* Checking start+end is sufficient. */ + return zone =3D=3D page_zone(pfn_to_page(start_pfn)) || + zone =3D=3D page_zone(pfn_to_page(end_pfn)); +} =20 - return start_pfn; +/* find the smallest valid pfn in the range [start_pfn, end_pfn) */ +static unsigned long find_smallest_section_pfn(struct zone *zone, + unsigned long start_pfn, unsigned long end_pfn) +{ + for (; start_pfn < end_pfn; start_pfn +=3D PAGES_PER_SUBSECTION) { + if (subsection_overlaps_zone(start_pfn, zone)) + return start_pfn; } - return 0; } =20 /* find the biggest valid pfn in the range [start_pfn, end_pfn). */ -static unsigned long find_biggest_section_pfn(int nid, struct zone *zone, - unsigned long start_pfn, - unsigned long end_pfn) +static unsigned long find_biggest_section_pfn(struct zone *zone, + unsigned long start_pfn, unsigned long end_pfn) { - unsigned long pfn; - - /* pfn is the end pfn of a memory section. */ - pfn =3D end_pfn - 1; - for (; pfn >=3D start_pfn; pfn -=3D PAGES_PER_SUBSECTION) { - if (unlikely(!pfn_to_online_page(pfn))) - continue; - - if (unlikely(pfn_to_nid(pfn) !=3D nid)) - continue; - - if (zone !=3D page_zone(pfn_to_page(pfn))) - continue; - - return pfn; + for (; end_pfn >=3D start_pfn; end_pfn -=3D PAGES_PER_SUBSECTION) { + if (subsection_overlaps_zone(end_pfn - 1, zone)) + return end_pfn - 1; } - return 0; } =20 @@ -472,7 +462,6 @@ static void shrink_zone_span(struct zone *zone, unsigne= d long start_pfn, unsigned long end_pfn) { unsigned long pfn; - int nid =3D zone_to_nid(zone); =20 if (zone->zone_start_pfn =3D=3D start_pfn) { /* @@ -481,7 +470,7 @@ static void shrink_zone_span(struct zone *zone, unsigne= d long start_pfn, * In this case, we find second smallest valid mem_section * for shrinking zone. */ - pfn =3D find_smallest_section_pfn(nid, zone, end_pfn, + pfn =3D find_smallest_section_pfn(zone, end_pfn, zone_end_pfn(zone)); if (pfn) { zone->spanned_pages =3D zone_end_pfn(zone) - pfn; @@ -497,7 +486,7 @@ static void shrink_zone_span(struct zone *zone, unsigne= d long start_pfn, * In this case, we find second biggest valid mem_section for * shrinking zone. */ - pfn =3D find_biggest_section_pfn(nid, zone, zone->zone_start_pfn, + pfn =3D find_biggest_section_pfn(zone, zone->zone_start_pfn, start_pfn); if (pfn) zone->spanned_pages =3D pfn - zone->zone_start_pfn + 1; --=20 2.47.3 From nobody Mon Sep 28 19:23:39 2026 Received: from mgamail.intel.com (mgamail.intel.com [198.175.65.20]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id AB0B143CEF8 for ; Tue, 18 Aug 2026 08:53:29 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=198.175.65.20 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1787043212; cv=none; b=CzwXmZAf42OhglSb/jAML/LmPz/JEFFSbxLzq/sFxc0ZCZ3uiRazSh0FCqmrZYCsAixd6pmbRROyH9NYno26vmI7bHT9DakdqnHAR4iQbI2Pl9DQog/McCzzAj6NXjZ96OVC2avmP0skKGGch6UFftp3QMIrRFlF5Jl7LFeqzw0= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1787043212; c=relaxed/simple; bh=Z+RYv5HKf5AEjLqXKBXVdkhl7Q55X+j0OFSE/dr+2xY=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=fv2KuoFuBHPAneJt3rwDgKhyo+86Pw9mefei/yzXKyWtfxI9y50RH/cnb2rJgvsurtVWa+JKi2nb3F4bPvT15lP388R7cAs2MBFjCB8xGa6lkVw79Zkmyt7oNudBz2jr5iRHegvgw3IFpgxJg4drHECXU3Ocq64pIkMH0zC2laU= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=intel.com; spf=pass smtp.mailfrom=intel.com; dkim=pass (2048-bit key) header.d=intel.com header.i=@intel.com header.b=jkM/GuFZ; arc=none smtp.client-ip=198.175.65.20 Authentication-Results: smtp.subspace.kernel.org; dmarc=pass (p=none dis=none) header.from=intel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=intel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=intel.com header.i=@intel.com header.b="jkM/GuFZ" DKIM-Signature: v=1; a=rsa-sha256; c=relaxed/simple; d=intel.com; i=@intel.com; q=dns/txt; s=Intel; t=1787043209; x=1818579209; h=from:to:cc:subject:date:message-id:in-reply-to: references:mime-version:content-transfer-encoding; bh=Z+RYv5HKf5AEjLqXKBXVdkhl7Q55X+j0OFSE/dr+2xY=; b=jkM/GuFZ1s3m+VgOC9mCbgQFd1o3CIouXnkgJ4DgyNoy8ebxqxL5Hkef VOnOgxhQ7L8sT+xM0KFxAmW6czELq6iGtJW1EudlY+1hyV/gSb63VAMVN GZVEwLCvdskkPHRXNzFH9hK0GM3xEpmrnoHOGJteQvMS0Ebyp7AOHsBC7 xAD7ws7imszdB4GcB6C2d7OKCKMvdY3gmh4PUVpBKeBurJwV/b5FcmOxR zn5hfufpkIpSpIviwKWa+4ra12VBF/NlQs1aXoW3zwFqTbGTb1kbXFk2+ rfCWec7gc8HoK9MMGjTxhDT0bw+99x4rz3+E0H3OnKTwdQ+FAH0UFUXoI w==; X-CSE-ConnectionGUID: t/2jM3HmRtqybdYBIZI+ZA== X-CSE-MsgGUID: deXtq/vURhSKo3gY196Heg== X-IronPort-AV: E=McAfee;i="6800,10657,11878"; a="87292234" X-IronPort-AV: E=Sophos;i="6.25,230,1779174000"; d="scan'208";a="87292234" Received: from fmviesa008.fm.intel.com ([10.60.135.148]) by orvoesa112.jf.intel.com with ESMTP/TLS/ECDHE-RSA-AES256-GCM-SHA384; 18 Aug 2026 01:53:29 -0700 X-CSE-ConnectionGUID: +WrKAR2iQQerXN6TqB9glw== X-CSE-MsgGUID: +31DWMrmSnCH1iYj1Vyfdg== X-ExtLoop1: 1 X-IronPort-AV: E=Sophos;i="6.25,230,1779174000"; d="scan'208";a="262539343" Received: from spr10.sh.intel.com (HELO localhost) ([10.239.23.75]) by fmviesa008.fm.intel.com with ESMTP; 18 Aug 2026 01:53:26 -0700 From: Yuan Liu To: David Hildenbrand , Oscar Salvador , Mike Rapoport , Wei Yang Cc: linux-mm@kvack.org, Nanhai Zou , Chen Zhang , Yuan Liu , Jason Zeng , Chen Yu , Pan Deng , Tianyou Li , linux-kernel@vger.kernel.org Subject: [PATCH v7 2/2] mm/memory_hotplug: optimize zone contiguous check when changing pfn range Date: Tue, 18 Aug 2026 04:57:02 -0400 Message-ID: <20260818085702.3395529-3-yuan1.liu@intel.com> X-Mailer: git-send-email 2.47.3 In-Reply-To: <20260818085702.3395529-1-yuan1.liu@intel.com> References: <20260818085702.3395529-1-yuan1.liu@intel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" When move_pfn_range_to_zone() or remove_pfn_range_from_zone() updates a zone, set_zone_contiguous() rescans the entire zone pageblock-by-pageblock to rebuild zone->contiguous. For large zones this is a significant cost during memory hotplug and hot-unplug. Add a new zone member, pages_with_online_memmap, that tracks the number of pages within the zone span that have an online memory map, including present pages and memory holes whose memory map has been initialized and for which pfn_to_online_page() succeeds. For early boot memory, pages_with_online_memmap is calculated in memmap_init_zone_range(). PFNs initialized by memmap_init_range() are included in pages_with_online_memmap, and hole PFNs for which pfn_to_online_page() succeeds are also counted in init_unavailable_range(). For hotplugged memory, pages_with_online_memmap is updated through adjust_present_page_count(), which is called during memory online and offline operations. When spanned_pages =3D=3D pages_with_online_memmap, every PFN in the zone span has a valid memmap entry, so pfn_to_page() can be called for any PFN within the zone span without an additional pfn_valid() check. The counter may temporarily undercount when pages with an online memory map exist outside the current zone span. This can only happen during boot, when initializing the memory map of pages that do not fall into any zone span. Growing the zone to cover such pages and later shrinking it back may result in a value that is too small. This is safe, as it merely prevents detecting a contiguous zone. The contiguity check using pages_with_online_memmap is stricter than the old pageblock-by-pageblock scan. The old set_zone_contiguous() iterated at pageblock granularity via pageblock_pfn_to_page(), so a zone could be marked contiguous even if a subsection-sized hole existed within a pageblock. The new check requires spanned_pages =3D=3D pages_with_online_memmap, meaning every PFN in the zone span must satisfy pfn_to_online_page(). The following test cases of memory hotplug for a VM [1], tested in the environment [2], show that this optimization can significantly reduce the memory hotplug time [3]. +----------------+------+---------------+--------------+----------------+ | | Size | Time (before) | Time (after) | Time Reduction | | +------+---------------+--------------+----------------+ | Plug Memory | 256G | 10s | 3s | 70% | | +------+---------------+--------------+----------------+ | | 512G | 36s | 7s | 81% | +----------------+------+---------------+--------------+----------------+ +----------------+------+---------------+--------------+----------------+ | | Size | Time (before) | Time (after) | Time Reduction | | +------+---------------+--------------+----------------+ | Unplug Memory | 256G | 11s | 4s | 64% | | +------+---------------+--------------+----------------+ | | 512G | 36s | 9s | 75% | +----------------+------+---------------+--------------+----------------+ [1] Qemu commands to hotplug 256G/512G memory for a VM: object_add memory-backend-ram,id=3Dhotmem0,size=3D256G/512G,share=3Don device_add virtio-mem-pci,id=3Dvmem1,memdev=3Dhotmem0,bus=3Dport1 qom-set vmem1 requested-size 256G/512G (Plug Memory) qom-set vmem1 requested-size 0G (Unplug Memory) [2] Hardware : Intel Icelake server Guest Kernel : v7.2-rc1 Qemu : v9.0.0 Launch VM : qemu-system-x86_64 -accel kvm -cpu host \ -drive file=3D./Centos10_cloud.qcow2,format=3Dqcow2,if=3Dvirtio \ -drive file=3D./seed.img,format=3Draw,if=3Dvirtio \ -smp 3,cores=3D3,threads=3D1,sockets=3D1,maxcpus=3D3 \ -m 2G,slots=3D10,maxmem=3D2052472M \ -device pcie-root-port,id=3Dport1,bus=3Dpcie.0,slot=3D1,multifunction= =3Don \ -device pcie-root-port,id=3Dport2,bus=3Dpcie.0,slot=3D2 \ -nographic -machine q35 \ -nic user,hostfwd=3Dtcp::3000-:22 Guest kernel auto-onlines newly added memory blocks: echo online > /sys/devices/system/memory/auto_online_blocks [3] The time from typing the QEMU commands in [1] to when the output of 'grep MemTotal /proc/meminfo' on Guest reflects that all hotplugged memory is recognized. Reported-by: Nanhai Zou Reported-by: Chen Zhang Tested-by: Yuan Liu Reviewed-by: Jason Zeng Reviewed-by: Chen Yu Reviewed-by: Pan Deng Co-developed-by: Tianyou Li Signed-off-by: Tianyou Li Signed-off-by: Yuan Liu --- Documentation/mm/physical_memory.rst | 6 +++ drivers/base/memory.c | 7 ++- include/linux/mmzone.h | 48 ++++++++++++++++++ mm/internal.h | 8 +-- mm/memory_hotplug.c | 12 +---- mm/mm_init.c | 73 +++++++++++++++++----------- 6 files changed, 108 insertions(+), 46 deletions(-) diff --git a/Documentation/mm/physical_memory.rst b/Documentation/mm/physic= al_memory.rst index b76183545e5b..ceea6c8b27cc 100644 --- a/Documentation/mm/physical_memory.rst +++ b/Documentation/mm/physical_memory.rst @@ -483,6 +483,12 @@ General ``present_pages`` should use ``get_online_mems()`` to get a stable value= . It is initialized by ``calculate_node_totalpages()``. =20 +``pages_with_online_memmap`` + Pages within the zone that have an online memory map: present pages and + memory holes whose memory map has been initialized and + ``pfn_to_online_page()`` succeeds. See the comment for + ``pages_with_online_memmap`` in ``include/linux/mmzone.h`` for more deta= ils. + ``present_early_pages`` The present pages existing within the zone located on memory available s= ince early boot, excluding hotplugged memory. Defined only when diff --git a/drivers/base/memory.c b/drivers/base/memory.c index bcfe2d9f4adb..97699be9a357 100644 --- a/drivers/base/memory.c +++ b/drivers/base/memory.c @@ -246,6 +246,7 @@ static int memory_block_online(struct memory_block *mem) nr_vmemmap_pages =3D mem->altmap->free; =20 mem_hotplug_begin(); + clear_zone_contiguous(zone); if (nr_vmemmap_pages) { ret =3D mhp_init_memmap_on_memory(start_pfn, nr_vmemmap_pages, zone); if (ret) @@ -270,6 +271,7 @@ static int memory_block_online(struct memory_block *mem) =20 mem->zone =3D zone; out: + set_zone_contiguous(zone); mem_hotplug_done(); return ret; } @@ -295,6 +297,7 @@ static int memory_block_offline(struct memory_block *me= m) nr_vmemmap_pages =3D mem->altmap->free; =20 mem_hotplug_begin(); + clear_zone_contiguous(mem->zone); if (nr_vmemmap_pages) adjust_present_page_count(pfn_to_page(start_pfn), mem->group, -nr_vmemmap_pages); @@ -312,8 +315,10 @@ static int memory_block_offline(struct memory_block *m= em) if (nr_vmemmap_pages) mhp_deinit_memmap_on_memory(start_pfn, nr_vmemmap_pages); =20 - mem->zone =3D NULL; out: + set_zone_contiguous(mem->zone); + if (!ret) + mem->zone =3D NULL; mem_hotplug_done(); return ret; } diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index ca2712187147..58f342de3bac 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -1032,6 +1032,21 @@ struct zone { * cma pages is present pages that are assigned for CMA use * (MIGRATE_CMA). * + * pages_with_online_memmap tracks pages within the zone that have + * an online memory map: present pages and memory holes whose + * memory map has been initialized and pfn_to_online_page() + * succeeds. When spanned_pages =3D=3D pages_with_online_memmap, + * pfn_to_page() can be performed without further checks on any + * PFN within the zone span. + * + * Note: this counter may temporarily undercount when pages with an + * online memory map exist outside the current zone span. This can + * only happen during boot, when initializing the memory map of + * pages that do not fall into any zone span. Growing the zone to + * cover such pages and later shrinking it back may result in a + * "too small" value. This is safe: it merely prevents detecting a + * contiguous zone. + * * So present_pages may be used by memory hotplug or memory power * management logic to figure out unmanaged pages by checking * (present_pages - managed_pages). And managed_pages should be used @@ -1056,6 +1071,7 @@ struct zone { atomic_long_t managed_pages; unsigned long spanned_pages; unsigned long present_pages; + unsigned long pages_with_online_memmap; #if defined(CONFIG_MEMORY_HOTPLUG) unsigned long present_early_pages; #endif @@ -1681,6 +1697,38 @@ static inline bool zone_is_zone_device(const struct = zone *zone) } #endif =20 +/** + * zone_is_contiguous - test whether a zone is contiguous + * @zone: the zone to test. + * + * In a contiguous zone, it is valid to call pfn_to_page() on any PFN in t= he + * spanned zone without requiring pfn_valid() or pfn_to_online_page() chec= ks. + * + * Note that missing synchronization with memory offlining makes any PFN + * traversal prone to races. + * + * ZONE_DEVICE zones are always marked non-contiguous. + * + * Return: true if contiguous, otherwise false. + */ +static inline bool zone_is_contiguous(const struct zone *zone) +{ + return READ_ONCE(zone->contiguous); +} + +static inline void set_zone_contiguous(struct zone *zone) +{ + if (zone_is_zone_device(zone)) + return; + if (zone->spanned_pages =3D=3D zone->pages_with_online_memmap) + WRITE_ONCE(zone->contiguous, true); +} + +static inline void clear_zone_contiguous(struct zone *zone) +{ + WRITE_ONCE(zone->contiguous, false); +} + /* * Returns true if a zone has pages managed by the buddy allocator. * All the reclaim decisions have to use this function rather than diff --git a/mm/internal.h b/mm/internal.h index 181e79f1d6a2..f932b7577c92 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -805,21 +805,15 @@ extern struct page *__pageblock_pfn_to_page(unsigned = long start_pfn, static inline struct page *pageblock_pfn_to_page(unsigned long start_pfn, unsigned long end_pfn, struct zone *zone) { - if (zone->contiguous) + if (zone_is_contiguous(zone)) return pfn_to_page(start_pfn); =20 return __pageblock_pfn_to_page(start_pfn, end_pfn, zone); } =20 -void set_zone_contiguous(struct zone *zone); bool pfn_range_intersects_zones(int nid, unsigned long start_pfn, unsigned long nr_pages); =20 -static inline void clear_zone_contiguous(struct zone *zone) -{ - zone->contiguous =3D false; -} - extern int __isolate_free_page(struct page *page, unsigned int order); extern void __putback_isolated_page(struct page *page, unsigned int order, int mt); diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index cd82e79f0782..8dbd50d88155 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -546,18 +546,13 @@ void remove_pfn_range_from_zone(struct zone *zone, =20 /* * Zone shrinking code cannot properly deal with ZONE_DEVICE. So - * we will not try to shrink the zones - which is okay as - * set_zone_contiguous() cannot deal with ZONE_DEVICE either way. + * we will not try to shrink it. */ if (zone_is_zone_device(zone)) return; =20 - clear_zone_contiguous(zone); - shrink_zone_span(zone, start_pfn, start_pfn + nr_pages); update_pgdat_span(pgdat); - - set_zone_contiguous(zone); } =20 /** @@ -735,8 +730,6 @@ void move_pfn_range_to_zone(struct zone *zone, unsigned= long start_pfn, struct pglist_data *pgdat =3D zone->zone_pgdat; int nid =3D pgdat->node_id; =20 - clear_zone_contiguous(zone); - if (zone_is_empty(zone)) init_currently_empty_zone(zone, start_pfn, nr_pages); resize_zone_range(zone, start_pfn, nr_pages); @@ -764,8 +757,6 @@ void move_pfn_range_to_zone(struct zone *zone, unsigned= long start_pfn, memmap_init_range(nr_pages, nid, zone_idx(zone), start_pfn, 0, MEMINIT_HOTPLUG, altmap, migratetype, isolate_pageblock); - - set_zone_contiguous(zone); } =20 struct auto_movable_stats { @@ -1061,6 +1052,7 @@ void adjust_present_page_count(struct page *page, str= uct memory_group *group, if (early_section(__pfn_to_section(page_to_pfn(page)))) zone->present_early_pages +=3D nr_pages; zone->present_pages +=3D nr_pages; + zone->pages_with_online_memmap +=3D nr_pages; zone->zone_pgdat->node_present_pages +=3D nr_pages; =20 if (group && movable) diff --git a/mm/mm_init.c b/mm/mm_init.c index f1afe023e4e7..d63ee554d7eb 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -799,6 +799,28 @@ void __meminit init_deferred_page(unsigned long pfn, i= nt nid) __init_deferred_page(pfn, nid); } =20 +#ifdef CONFIG_SPARSEMEM_VMEMMAP +static bool __init unavailable_pfn_is_online(unsigned long pfn, + unsigned long *last_subsection, + bool *is_online) +{ + unsigned long subsection =3D pfn & PAGE_SUBSECTION_MASK; + + if (subsection !=3D *last_subsection) { + *is_online =3D !!pfn_to_online_page(pfn); + *last_subsection =3D subsection; + } + return *is_online; +} +#else +static inline bool unavailable_pfn_is_online(unsigned long pfn, + unsigned long *last_subsection, + bool *is_online) +{ + return true; +} +#endif + /* * Only struct pages that correspond to ranges defined by memblock.memory * are zeroed and initialized by going through __init_single_page() during @@ -822,22 +844,27 @@ void __meminit init_deferred_page(unsigned long pfn, = int nid) * zone/node above the hole except for the trailing pages in the last * section that will be appended to the zone/node below. */ -static void __init init_unavailable_range(unsigned long spfn, - unsigned long epfn, - int zone, int node) +static unsigned long __init init_unavailable_range(unsigned long spfn, + unsigned long epfn, + int zone, int node) { unsigned long pfn; - u64 pgcnt =3D 0; + u64 pgcnt =3D 0, online_pgcnt =3D 0; + unsigned long last_subsection =3D -1; + bool is_online =3D false; =20 for_each_valid_pfn(pfn, spfn, epfn) { __init_single_page(pfn_to_page(pfn), pfn, zone, node); __SetPageReserved(pfn_to_page(pfn)); + if (unavailable_pfn_is_online(pfn, &last_subsection, &is_online)) + online_pgcnt++; pgcnt++; } =20 if (pgcnt) pr_info("On node %d, zone %s: %lld pages in unavailable ranges\n", node, zone_names[zone], pgcnt); + return online_pgcnt; } =20 /* @@ -934,9 +961,21 @@ static void __init memmap_init_zone_range(struct zone = *zone, =20 memmap_init_range(end_pfn - start_pfn, nid, zone_id, start_pfn, zone_end_pfn, MEMINIT_EARLY, NULL, mt, false); + zone->pages_with_online_memmap +=3D end_pfn - start_pfn; + + if (*hole_pfn < start_pfn) { + unsigned long hole_start_pfn =3D *hole_pfn; + unsigned long pgcnt; =20 - if (*hole_pfn < start_pfn) - init_unavailable_range(*hole_pfn, start_pfn, zone_id, nid); + if (hole_start_pfn < zone_start_pfn) { + init_unavailable_range(hole_start_pfn, zone_start_pfn, + zone_id, nid); + hole_start_pfn =3D zone_start_pfn; + } + pgcnt =3D init_unavailable_range(hole_start_pfn, start_pfn, + zone_id, nid); + zone->pages_with_online_memmap +=3D pgcnt; + } =20 *hole_pfn =3D end_pfn; } @@ -2195,28 +2234,6 @@ void __init init_cma_pageblock(struct page *page) } #endif =20 -void set_zone_contiguous(struct zone *zone) -{ - unsigned long block_start_pfn =3D zone->zone_start_pfn; - unsigned long block_end_pfn; - - block_end_pfn =3D pageblock_end_pfn(block_start_pfn); - for (; block_start_pfn < zone_end_pfn(zone); - block_start_pfn =3D block_end_pfn, - block_end_pfn +=3D pageblock_nr_pages) { - - block_end_pfn =3D min(block_end_pfn, zone_end_pfn(zone)); - - if (!__pageblock_pfn_to_page(block_start_pfn, - block_end_pfn, zone)) - return; - cond_resched(); - } - - /* We confirm that there is no hole */ - zone->contiguous =3D true; -} - /* * Check if a PFN range intersects multiple zones on one or more * NUMA nodes. Specify the @nid argument if it is known that this --=20 2.47.3