[PATCH v4] mm/migrate_device: consolidate compound folio handling

Hui Su posted 1 patch 7 hours ago
mm/migrate_device.c | 92 +++++++++++++++++++++++----------------------
1 file changed, 48 insertions(+), 44 deletions(-)
[PATCH v4] mm/migrate_device: consolidate compound folio handling
Posted by Hui Su 7 hours ago
Commit dc41e961a269 ("mm/migrate_device: avoid out-of-bounds writes for
compound folios") added handling for compound folios that do not fit in
the remaining PFN array.

migrate_device_range() and migrate_device_pfns() duplicate the logic for
locking device PFNs, encoding compound folios, and handling this boundary
condition.

A compound folio cannot be represented partially for migration. Use
WARN_ON_ONCE() when one does not fit in the remaining PFN array, while
retaining the defensive handling: release the current folio's lock and
reference, clear the remaining entries, and stop collecting.

Move the shared collection and encoding logic into a helper so both
interfaces handle compound folios consistently. Keep the reference,
lock, size lookup, and PFN encoding together so cleanup acts on the folio
whose reference and lock were acquired. Also use memset() for compound
folio tail entries instead of open-coding the clearing loop.

If a PFN cannot be referenced or its folio cannot be locked, leave that
source entry clear and continue at page granularity. Compound tail pages
have a zero page->_refcount, so folio_get_nontail_page() cannot acquire
them. They are therefore skipped one slot at a time. This preserves
collection of later allocated pages in a device chunk containing free
PFNs.

Document that an encountered compound folio must fit entirely in the
remaining range or PFN array.

Link: https://lore.kernel.org/r/c99ca53a-73ef-4a0c-8738-eba1cc89bea2@kernel.org
Link: https://lore.kernel.org/r/60634ab5-3fcc-42fb-8722-4ba6771acfda@kernel.org
Suggested-by: David Hildenbrand <david@kernel.org>
Signed-off-by: Hui Su <sh_def@163.com>
---
Tested on x86_64 with KASAN:
- 7/7 private-device HMM THP cases passed on the exact v4 tree.
- Local KUnit validation covered success, get/trylock failure, and
  truncated compound-folio paths.

Changes in v4:
- Fold migrate_device_pfn_lock() into migrate_device_collect_folio() so the
  folio whose reference and lock are acquired is also used for size and
  cleanup decisions.
- When a PFN cannot be referenced or its folio cannot be locked, leave only
  the current entry clear and continue at page granularity. Compound tails
  have a zero page->_refcount, so folio_get_nontail_page() cannot acquire them.
- Use WARN_ON_ONCE() with defensive cleanup when a compound folio does not
  fit in the remaining array, as suggested by David Hildenbrand.
- Document that a compound folio occupies consecutive page-granular entries
  and must fit entirely in the remaining PFN array.

Changes in v3:
- Keep the source entry zero when migrate_device_pfn_lock() fails instead
  of setting MIGRATE_PFN_COMPOUND, while still consuming and clearing all
  slots belonging to the compound folio.
- Use VM_WARN_ON_ONCE() for the truncated compound-folio invariant while
  keeping defensive cleanup independent of CONFIG_DEBUG_VM, as suggested by
  Balbir Singh.

Changes in v2:
- Fix the helper parameter indentation and keep the declaration to two
  lines, as suggested by David Hildenbrand.

 mm/migrate_device.c | 92 +++++++++++++++++++++++----------------------
 1 file changed, 48 insertions(+), 44 deletions(-)

diff --git a/mm/migrate_device.c b/mm/migrate_device.c
index 009bfa8b212d..34d687882d85 100644
--- a/mm/migrate_device.c
+++ b/mm/migrate_device.c
@@ -1376,20 +1376,47 @@ void migrate_vma_finalize(struct migrate_vma *migrate)
 }
 EXPORT_SYMBOL(migrate_vma_finalize);
 
-static unsigned long migrate_device_pfn_lock(unsigned long pfn)
+/*
+ * Collect a device folio into the page-granular PFN array.
+ *
+ * Return the number of entries consumed. Return 1 with a clear source entry
+ * if the current PFN cannot be referenced or locked. Return 0 if a compound
+ * folio does not fit in the remaining array and collection should stop.
+ */
+static unsigned int migrate_device_collect_folio(unsigned long *src_pfn,
+		unsigned long pfn, unsigned long remaining)
 {
 	struct folio *folio;
+	unsigned int nr;
+
+	*src_pfn = 0;
 
 	folio = folio_get_nontail_page(pfn_to_page(pfn));
 	if (!folio)
-		return 0;
+		return 1;
 
 	if (!folio_trylock(folio)) {
 		folio_put(folio);
+		return 1;
+	}
+
+	nr = folio_nr_pages(folio);
+
+	if (WARN_ON_ONCE(nr > remaining)) {
+		folio_unlock(folio);
+		folio_put(folio);
+		memset(src_pfn, 0, remaining * sizeof(*src_pfn));
 		return 0;
 	}
 
-	return migrate_pfn(pfn) | MIGRATE_PFN_MIGRATE;
+	*src_pfn = migrate_pfn(pfn) | MIGRATE_PFN_MIGRATE;
+
+	if (nr > 1) {
+		*src_pfn |= MIGRATE_PFN_COMPOUND;
+		memset(src_pfn + 1, 0, (nr - 1) * sizeof(*src_pfn));
+	}
+
+	return nr;
 }
 
 /**
@@ -1410,35 +1437,22 @@ static unsigned long migrate_device_pfn_lock(unsigned long pfn)
  * migrating pages that aren't free before unmapping them. Drivers may then
  * allocate destination pages and start copying data from the device to CPU
  * memory before calling migrate_device_pages().
+ *
+ * A compound folio must fit entirely in the remaining range.
  */
 int migrate_device_range(unsigned long *src_pfns, unsigned long start,
 			unsigned long npages)
 {
-	unsigned long i, j, pfn;
+	unsigned long i, pfn;
 
 	for (pfn = start, i = 0; i < npages; pfn++, i++) {
-		struct page *page = pfn_to_page(pfn);
-		struct folio *folio = page_folio(page);
-		unsigned int nr = 1;
+		unsigned int nr;
 
-		src_pfns[i] = migrate_device_pfn_lock(pfn);
-		nr = folio_nr_pages(folio);
-		if (nr > npages - i) {
-			if (src_pfns[i] & MIGRATE_PFN_MIGRATE) {
-				folio_unlock(folio);
-				folio_put(folio);
-			}
-			memset(&src_pfns[i], 0,
-			       (npages - i) * sizeof(*src_pfns));
+		nr = migrate_device_collect_folio(&src_pfns[i], pfn, npages - i);
+		if (!nr)
 			break;
-		}
-		if (nr > 1) {
-			src_pfns[i] |= MIGRATE_PFN_COMPOUND;
-			for (j = 1; j < nr; j++)
-				src_pfns[i+j] = 0;
-			i += j - 1;
-			pfn += j - 1;
-		}
+		i += nr - 1;
+		pfn += nr - 1;
 	}
 
 	migrate_device_unmap(src_pfns, npages, NULL);
@@ -1454,33 +1468,23 @@ EXPORT_SYMBOL(migrate_device_range);
  *
  * Similar to migrate_device_range() but supports non-contiguous pre-populated
  * array of device pages to migrate.
+ *
+ * Entries for different folios may be non-contiguous, but a compound folio
+ * must occupy consecutive page-granular entries and fit entirely in the
+ * remaining PFN array.
  */
 int migrate_device_pfns(unsigned long *src_pfns, unsigned long npages)
 {
-	unsigned long i, j;
+	unsigned long i;
 
 	for (i = 0; i < npages; i++) {
-		struct page *page = pfn_to_page(src_pfns[i]);
-		struct folio *folio = page_folio(page);
-		unsigned int nr = 1;
+		unsigned long pfn = src_pfns[i];
+		unsigned int nr;
 
-		src_pfns[i] = migrate_device_pfn_lock(src_pfns[i]);
-		nr = folio_nr_pages(folio);
-		if (nr > npages - i) {
-			if (src_pfns[i] & MIGRATE_PFN_MIGRATE) {
-				folio_unlock(folio);
-				folio_put(folio);
-			}
-			memset(&src_pfns[i], 0,
-			       (npages - i) * sizeof(*src_pfns));
+		nr = migrate_device_collect_folio(&src_pfns[i], pfn, npages - i);
+		if (!nr)
 			break;
-		}
-		if (nr > 1) {
-			src_pfns[i] |= MIGRATE_PFN_COMPOUND;
-			for (j = 1; j < nr; j++)
-				src_pfns[i+j] = 0;
-			i += j - 1;
-		}
+		i += nr - 1;
 	}
 
 	migrate_device_unmap(src_pfns, npages, NULL);

base-commit: fe2ec83746e501645709761605c2464a44fd2929
-- 
2.55.0