From nobody Fri Oct 2 12:19:50 2026 Received: from shelob.surriel.com (shelob.surriel.com [96.67.55.147]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id D1C603932CE for ; Sat, 1 Aug 2026 03:15:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=96.67.55.147 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554166; cv=none; b=XcYcp2VyMV0XrAZh71JSYjV19lcP6LTndm9TQTNXIsWtP0DHXhAvNifaQjhFY8KRFn/jooeiVU2c93gCdzONvyxLdelOAfaNgSQMnA6hybErrHwdK/IIvkwM6TmxTZKq2ClXOI1VnfY9WIPKqN5hrfTRI5BdAjSenmHO4ceTBZM= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554166; c=relaxed/simple; bh=w/pOhe0J26uMQQ6rRYrgQ5SYLcw8KZX5aCrjcQtYcew=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=i6jWnHXqnaGSiY/clWpF+TAF63jn7mTu6JoYFKBExWBiQEwCUpAcJNeOPue80+hv591F3moFqTxj0V6o01/bU2bqdT9F4oBne+argPnGPprlGehmE64NvBpQP88hOFzJl4kxtaGQjbyet6wWRpQcteWQ09YNKV5svNBYZYn8W/8= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com; spf=pass smtp.mailfrom=surriel.com; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b=IfOz/lDo; arc=none smtp.client-ip=96.67.55.147 Authentication-Results: smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=surriel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b="IfOz/lDo" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=surriel.com ; s=mail; h=Content-Transfer-Encoding:MIME-Version:References:In-Reply-To: Message-ID:Date:Subject:Cc:To:From:Sender:Reply-To:Content-Type:Content-ID: Content-Description:Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc :Resent-Message-ID:List-Id:List-Help:List-Unsubscribe:List-Subscribe: List-Post:List-Owner:List-Archive; bh=bR/vAGfDb9MXggb5yxUq+nu3vsIKwtkqbsmt5GvQiXQ=; b=IfOz/lDosv3Cp2TErnHOwxM3rr MJXIlmAY8t/Wyi0IRlDqhtKjQ2I0qy6G634Rta8Eh5OTWdcdp71qF9UdG1+vPuKx62LdyoKsDfPrX 3goVqp9/kQK4CC6RshBPCLhPt0xaM9J3++I4gYor2C7lBugQsC1JJMiG8wfqJhe2A1HNeRFvEbwP3 t8d51n5D1bPzZJRLeDvZhVjmhqiY44uZGKzZ3C2wb46utjdlcPUVrNj0fJ4J1QqHrvH10+E4A5CZ6 We+U123cuMHElmphHo3uYzNHXDxv0yCQdZ3OhbxkFZE70weIrwRi4aRubpXcYC6WD1BPo2ZMsHbmc Fbtq+45w==; Received: from [96.67.55.146] (helo=fangorn.surriel.com) by shelob.surriel.com with esmtpsa (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384 (Exim 4.97.1) (envelope-from ) id 1wq0CU-000000001Q5-3ezp; Fri, 31 Jul 2026 23:15:50 -0400 From: Rik van Riel To: linux-kernel@vger.kernel.org, Andrew Morton , David Hildenbrand Cc: kernel-team@meta.com, Rik van Riel , Jason Gunthorpe , John Hubbard , Peter Xu , linux-mm@kvack.org, Lorenzo Stoakes Subject: [PATCH 1/5] mm/gup: convert follow_page_mask() to return a long Date: Fri, 31 Jul 2026 23:15:36 -0400 Message-ID: <20260801031540.2742891-2-riel@surriel.com> X-Mailer: git-send-email 2.54.0 In-Reply-To: <20260801031540.2742891-1-riel@surriel.com> References: <20260801031540.2742891-1-riel@surriel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" follow_page_mask() and its helpers return a struct page pointer: NULL, ERR_PTR(), or the page found. Change the return type to long instead: 0, a negative errno, or 1 with the page stored in a new pages[0] slot. This lets the return value carry a page count in a later change, rather than only ever a single struct page pointer. follow_huge_pud() and follow_huge_pmd() now store the found page and flush its caches themselves; __get_user_pages() reads pages[i] back to still expand a large folio's remaining subpages itself. The vsyscall gate area, which bypasses follow_page_mask(), fills its own slot the same way. *page_mask and __get_user_pages()'s handling of a large folio's remaining subpages are untouched, and mm/gup_test.c (PIN_LONGTERM_BENCHMARK) shows no measurable difference for 4 kB, 64 kB mTHP, or 2 MB THP. No functional changes intended. Suggested-by: David Hildenbrand Assisted-by: Claude:claude-opus-4.8 Signed-off-by: Rik van Riel --- mm/gup.c | 269 ++++++++++++++++++++++++++++++------------------------- 1 file changed, 148 insertions(+), 121 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index 0692119b7904..09c64ef2f57c 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -608,15 +608,15 @@ static inline bool can_follow_write_common(struct pag= e *page, return page && PageAnon(page) && PageAnonExclusive(page); } =20 -static struct page *no_page_table(struct vm_area_struct *vma, - unsigned int flags, unsigned long address) +static long no_page_table(struct vm_area_struct *vma, + unsigned int flags, unsigned long address) { if (!(flags & FOLL_DUMP)) - return NULL; + return 0; =20 /* * When core dumping, we don't want to allocate unnecessary pages or - * page tables. Return error instead of NULL to skip handle_mm_fault, + * page tables. Return error instead of 0 to skip handle_mm_fault, * then get_dump_page() will return NULL to leave a hole in the dump. * But we can only make this optimization where a hole would surely * be zero-filled if handle_mm_fault() actually did handle it. @@ -625,12 +625,12 @@ static struct page *no_page_table(struct vm_area_stru= ct *vma, struct hstate *h =3D hstate_vma(vma); =20 if (!hugetlbfs_pagecache_present(h, vma, address)) - return ERR_PTR(-EFAULT); + return -EFAULT; } else if ((vma_is_anonymous(vma) || !vma->vm_ops->fault)) { - return ERR_PTR(-EFAULT); + return -EFAULT; } =20 - return NULL; + return 0; } =20 #ifdef CONFIG_PGTABLE_HAS_HUGE_LEAVES @@ -646,9 +646,10 @@ static inline bool can_follow_write_pud(pud_t pud, str= uct page *page, return can_follow_write_common(page, vma, flags); } =20 -static struct page *follow_huge_pud(struct vm_area_struct *vma, - unsigned long addr, pud_t *pudp, - int flags, unsigned long *page_mask) +static long follow_huge_pud(struct vm_area_struct *vma, + unsigned long addr, pud_t *pudp, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { struct mm_struct *mm =3D vma->vm_mm; struct page *page; @@ -659,25 +660,31 @@ static struct page *follow_huge_pud(struct vm_area_st= ruct *vma, assert_spin_locked(pud_lockptr(mm, pudp)); =20 if (!pud_present(pud)) - return NULL; + return 0; =20 if ((flags & FOLL_WRITE) && !can_follow_write_pud(pud, pfn_to_page(pfn), vma, flags)) - return NULL; + return 0; =20 pfn +=3D (addr & ~PUD_MASK) >> PAGE_SHIFT; page =3D pfn_to_page(pfn); =20 if (!pud_write(pud) && gup_must_unshare(vma, flags, page)) - return ERR_PTR(-EMLINK); + return -EMLINK; =20 ret =3D try_grab_folio(page_folio(page), 1, flags); if (ret) - page =3D ERR_PTR(ret); - else - *page_mask =3D HPAGE_PUD_NR - 1; + return ret; + + *page_mask =3D HPAGE_PUD_NR - 1; + + if (pages) { + pages[0] =3D page; + flush_anon_page(vma, page, addr); + flush_dcache_page(page); + } =20 - return page; + return 1; } =20 /* FOLL_FORCE can write to even unwritable PMDs in COW mappings. */ @@ -698,10 +705,10 @@ static inline bool can_follow_write_pmd(pmd_t pmd, st= ruct page *page, return !userfaultfd_huge_pmd_wp(vma, pmd); } =20 -static struct page *follow_huge_pmd(struct vm_area_struct *vma, - unsigned long addr, pmd_t *pmd, - unsigned int flags, - unsigned long *page_mask) +static long follow_huge_pmd(struct vm_area_struct *vma, + unsigned long addr, pmd_t *pmd, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { struct mm_struct *mm =3D vma->vm_mm; pmd_t pmdval =3D *pmd; @@ -713,24 +720,24 @@ static struct page *follow_huge_pmd(struct vm_area_st= ruct *vma, page =3D pmd_page(pmdval); if ((flags & FOLL_WRITE) && !can_follow_write_pmd(pmdval, page, vma, flags)) - return NULL; + return 0; =20 /* Avoid dumping huge zero page */ if ((flags & FOLL_DUMP) && is_huge_zero_pmd(pmdval)) - return ERR_PTR(-EFAULT); + return -EFAULT; =20 if (pmd_protnone(*pmd) && !gup_can_follow_protnone(vma, flags)) - return NULL; + return 0; =20 if (!pmd_write(pmdval) && gup_must_unshare(vma, flags, page)) - return ERR_PTR(-EMLINK); + return -EMLINK; =20 VM_WARN_ON_ONCE_PAGE((flags & FOLL_PIN) && PageAnon(page) && !PageAnonExclusive(page), page); =20 ret =3D try_grab_folio(page_folio(page), 1, flags); if (ret) - return ERR_PTR(ret); + return ret; =20 #ifdef CONFIG_TRANSPARENT_HUGEPAGE if (pmd_trans_huge(pmdval) && (flags & FOLL_TOUCH)) @@ -740,23 +747,30 @@ static struct page *follow_huge_pmd(struct vm_area_st= ruct *vma, page +=3D (addr & ~HPAGE_PMD_MASK) >> PAGE_SHIFT; *page_mask =3D HPAGE_PMD_NR - 1; =20 - return page; + if (pages) { + pages[0] =3D page; + flush_anon_page(vma, page, addr); + flush_dcache_page(page); + } + + return 1; } =20 #else /* CONFIG_PGTABLE_HAS_HUGE_LEAVES */ -static struct page *follow_huge_pud(struct vm_area_struct *vma, - unsigned long addr, pud_t *pudp, - int flags, unsigned long *page_mask) +static long follow_huge_pud(struct vm_area_struct *vma, + unsigned long addr, pud_t *pudp, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { - return NULL; + return 0; } =20 -static struct page *follow_huge_pmd(struct vm_area_struct *vma, - unsigned long addr, pmd_t *pmd, - unsigned int flags, - unsigned long *page_mask) +static long follow_huge_pmd(struct vm_area_struct *vma, + unsigned long addr, pmd_t *pmd, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { - return NULL; + return 0; } #endif /* CONFIG_PGTABLE_HAS_HUGE_LEAVES */ =20 @@ -799,15 +813,16 @@ static inline bool can_follow_write_pte(pte_t pte, st= ruct page *page, return !userfaultfd_pte_wp(vma, pte); } =20 -static struct page *follow_page_pte(struct vm_area_struct *vma, - unsigned long address, pmd_t *pmd, unsigned int flags) +static long follow_page_pte(struct vm_area_struct *vma, + unsigned long address, pmd_t *pmd, unsigned int flags, + struct page **pages) { struct mm_struct *mm =3D vma->vm_mm; struct folio *folio; struct page *page; spinlock_t *ptl; pte_t *ptep, pte; - int ret; + long ret; =20 ptep =3D pte_offset_map_lock(mm, pmd, address, &ptl); if (!ptep) @@ -825,14 +840,14 @@ static struct page *follow_page_pte(struct vm_area_st= ruct *vma, */ if ((flags & FOLL_WRITE) && !can_follow_write_pte(pte, page, vma, flags)) { - page =3D NULL; + ret =3D 0; goto out; } =20 if (unlikely(!page)) { if (flags & FOLL_DUMP) { /* Avoid special (like zero) pages in core dumps */ - page =3D ERR_PTR(-EFAULT); + ret =3D -EFAULT; goto out; } =20 @@ -840,14 +855,13 @@ static struct page *follow_page_pte(struct vm_area_st= ruct *vma, page =3D pte_page(pte); } else { ret =3D follow_pfn_pte(vma, address, ptep, flags); - page =3D ERR_PTR(ret); goto out; } } folio =3D page_folio(page); =20 if (!pte_write(pte) && gup_must_unshare(vma, flags, page)) { - page =3D ERR_PTR(-EMLINK); + ret =3D -EMLINK; goto out; } =20 @@ -856,10 +870,8 @@ static struct page *follow_page_pte(struct vm_area_str= uct *vma, =20 /* try_grab_folio() does nothing unless FOLL_GET or FOLL_PIN is set. */ ret =3D try_grab_folio(folio, 1, flags); - if (unlikely(ret)) { - page =3D ERR_PTR(ret); + if (unlikely(ret)) goto out; - } =20 /* * We need to make the page accessible if and only if we are going @@ -869,8 +881,7 @@ static struct page *follow_page_pte(struct vm_area_stru= ct *vma, if (flags & FOLL_PIN) { ret =3D arch_make_folio_accessible(folio); if (ret) { - unpin_user_page(page); - page =3D ERR_PTR(ret); + gup_put_folio(folio, 1, flags); goto out; } } @@ -885,24 +896,31 @@ static struct page *follow_page_pte(struct vm_area_st= ruct *vma, */ folio_mark_accessed(folio); } + + if (pages) { + pages[0] =3D page; + flush_anon_page(vma, page, address); + flush_dcache_page(page); + } + ret =3D 1; out: pte_unmap_unlock(ptep, ptl); - return page; + return ret; no_page: pte_unmap_unlock(ptep, ptl); if (!pte_none(pte)) - return NULL; + return 0; return no_page_table(vma, flags, address); } =20 -static struct page *follow_pmd_mask(struct vm_area_struct *vma, - unsigned long address, pud_t *pudp, - unsigned int flags, - unsigned long *page_mask) +static long follow_pmd_mask(struct vm_area_struct *vma, + unsigned long address, pud_t *pudp, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { pmd_t *pmd, pmdval; spinlock_t *ptl; - struct page *page; + long ret; struct mm_struct *mm =3D vma->vm_mm; =20 pmd =3D pmd_offset(pudp, address); @@ -912,7 +930,7 @@ static struct page *follow_pmd_mask(struct vm_area_stru= ct *vma, if (!pmd_present(pmdval)) return no_page_table(vma, flags, address); if (likely(!pmd_leaf(pmdval))) - return follow_page_pte(vma, address, pmd, flags); + return follow_page_pte(vma, address, pmd, flags, pages); =20 if (pmd_protnone(pmdval) && !gup_can_follow_protnone(vma, flags)) return no_page_table(vma, flags, address); @@ -925,28 +943,28 @@ static struct page *follow_pmd_mask(struct vm_area_st= ruct *vma, } if (unlikely(!pmd_leaf(pmdval))) { spin_unlock(ptl); - return follow_page_pte(vma, address, pmd, flags); + return follow_page_pte(vma, address, pmd, flags, pages); } if (pmd_trans_huge(pmdval) && (flags & FOLL_SPLIT_PMD)) { spin_unlock(ptl); split_huge_pmd(vma, pmd, address); /* If pmd was left empty, stuff a page table in there quickly */ - return pte_alloc(mm, pmd) ? ERR_PTR(-ENOMEM) : - follow_page_pte(vma, address, pmd, flags); + return pte_alloc(mm, pmd) ? -ENOMEM : + follow_page_pte(vma, address, pmd, flags, pages); } - page =3D follow_huge_pmd(vma, address, pmd, flags, page_mask); + ret =3D follow_huge_pmd(vma, address, pmd, flags, page_mask, pages); spin_unlock(ptl); - return page; + return ret; } =20 -static struct page *follow_pud_mask(struct vm_area_struct *vma, - unsigned long address, p4d_t *p4dp, - unsigned int flags, - unsigned long *page_mask) +static long follow_pud_mask(struct vm_area_struct *vma, + unsigned long address, p4d_t *p4dp, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { pud_t *pudp, pud; spinlock_t *ptl; - struct page *page; + long ret; struct mm_struct *mm =3D vma->vm_mm; =20 pudp =3D pud_offset(p4dp, address); @@ -955,22 +973,22 @@ static struct page *follow_pud_mask(struct vm_area_st= ruct *vma, return no_page_table(vma, flags, address); if (pud_leaf(pud)) { ptl =3D pud_lock(mm, pudp); - page =3D follow_huge_pud(vma, address, pudp, flags, page_mask); + ret =3D follow_huge_pud(vma, address, pudp, flags, page_mask, pages); spin_unlock(ptl); - if (page) - return page; + if (ret) + return ret; return no_page_table(vma, flags, address); } if (unlikely(pud_bad(pud))) return no_page_table(vma, flags, address); =20 - return follow_pmd_mask(vma, address, pudp, flags, page_mask); + return follow_pmd_mask(vma, address, pudp, flags, page_mask, pages); } =20 -static struct page *follow_p4d_mask(struct vm_area_struct *vma, - unsigned long address, pgd_t *pgdp, - unsigned int flags, - unsigned long *page_mask) +static long follow_p4d_mask(struct vm_area_struct *vma, + unsigned long address, pgd_t *pgdp, + unsigned int flags, unsigned long *page_mask, + struct page **pages) { p4d_t *p4dp, p4d; =20 @@ -981,7 +999,7 @@ static struct page *follow_p4d_mask(struct vm_area_stru= ct *vma, if (!p4d_present(p4d) || p4d_bad(p4d)) return no_page_table(vma, flags, address); =20 - return follow_pud_mask(vma, address, p4dp, flags, page_mask); + return follow_pud_mask(vma, address, p4dp, flags, page_mask, pages); } =20 /** @@ -990,6 +1008,9 @@ static struct page *follow_p4d_mask(struct vm_area_str= uct *vma, * @address: virtual address to look up * @flags: flags modifying lookup behaviour * @page_mask: a pointer to output page_mask + * @pages: array to receive the page found, refcounted per @flags, or NULL + * to walk the page tables (e.g. to fault pages in) without + * collecting or refcounting them * * @flags can have FOLL_ flags set, defined in * @@ -1000,17 +1021,17 @@ static struct page *follow_p4d_mask(struct vm_area_= struct *vma, * * On output, @page_mask is set according to the size of the page. * - * Return: the mapped (struct page *), %NULL if no mapping exists, or - * an error pointer if there is a mapping to something not represented - * by a page descriptor (see also vm_normal_page()). + * Return: 1 with @pages[0] filled in if a page was found, 0 if no mapping + * exists at @address, or a negative errno for a mapping to something not + * represented by a page descriptor (see also vm_normal_page()). */ -static struct page *follow_page_mask(struct vm_area_struct *vma, - unsigned long address, unsigned int flags, - unsigned long *page_mask) +static long follow_page_mask(struct vm_area_struct *vma, + unsigned long address, unsigned int flags, + unsigned long *page_mask, struct page **pages) { pgd_t *pgd; struct mm_struct *mm =3D vma->vm_mm; - struct page *page; + long ret; =20 vma_pgtable_walk_begin(vma); =20 @@ -1018,13 +1039,13 @@ static struct page *follow_page_mask(struct vm_area= _struct *vma, pgd =3D pgd_offset(mm, address); =20 if (pgd_none(*pgd) || unlikely(pgd_bad(*pgd))) - page =3D no_page_table(vma, flags, address); + ret =3D no_page_table(vma, flags, address); else - page =3D follow_p4d_mask(vma, address, pgd, flags, page_mask); + ret =3D follow_p4d_mask(vma, address, pgd, flags, page_mask, pages); =20 vma_pgtable_walk_end(vma); =20 - return page; + return ret; } =20 static int get_gate_page(struct mm_struct *mm, unsigned long address, @@ -1374,6 +1395,7 @@ static long __get_user_pages(struct mm_struct *mm, do { struct page *page; unsigned int page_increm; + long nr; =20 /* first iteration or cross vma bound */ if (!vma || start >=3D vma->vm_end) { @@ -1400,8 +1422,15 @@ static long __get_user_pages(struct mm_struct *mm, pages ? &page : NULL); if (ret) goto out; - page_mask =3D 0; - goto next_page; + if (pages) { + pages[i] =3D page; + flush_anon_page(vma, page, start); + flush_dcache_page(page); + } + i++; + start +=3D PAGE_SIZE; + nr_pages--; + continue; } =20 if (!vma) { @@ -1423,10 +1452,11 @@ static long __get_user_pages(struct mm_struct *mm, } cond_resched(); =20 - page =3D follow_page_mask(vma, start, gup_flags, &page_mask); - if (!page || PTR_ERR(page) =3D=3D -EMLINK) { + nr =3D follow_page_mask(vma, start, gup_flags, &page_mask, + pages ? &pages[i] : NULL); + if (!nr || nr =3D=3D -EMLINK) { ret =3D faultin_page(vma, start, gup_flags, - PTR_ERR(page) =3D=3D -EMLINK, locked); + nr =3D=3D -EMLINK, locked); switch (ret) { case 0: goto retry; @@ -1440,7 +1470,7 @@ static long __get_user_pages(struct mm_struct *mm, goto out; } BUG(); - } else if (PTR_ERR(page) =3D=3D -EEXIST) { + } else if (nr =3D=3D -EEXIST) { /* * Proper page table entry exists, but no corresponding * struct page. If the caller expects **pages to be @@ -1448,53 +1478,50 @@ static long __get_user_pages(struct mm_struct *mm, * for this page. */ if (pages) { - ret =3D PTR_ERR(page); + ret =3D nr; goto out; } - } else if (IS_ERR(page)) { - ret =3D PTR_ERR(page); + } else if (nr < 0) { + ret =3D nr; goto out; } -next_page: + page_increm =3D 1 + (~(start >> PAGE_SHIFT) & page_mask); if (page_increm > nr_pages) page_increm =3D nr_pages; =20 - if (pages) { + /* + * This must be a large folio (and doesn't need to + * be the whole folio; it can be part of it), do + * the refcount work for all the subpages too. + * + * NOTE: here the page may not be the head page + * e.g. when start addr is not thp-size aligned. + * try_grab_folio() should have taken care of tail + * pages. + */ + if (pages && page_increm > 1) { struct page *subpage; unsigned int j; + struct folio *folio =3D page_folio(pages[i]); =20 /* - * This must be a large folio (and doesn't need to - * be the whole folio; it can be part of it), do - * the refcount work for all the subpages too. - * - * NOTE: here the page may not be the head page - * e.g. when start addr is not thp-size aligned. - * try_grab_folio() should have taken care of tail - * pages. + * Since we already hold refcount on the + * large folio, this should never fail. */ - if (page_increm > 1) { - struct folio *folio =3D page_folio(page); - + if (try_grab_folio(folio, page_increm - 1, + gup_flags)) { /* - * Since we already hold refcount on the - * large folio, this should never fail. + * Release the 1st page ref if the + * folio is problematic, fail hard. */ - if (try_grab_folio(folio, page_increm - 1, - gup_flags)) { - /* - * Release the 1st page ref if the - * folio is problematic, fail hard. - */ - gup_put_folio(folio, 1, gup_flags); - ret =3D -EFAULT; - goto out; - } + gup_put_folio(folio, 1, gup_flags); + ret =3D -EFAULT; + goto out; } =20 - for (j =3D 0; j < page_increm; j++) { - subpage =3D page + j; + for (j =3D 1; j < page_increm; j++) { + subpage =3D pages[i] + j; pages[i + j] =3D subpage; flush_anon_page(vma, subpage, start + j * PAGE_SIZE); flush_dcache_page(subpage); --=20 2.53.0-Meta From nobody Fri Oct 2 12:19:50 2026 Received: from shelob.surriel.com (shelob.surriel.com [96.67.55.147]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 8BFD23DDDAF for ; Sat, 1 Aug 2026 03:15:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=96.67.55.147 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554170; cv=none; b=aM4iwSGmWaQjSF2ZPRN+TkF0VVNx27ddUw5fOPslSMzwokhr6QAq/OeGzfJmq2H9A1w61zoolKhiuzjYuVyp4q4X4iYPadNqm60r/rciDZXCDUCYVkO46QAKGFoIwpsWfktEl8mMwlcXoP1Dv1hYfOnQQbr+tnn4Y6qKOKVcpV8= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554170; c=relaxed/simple; bh=E1bfLYtI2RZWnD5lfzUqEGDGV77yVvMK6NCn5AqvZhI=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=criQ2ConCrTwvpa9SGW2CIWA2WZgkfqng3/AUtOh45djliiCA8PkXbrkgcEEpGd8LjMQEbkT3cPUiOY1wtV8TgIJ1UtKnZRjZ61H6RRkH0doe/T9y/QVCvL6PYExO4hvj0vmxjLrNW6Cp95Yq6Wrm+gMjNeEPY60sbEve3tp5rI= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com; spf=pass smtp.mailfrom=surriel.com; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b=bMMULlZD; arc=none smtp.client-ip=96.67.55.147 Authentication-Results: smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=surriel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b="bMMULlZD" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=surriel.com ; s=mail; h=Content-Transfer-Encoding:MIME-Version:References:In-Reply-To: Message-ID:Date:Subject:Cc:To:From:Sender:Reply-To:Content-Type:Content-ID: Content-Description:Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc :Resent-Message-ID:List-Id:List-Help:List-Unsubscribe:List-Subscribe: List-Post:List-Owner:List-Archive; bh=M1pHvAK4N9+3KUvGzv0LO80HoGs7+eXLX4dwzNp1IZM=; b=bMMULlZD2Ev3kM3uCYDeagi7uu SGxDMlBuA+9EC1H10raJQGthPKNwhRU/ByxUaR8mzy4Pct6RgOIcGMUv+2zX9H4D3xBhOCic0mypU pVI1HwOm6NVbJgW3QO36LLC8eaUzW19upwDg4XfYc5XXnsKgI59fjNScFpZGCa33d47YowjBMU3kY +iINXrsIg7lb4r3iUrvP4UB9zNWK51o++i77raAAmcGYPUx/t6ryctUgrrGNJ9OYPIxiKE9W7jONp 6J6UKhulw3MrIoaqs21CpiYFmHYnOmEaVqmjvk1XK0nWBqhJfiRJpZB27pNbbKKkpRul0TOCw5z3K QE07Sn+w==; Received: from [96.67.55.146] (helo=fangorn.surriel.com) by shelob.surriel.com with esmtpsa (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384 (Exim 4.97.1) (envelope-from ) id 1wq0CU-000000001Q5-3m7I; Fri, 31 Jul 2026 23:15:50 -0400 From: Rik van Riel To: linux-kernel@vger.kernel.org, Andrew Morton , David Hildenbrand Cc: kernel-team@meta.com, Rik van Riel , Jason Gunthorpe , John Hubbard , Peter Xu , linux-mm@kvack.org, Lorenzo Stoakes Subject: [PATCH 2/5] mm/gup: split follow_page_pte_commit() out of follow_page_pte() Date: Fri, 31 Jul 2026 23:15:37 -0400 Message-ID: <20260801031540.2742891-3-riel@surriel.com> X-Mailer: git-send-email 2.54.0 In-Reply-To: <20260801031540.2742891-1-riel@surriel.com> References: <20260801031540.2742891-1-riel@surriel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" follow_page_pte() does two things once it has resolved a present PTE to a page: run the per-PTE safety checks (write-fault, unshare), then commit to that page: grab a ref, fault it in if pinning, mark it dirty/accessed, and hand it back to the caller. Split the second part into its own follow_page_pte_commit(), unchanged except for taking its inputs as parameters instead of local variables. A later change teaches it to commit more than one page at a time. No functional changes intended. Suggested-by: David Hildenbrand Assisted-by: Claude:claude-opus-4.8 Signed-off-by: Rik van Riel --- mm/gup.c | 86 ++++++++++++++++++++++++++++++++++---------------------- 1 file changed, 53 insertions(+), 33 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index 09c64ef2f57c..053da43760a2 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -813,6 +813,56 @@ static inline bool can_follow_write_pte(pte_t pte, str= uct page *page, return !userfaultfd_pte_wp(vma, pte); } =20 +/* + * The caller has already run every per-PTE safety check (present, + * write-fault, gup_must_unshare()) on the PTE, so this only does the + * per-folio work: the refcount grab, the FOLL_PIN accessibility fault-in, + * dirty/accessed marking, and the array fill with the cache flush. + */ +static long follow_page_pte_commit(struct vm_area_struct *vma, + unsigned long address, struct folio *folio, struct page *page, + pte_t pte, unsigned int flags, struct page **pages) +{ + long ret; + + /* try_grab_folio() does nothing unless FOLL_GET or FOLL_PIN is set. */ + ret =3D try_grab_folio(folio, 1, flags); + if (unlikely(ret)) + return ret; + + /* + * We need to make the page accessible if and only if we are going + * to access its content (the FOLL_PIN case). Please see + * Documentation/core-api/pin_user_pages.rst for details. + */ + if (flags & FOLL_PIN) { + ret =3D arch_make_folio_accessible(folio); + if (ret) { + gup_put_folio(folio, 1, flags); + return ret; + } + } + if (flags & FOLL_TOUCH) { + if ((flags & FOLL_WRITE) && + !pte_dirty(pte) && !folio_test_dirty(folio)) + folio_mark_dirty(folio); + /* + * pte_mkyoung() would be more correct here, but atomic care + * is needed to avoid losing the dirty bit: it is easier to use + * folio_mark_accessed(). + */ + folio_mark_accessed(folio); + } + + if (pages) { + pages[0] =3D page; + flush_anon_page(vma, page, address); + flush_dcache_page(page); + } + + return 0; +} + static long follow_page_pte(struct vm_area_struct *vma, unsigned long address, pmd_t *pmd, unsigned int flags, struct page **pages) @@ -868,40 +918,10 @@ static long follow_page_pte(struct vm_area_struct *vm= a, VM_WARN_ON_ONCE_PAGE((flags & FOLL_PIN) && PageAnon(page) && !PageAnonExclusive(page), page); =20 - /* try_grab_folio() does nothing unless FOLL_GET or FOLL_PIN is set. */ - ret =3D try_grab_folio(folio, 1, flags); - if (unlikely(ret)) + ret =3D follow_page_pte_commit(vma, address, folio, page, pte, flags, + pages); + if (ret) goto out; - - /* - * We need to make the page accessible if and only if we are going - * to access its content (the FOLL_PIN case). Please see - * Documentation/core-api/pin_user_pages.rst for details. - */ - if (flags & FOLL_PIN) { - ret =3D arch_make_folio_accessible(folio); - if (ret) { - gup_put_folio(folio, 1, flags); - goto out; - } - } - if (flags & FOLL_TOUCH) { - if ((flags & FOLL_WRITE) && - !pte_dirty(pte) && !folio_test_dirty(folio)) - folio_mark_dirty(folio); - /* - * pte_mkyoung() would be more correct here, but atomic care - * is needed to avoid losing the dirty bit: it is easier to use - * folio_mark_accessed(). - */ - folio_mark_accessed(folio); - } - - if (pages) { - pages[0] =3D page; - flush_anon_page(vma, page, address); - flush_dcache_page(page); - } ret =3D 1; out: pte_unmap_unlock(ptep, ptl); --=20 2.53.0-Meta From nobody Fri Oct 2 12:19:50 2026 Received: from shelob.surriel.com (shelob.surriel.com [96.67.55.147]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 2B6313D47C5 for ; Sat, 1 Aug 2026 03:15:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=96.67.55.147 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554166; cv=none; b=YXQ5sXko7PQPKsJt0Pz7VGLly/PLtbntCra2yZcbufLHQ7JG+hzlanUfqsIa4jYnlwVplsVoJFbSMNqO9AJi5hl9EFFAwTOmVFNAt9sKL6jzQkg1xdaeijPEV6Gn4RlTaktA0M0dQL75TJE/PqWZddhq6tXhAevP4d1Ty0fUXbE= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554166; c=relaxed/simple; bh=Ru7dCAuTLRIyhPgJukpe+qV+gOQTLImC+os1Np18NfQ=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=I6BuEEf8nCUEukdUyh3411+YmskhjdeqIpZdqV+zsv1Lz1TMvkjDKw5799Wy2UGivlvgoZmDrmEK9DZ2f2aX4RP+0ekw7C9JQvcLEk1GD8lh/Ytxl0Z9J42QqjrMDB0+KB4rxpxPaFAOQP2JM6nvSU/Q1Lhz6VAx04h1fKA7nQU= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com; spf=pass smtp.mailfrom=surriel.com; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b=oPTdc7GY; arc=none smtp.client-ip=96.67.55.147 Authentication-Results: smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=surriel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b="oPTdc7GY" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=surriel.com ; s=mail; h=Content-Transfer-Encoding:MIME-Version:References:In-Reply-To: Message-ID:Date:Subject:Cc:To:From:Sender:Reply-To:Content-Type:Content-ID: Content-Description:Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc :Resent-Message-ID:List-Id:List-Help:List-Unsubscribe:List-Subscribe: List-Post:List-Owner:List-Archive; bh=Wa9J/+YO6I4dfQddtw/Ia7wDAxt2roUtbkMuKbEd/24=; b=oPTdc7GY6cMkQue14keVS8PtXA XMgVlDpD6bsEpPAG6VEf8wL6uLFYKX3uT+NN7NfkpFJrlur0MchHYWQp/yszM0i7ThzXFFnZKYb2K gFjH+MHq/WheyPnqmiZs3EG9FHyiGV/4x9vk2CTpKt+M+361VSCPsqcaHyL0ESyAF7XR4O2EBvjH6 lGjMyzugkU9m8drjEWEicpM83GwnpbrfJnIIb7hrFcTxS6nYtVEsGV7bSSE6wUHXT+oVOlaOKqyaP Q/tZ0yvqk/OTV/1nZEJfJQPRYqQFw3MXudWgc5tm2v38cf5/tmYOCOeMo4ZYH/1ePa5acd/sM5dhu UGT4dcjA==; Received: from [96.67.55.146] (helo=fangorn.surriel.com) by shelob.surriel.com with esmtpsa (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384 (Exim 4.97.1) (envelope-from ) id 1wq0CU-000000001Q5-3rhq; Fri, 31 Jul 2026 23:15:50 -0400 From: Rik van Riel To: linux-kernel@vger.kernel.org, Andrew Morton , David Hildenbrand Cc: kernel-team@meta.com, Rik van Riel , Jason Gunthorpe , John Hubbard , Peter Xu , linux-mm@kvack.org, Lorenzo Stoakes Subject: [PATCH 3/5] mm/gup: add gup_fill_pages() and use it Date: Fri, 31 Jul 2026 23:15:38 -0400 Message-ID: <20260801031540.2742891-4-riel@surriel.com> X-Mailer: git-send-email 2.54.0 In-Reply-To: <20260801031540.2742891-1-riel@surriel.com> References: <20260801031540.2742891-1-riel@surriel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" follow_huge_pud(), follow_huge_pmd(), and follow_page_pte_commit() each store the page they found into pages[0] and flush its caches with the same three lines. Add a gup_fill_pages() helper for that, and call it from all three instead of repeating the loop. No functional changes intended. Suggested-by: David Hildenbrand Assisted-by: Claude:claude-opus-4.8 Signed-off-by: Rik van Riel --- mm/gup.c | 35 ++++++++++++++++++++--------------- 1 file changed, 20 insertions(+), 15 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index 053da43760a2..b6d508a44ced 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -633,6 +633,23 @@ static long no_page_table(struct vm_area_struct *vma, return 0; } =20 +static void gup_fill_pages(struct vm_area_struct *vma, unsigned long addre= ss, + struct page *page, unsigned long nr, struct page **pages) +{ + unsigned long i; + + if (!pages) + return; + + for (i =3D 0; i < nr; i++) { + struct page *subpage =3D page + i; + + pages[i] =3D subpage; + flush_anon_page(vma, subpage, address + i * PAGE_SIZE); + flush_dcache_page(subpage); + } +} + #ifdef CONFIG_PGTABLE_HAS_HUGE_LEAVES /* FOLL_FORCE can write to even unwritable PUDs in COW mappings. */ static inline bool can_follow_write_pud(pud_t pud, struct page *page, @@ -678,11 +695,7 @@ static long follow_huge_pud(struct vm_area_struct *vma, =20 *page_mask =3D HPAGE_PUD_NR - 1; =20 - if (pages) { - pages[0] =3D page; - flush_anon_page(vma, page, addr); - flush_dcache_page(page); - } + gup_fill_pages(vma, addr, page, 1, pages); =20 return 1; } @@ -747,11 +760,7 @@ static long follow_huge_pmd(struct vm_area_struct *vma, page +=3D (addr & ~HPAGE_PMD_MASK) >> PAGE_SHIFT; *page_mask =3D HPAGE_PMD_NR - 1; =20 - if (pages) { - pages[0] =3D page; - flush_anon_page(vma, page, addr); - flush_dcache_page(page); - } + gup_fill_pages(vma, addr, page, 1, pages); =20 return 1; } @@ -854,11 +863,7 @@ static long follow_page_pte_commit(struct vm_area_stru= ct *vma, folio_mark_accessed(folio); } =20 - if (pages) { - pages[0] =3D page; - flush_anon_page(vma, page, address); - flush_dcache_page(page); - } + gup_fill_pages(vma, address, page, 1, pages); =20 return 0; } --=20 2.53.0-Meta From nobody Fri Oct 2 12:19:50 2026 Received: from shelob.surriel.com (shelob.surriel.com [96.67.55.147]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 970113DB960 for ; Sat, 1 Aug 2026 03:15:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=96.67.55.147 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554165; cv=none; b=AocA1IlB0t9/rtxURTeJsY8sCR90SojpfP3Wqrxjr1IRva8+mOntLT7v10i2z5Z8Kjr2X4dVr+w4oR3bQFjLhNxaFULSWAEeYC6r0FV3DW9tSSKfaMezqez1Tn3CNKj7cnSOwvEDEzNHWhswSUPjj5PTFIClnFmTSrpvFuT1Sn8= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554165; c=relaxed/simple; bh=+CUGNDgOJtSlCrQgyw3ZXXZDNVrG0aoJYwvhdf9SsIc=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=o4Xwsx7qEp+0z5PtCq/CY20E8Edq6J57U0uGZGb0dvEbt9r3n/R1bwkpJGw7foUjWQVzxhnF3yoU0/0FmQnBKkfe886P/nSaZnHTA/48U7YXmxwWt+HNRmSaSGVWqeJBuWxF+8WK8ZWdomFS0WRHq0b7N/WxpGz9E4ZuPbUgUIQ= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com; spf=pass smtp.mailfrom=surriel.com; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b=kLLQRNG3; arc=none smtp.client-ip=96.67.55.147 Authentication-Results: smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=surriel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b="kLLQRNG3" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=surriel.com ; s=mail; h=Content-Transfer-Encoding:MIME-Version:References:In-Reply-To: Message-ID:Date:Subject:Cc:To:From:Sender:Reply-To:Content-Type:Content-ID: Content-Description:Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc :Resent-Message-ID:List-Id:List-Help:List-Unsubscribe:List-Subscribe: List-Post:List-Owner:List-Archive; bh=frJWwvl6Omv6CrI+LKD0Y+Mv87IoxqrmxcgZCvlyOIA=; b=kLLQRNG3v5Z2B5GMFumP0Wm1qu 3b3YxLgXEktuylf8uXqpOgYAppzJ9EZFpSHymnJya/VDqO+XvxjgpkH4pFqKBb8BcsLRk7NgFlrW4 V9iD/cSwDrHDTVV4Y7tAZ9fWvv0+5agVV7tJqmeYdGgWJxuRz3d9v6HizJchxGG/cZvw2JwyBoVgh hrLEjWvyvhdz9uIWUZX6c4CGadQivOUsn5Jhy3qd4d0MiQYHS8NVBIJvCRe05qF0rrb9RtE0YSOdv DaMaFqH6qOCmekRIE9hqxNMxJ5tn5aA88ZPu2EEP1VvC+daC/6PWGGWzpaInqJZTLQlo0X1WAmnKc IuulJlvQ==; Received: from [96.67.55.146] (helo=fangorn.surriel.com) by shelob.surriel.com with esmtpsa (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384 (Exim 4.97.1) (envelope-from ) id 1wq0CU-000000001Q5-3x3V; Fri, 31 Jul 2026 23:15:50 -0400 From: Rik van Riel To: linux-kernel@vger.kernel.org, Andrew Morton , David Hildenbrand Cc: kernel-team@meta.com, Rik van Riel , Jason Gunthorpe , John Hubbard , Peter Xu , linux-mm@kvack.org, Lorenzo Stoakes Subject: [PATCH 4/5] mm/gup: return a huge page's full count from follow_page_mask() Date: Fri, 31 Jul 2026 23:15:39 -0400 Message-ID: <20260801031540.2742891-5-riel@surriel.com> X-Mailer: git-send-email 2.54.0 In-Reply-To: <20260801031540.2742891-1-riel@surriel.com> References: <20260801031540.2742891-1-riel@surriel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" follow_huge_pud()/follow_huge_pmd() already know the huge page's full size but report it via a separate *page_mask output; __get_user_pages() does a second try_grab_folio() call and subpage loop for everything past the first page. Add an @end argument, have the huge paths report their count as the return value instead, clamped to the huge page's size and @end, and fill every subpage via gup_fill_pages(). The huge paths grab refs under the pud/pmd lock, then store the first page in pages[0] and return; follow_pud_mask()/follow_pmd_mask() fill the rest of the array and flush caches after releasing the lock. The pages are already pinned by then, so this is safe. This keeps the lock held for a fixed amount of work regardless of folio size: a 1 GB PUD-mapped folio can be up to HPAGE_PUD_NR pages, too long to flush while blocking every other user of that lock. This merges two try_grab_folio() calls into one, and returns whatever error try_grab_folio() gives instead of forcing -EFAULT on the second call's failure. *page_mask and __get_user_pages()'s second-grab/subpage loop are now dead; remove them. The old silent page_increm clamp becomes a VM_WARN_ON_ONCE, since @end already bounds the count and refs/pages[] are already committed by the time the caller sees it -- truncating here would leak references, not just waste a comparison. follow_page_pte() is unaffected, still returning at most 1 page. __get_user_pages()'s -EEXIST handling also needs an explicit nr =3D 1 for the pages =3D=3D NULL case now: it used to get that for free from page_mask staying 0 for a PFN-special PTE, but nr is -EEXIST there without it, which would grow nr_pages instead of shrinking it. Verified with mm/gup_test.c (PIN_LONGTERM_BENCHMARK): no measurable change for 4 kB, 64 kB mTHP, or 2 MB THP -- these paths don't reach follow_huge_pud()/follow_huge_pmd(), or the merged call is too small to measure. This single-threaded benchmark can't show a lock-hold-time change; the fix above is by inspection, not measurement. Suggested-by: David Hildenbrand Assisted-by: Claude:claude-opus-4.8 Signed-off-by: Rik van Riel --- mm/gup.c | 177 ++++++++++++++++++++++++------------------------------- 1 file changed, 78 insertions(+), 99 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index b6d508a44ced..ccbef9476ff6 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -664,14 +664,14 @@ static inline bool can_follow_write_pud(pud_t pud, st= ruct page *page, } =20 static long follow_huge_pud(struct vm_area_struct *vma, - unsigned long addr, pud_t *pudp, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long addr, unsigned long end, pud_t *pudp, + unsigned int flags, struct page **pages) { struct mm_struct *mm =3D vma->vm_mm; struct page *page; pud_t pud =3D *pudp; unsigned long pfn =3D pud_pfn(pud); + unsigned long off, nr; int ret; =20 assert_spin_locked(pud_lockptr(mm, pudp)); @@ -683,21 +683,23 @@ static long follow_huge_pud(struct vm_area_struct *vm= a, !can_follow_write_pud(pud, pfn_to_page(pfn), vma, flags)) return 0; =20 - pfn +=3D (addr & ~PUD_MASK) >> PAGE_SHIFT; + off =3D PFN_DOWN(addr & ~PUD_MASK); + pfn +=3D off; page =3D pfn_to_page(pfn); =20 if (!pud_write(pud) && gup_must_unshare(vma, flags, page)) return -EMLINK; =20 - ret =3D try_grab_folio(page_folio(page), 1, flags); + nr =3D min(HPAGE_PUD_NR - off, PFN_DOWN(end - addr)); + + ret =3D try_grab_folio(page_folio(page), nr, flags); if (ret) return ret; =20 - *page_mask =3D HPAGE_PUD_NR - 1; - - gup_fill_pages(vma, addr, page, 1, pages); + if (pages) + pages[0] =3D page; =20 - return 1; + return nr; } =20 /* FOLL_FORCE can write to even unwritable PMDs in COW mappings. */ @@ -719,13 +721,13 @@ static inline bool can_follow_write_pmd(pmd_t pmd, st= ruct page *page, } =20 static long follow_huge_pmd(struct vm_area_struct *vma, - unsigned long addr, pmd_t *pmd, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long addr, unsigned long end, pmd_t *pmd, + unsigned int flags, struct page **pages) { struct mm_struct *mm =3D vma->vm_mm; pmd_t pmdval =3D *pmd; struct page *page; + unsigned long off, nr; int ret; =20 assert_spin_locked(pmd_lockptr(mm, pmd)); @@ -748,7 +750,10 @@ static long follow_huge_pmd(struct vm_area_struct *vma, VM_WARN_ON_ONCE_PAGE((flags & FOLL_PIN) && PageAnon(page) && !PageAnonExclusive(page), page); =20 - ret =3D try_grab_folio(page_folio(page), 1, flags); + off =3D PFN_DOWN(addr & ~HPAGE_PMD_MASK); + nr =3D min(HPAGE_PMD_NR - off, PFN_DOWN(end - addr)); + + ret =3D try_grab_folio(page_folio(page), nr, flags); if (ret) return ret; =20 @@ -757,27 +762,25 @@ static long follow_huge_pmd(struct vm_area_struct *vm= a, touch_pmd(vma, addr, pmd, flags & FOLL_WRITE); #endif /* CONFIG_TRANSPARENT_HUGEPAGE */ =20 - page +=3D (addr & ~HPAGE_PMD_MASK) >> PAGE_SHIFT; - *page_mask =3D HPAGE_PMD_NR - 1; + page +=3D off; =20 - gup_fill_pages(vma, addr, page, 1, pages); + if (pages) + pages[0] =3D page; =20 - return 1; + return nr; } =20 #else /* CONFIG_PGTABLE_HAS_HUGE_LEAVES */ static long follow_huge_pud(struct vm_area_struct *vma, - unsigned long addr, pud_t *pudp, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long addr, unsigned long end, pud_t *pudp, + unsigned int flags, struct page **pages) { return 0; } =20 static long follow_huge_pmd(struct vm_area_struct *vma, - unsigned long addr, pmd_t *pmd, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long addr, unsigned long end, pmd_t *pmd, + unsigned int flags, struct page **pages) { return 0; } @@ -939,9 +942,8 @@ static long follow_page_pte(struct vm_area_struct *vma, } =20 static long follow_pmd_mask(struct vm_area_struct *vma, - unsigned long address, pud_t *pudp, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long address, unsigned long end, pud_t *pudp, + unsigned int flags, struct page **pages) { pmd_t *pmd, pmdval; spinlock_t *ptl; @@ -977,15 +979,23 @@ static long follow_pmd_mask(struct vm_area_struct *vm= a, return pte_alloc(mm, pmd) ? -ENOMEM : follow_page_pte(vma, address, pmd, flags, pages); } - ret =3D follow_huge_pmd(vma, address, pmd, flags, page_mask, pages); + ret =3D follow_huge_pmd(vma, address, end, pmd, flags, pages); spin_unlock(ptl); + + /* + * Refs are already grabbed above; the array fill and cache + * flushes only touch the now-pinned pages, so do them without + * the pmd lock held. + */ + if (ret > 0 && pages) + gup_fill_pages(vma, address, pages[0], ret, pages); + return ret; } =20 static long follow_pud_mask(struct vm_area_struct *vma, - unsigned long address, p4d_t *p4dp, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long address, unsigned long end, p4d_t *p4dp, + unsigned int flags, struct page **pages) { pud_t *pudp, pud; spinlock_t *ptl; @@ -998,8 +1008,18 @@ static long follow_pud_mask(struct vm_area_struct *vm= a, return no_page_table(vma, flags, address); if (pud_leaf(pud)) { ptl =3D pud_lock(mm, pudp); - ret =3D follow_huge_pud(vma, address, pudp, flags, page_mask, pages); + ret =3D follow_huge_pud(vma, address, end, pudp, flags, pages); spin_unlock(ptl); + + /* + * Refs are already grabbed above; the array fill and cache + * flushes only touch the now-pinned pages, so do them + * without the pud lock held -- a 1 GB folio can be up to + * HPAGE_PUD_NR pages, too long to flush under a spinlock. + */ + if (ret > 0 && pages) + gup_fill_pages(vma, address, pages[0], ret, pages); + if (ret) return ret; return no_page_table(vma, flags, address); @@ -1007,13 +1027,12 @@ static long follow_pud_mask(struct vm_area_struct *= vma, if (unlikely(pud_bad(pud))) return no_page_table(vma, flags, address); =20 - return follow_pmd_mask(vma, address, pudp, flags, page_mask, pages); + return follow_pmd_mask(vma, address, end, pudp, flags, pages); } =20 static long follow_p4d_mask(struct vm_area_struct *vma, - unsigned long address, pgd_t *pgdp, - unsigned int flags, unsigned long *page_mask, - struct page **pages) + unsigned long address, unsigned long end, pgd_t *pgdp, + unsigned int flags, struct page **pages) { p4d_t *p4dp, p4d; =20 @@ -1024,18 +1043,18 @@ static long follow_p4d_mask(struct vm_area_struct *= vma, if (!p4d_present(p4d) || p4d_bad(p4d)) return no_page_table(vma, flags, address); =20 - return follow_pud_mask(vma, address, p4dp, flags, page_mask, pages); + return follow_pud_mask(vma, address, end, p4dp, flags, pages); } =20 /** - * follow_page_mask - look up a page descriptor from a user-virtual address + * follow_page_mask - look up pages at a user-virtual address * @vma: vm_area_struct mapping @address * @address: virtual address to look up + * @end: virtual address at which to stop batching contiguous pages * @flags: flags modifying lookup behaviour - * @page_mask: a pointer to output page_mask - * @pages: array to receive the page found, refcounted per @flags, or NULL - * to walk the page tables (e.g. to fault pages in) without - * collecting or refcounting them + * @pages: array to receive the pages, refcounted per @flags, or NULL to + * walk the page tables (e.g. to fault pages in) without collecting + * or refcounting them * * @flags can have FOLL_ flags set, defined in * @@ -1044,15 +1063,15 @@ static long follow_p4d_mask(struct vm_area_struct *= vma, * trigger a fault with FAULT_FLAG_UNSHARE set. Note that unsharing is only * relevant with FOLL_PIN and !FOLL_WRITE. * - * On output, @page_mask is set according to the size of the page. - * - * Return: 1 with @pages[0] filled in if a page was found, 0 if no mapping - * exists at @address, or a negative errno for a mapping to something not - * represented by a page descriptor (see also vm_normal_page()). + * Return: the number of contiguous pages starting at @address that were + * placed into @pages (if non-NULL), which may be fewer than the pages + * requested via @end; 0 if no mapping exists at @address; or a negative + * errno for a mapping to something not represented by a page descriptor + * (see also vm_normal_page()). */ static long follow_page_mask(struct vm_area_struct *vma, - unsigned long address, unsigned int flags, - unsigned long *page_mask, struct page **pages) + unsigned long address, unsigned long end, + unsigned int flags, struct page **pages) { pgd_t *pgd; struct mm_struct *mm =3D vma->vm_mm; @@ -1060,13 +1079,12 @@ static long follow_page_mask(struct vm_area_struct = *vma, =20 vma_pgtable_walk_begin(vma); =20 - *page_mask =3D 0; pgd =3D pgd_offset(mm, address); =20 if (pgd_none(*pgd) || unlikely(pgd_bad(*pgd))) ret =3D no_page_table(vma, flags, address); else - ret =3D follow_p4d_mask(vma, address, pgd, flags, page_mask, pages); + ret =3D follow_p4d_mask(vma, address, end, pgd, flags, pages); =20 vma_pgtable_walk_end(vma); =20 @@ -1404,7 +1422,6 @@ static long __get_user_pages(struct mm_struct *mm, { long ret =3D 0, i =3D 0; struct vm_area_struct *vma =3D NULL; - unsigned long page_mask =3D 0; =20 if (!nr_pages) return 0; @@ -1419,7 +1436,6 @@ static long __get_user_pages(struct mm_struct *mm, =20 do { struct page *page; - unsigned int page_increm; long nr; =20 /* first iteration or cross vma bound */ @@ -1477,8 +1493,8 @@ static long __get_user_pages(struct mm_struct *mm, } cond_resched(); =20 - nr =3D follow_page_mask(vma, start, gup_flags, &page_mask, - pages ? &pages[i] : NULL); + nr =3D follow_page_mask(vma, start, start + nr_pages * PAGE_SIZE, + gup_flags, pages ? &pages[i] : NULL); if (!nr || nr =3D=3D -EMLINK) { ret =3D faultin_page(vma, start, gup_flags, nr =3D=3D -EMLINK, locked); @@ -1500,62 +1516,25 @@ static long __get_user_pages(struct mm_struct *mm, * Proper page table entry exists, but no corresponding * struct page. If the caller expects **pages to be * filled in, bail out now, because that can't be done - * for this page. + * for this page. Otherwise advance by the one page + * follow_page_mask() looked at. */ if (pages) { ret =3D nr; goto out; } + nr =3D 1; } else if (nr < 0) { ret =3D nr; goto out; } =20 - page_increm =3D 1 + (~(start >> PAGE_SHIFT) & page_mask); - if (page_increm > nr_pages) - page_increm =3D nr_pages; - - /* - * This must be a large folio (and doesn't need to - * be the whole folio; it can be part of it), do - * the refcount work for all the subpages too. - * - * NOTE: here the page may not be the head page - * e.g. when start addr is not thp-size aligned. - * try_grab_folio() should have taken care of tail - * pages. - */ - if (pages && page_increm > 1) { - struct page *subpage; - unsigned int j; - struct folio *folio =3D page_folio(pages[i]); - - /* - * Since we already hold refcount on the - * large folio, this should never fail. - */ - if (try_grab_folio(folio, page_increm - 1, - gup_flags)) { - /* - * Release the 1st page ref if the - * folio is problematic, fail hard. - */ - gup_put_folio(folio, 1, gup_flags); - ret =3D -EFAULT; - goto out; - } - - for (j =3D 1; j < page_increm; j++) { - subpage =3D pages[i] + j; - pages[i + j] =3D subpage; - flush_anon_page(vma, subpage, start + j * PAGE_SIZE); - flush_dcache_page(subpage); - } - } + /* Check that we didn't pin more pages than the caller will free. */ + VM_WARN_ON_ONCE(nr > nr_pages); =20 - i +=3D page_increm; - start +=3D page_increm * PAGE_SIZE; - nr_pages -=3D page_increm; + i +=3D nr; + start +=3D nr * PAGE_SIZE; + nr_pages -=3D nr; } while (nr_pages); out: return i ? i : ret; --=20 2.53.0-Meta From nobody Fri Oct 2 12:19:50 2026 Received: from shelob.surriel.com (shelob.surriel.com [96.67.55.147]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.subspace.kernel.org (Postfix) with ESMTPS id 7E6DE36E498 for ; Sat, 1 Aug 2026 03:15:58 +0000 (UTC) Authentication-Results: smtp.subspace.kernel.org; arc=none smtp.client-ip=96.67.55.147 ARC-Seal: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554168; cv=none; b=B3APyy7xspyGzDG9caf3P9YtyuqUgikDgZAPce8/lWjbO/eRnd/YueTEmEQKhrScbQkYfvIr1yIyRmG7F80p6UYpqkDAv7EEGHXPAPDlOW7swpb6+aBPmNCC6sYp97RFJqztFlvIs1pXSnUVhcVShHuI/bcfl6DrBVw1RsuFRe0= ARC-Message-Signature: i=1; a=rsa-sha256; d=subspace.kernel.org; s=arc-20240116; t=1785554168; c=relaxed/simple; bh=XaJog3hQ1l148QFtycSshx6Eg1rh2OhIah+Am+hcQbU=; h=From:To:Cc:Subject:Date:Message-ID:In-Reply-To:References: MIME-Version; b=TYlW5YL34xNXdAXwiF3RMY0pH37ZRVqMGWxHmEeFEo5llXr65998pozSbfTK1hBSW0YWhqhnUN2CflQROCoV1cgAM00CLTOZU3tzqjToCpose/fct1QV9rTmNLdvbGITiP35VuHWXIXE0RNRsK/wQr3oX3KdHk86xtAIHVXrnjI= ARC-Authentication-Results: i=1; smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com; spf=pass smtp.mailfrom=surriel.com; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b=HZgRLPVW; arc=none smtp.client-ip=96.67.55.147 Authentication-Results: smtp.subspace.kernel.org; dmarc=none (p=none dis=none) header.from=surriel.com Authentication-Results: smtp.subspace.kernel.org; spf=pass smtp.mailfrom=surriel.com Authentication-Results: smtp.subspace.kernel.org; dkim=pass (2048-bit key) header.d=surriel.com header.i=@surriel.com header.b="HZgRLPVW" DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=surriel.com ; s=mail; h=Content-Transfer-Encoding:MIME-Version:References:In-Reply-To: Message-ID:Date:Subject:Cc:To:From:Sender:Reply-To:Content-Type:Content-ID: Content-Description:Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc :Resent-Message-ID:List-Id:List-Help:List-Unsubscribe:List-Subscribe: List-Post:List-Owner:List-Archive; bh=ayUmkBGSzdWGbdkSooD5p8ZAVjeLT0Ckq7fwnbKZLFE=; b=HZgRLPVW8/Xv7+jUG3Qa3KUd/P 160h1qW84hR7wqom/V4sMQvCeaq6wn0S6JHgYFl8MwvE7GrksU504ebaqYaJBk5bTcSlXk/eIVSMr ofpO5qVP6gFek/Vb1kflwtYNLegf59qda/P893dblvXE3jReBDBDfaBI8wAifRYs23WvzUBO5cz+g GWp/xBubzqS17+MGWS4pVMqi5ezT+57SqsaXjwTaYTF2rlgPaFuw7641YlQUm6OhAlCcNH99g461r Qd4pBJXoj5KXgSrmbI6XRexvNKjPoRSQiUkEtAg5yhd1SNf2OkNqj6eYVNX+Oq6OmdtiafnKwrQ/u tANJ8wyA==; Received: from [96.67.55.146] (helo=fangorn.surriel.com) by shelob.surriel.com with esmtpsa (TLS1.2) tls TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384 (Exim 4.97.1) (envelope-from ) id 1wq0CU-000000001Q5-43PR; Fri, 31 Jul 2026 23:15:50 -0400 From: Rik van Riel To: linux-kernel@vger.kernel.org, Andrew Morton , David Hildenbrand Cc: kernel-team@meta.com, Rik van Riel , Jason Gunthorpe , John Hubbard , Peter Xu , linux-mm@kvack.org, Lorenzo Stoakes Subject: [PATCH 5/5] mm/gup: walk multiple PTEs per follow_page_pte() call Date: Fri, 31 Jul 2026 23:15:40 -0400 Message-ID: <20260801031540.2742891-6-riel@surriel.com> X-Mailer: git-send-email 2.54.0 In-Reply-To: <20260801031540.2742891-1-riel@surriel.com> References: <20260801031540.2742891-1-riel@surriel.com> Precedence: bulk X-Mailing-List: linux-kernel@vger.kernel.org List-Id: List-Subscribe: List-Unsubscribe: MIME-Version: 1.0 Content-Transfer-Encoding: quoted-printable Content-Type: text/plain; charset="utf-8" follow_page_pte() still looks at one PTE per call, so __get_user_pages() calls it once per page even for a PTE-mapped large folio (mTHP), restarting the pgd/p4d/pud/pmd descent and retaking the PTE lock each time. Walk every PTE from @address to the page-table/VMA/@end boundary in one call instead, without stopping at folio boundaries -- adjacent pages from different folios, or plain base pages with no folio relationship at all, are covered by the same call under one lock. Within that walk, contiguous same-folio pages still get one combined refcount grab: each run starts with a full per-PTE resolve, then follow_pte_batch() finds how many more PTEs extend it via a cheap comparison scan, not a re-derivation of each page. Measured with mm/gup_test.c (PIN_LONGTERM_BENCHMARK) on a 256 MB MADV_HUGEPAGE region in a 4 CPU VM, median get time over 16 iterations, before/after back to back in the same VM. Each folio size was confirmed through the per-size anon_fault_alloc counters (4096 folios for 64 kB, 128 for 2 MB): gup_test -L -m 256 -n 65536 -r 16 -t before after 64 kB mTHP 3000 us 207 us (14.5x) 2 MB THP (control) 70 us 69 us 4 kB base (control) 2801 us 1188 us (2.4x) The 4 kB case shares no folio, so gets no refcount-batching benefit -- yet it still improves 2.4x purely from walking the page table once instead of restarting per page. The 64 kB run gets that saving plus refcount batching on top, reaching 14.5x. The 2 MB THP case doesn't reach follow_page_pte() at all, so it stays flat. Suggested-by: David Hildenbrand Assisted-by: Claude:claude-opus-4.8 Signed-off-by: Rik van Riel --- mm/gup.c | 209 ++++++++++++++++++++++++++++++++++++++----------------- 1 file changed, 145 insertions(+), 64 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index ccbef9476ff6..55bbeaa52b13 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -827,18 +827,20 @@ static inline bool can_follow_write_pte(pte_t pte, st= ruct page *page, =20 /* * The caller has already run every per-PTE safety check (present, - * write-fault, gup_must_unshare()) on the PTE, so this only does the - * per-folio work: the refcount grab, the FOLL_PIN accessibility fault-in, - * dirty/accessed marking, and the array fill with the cache flush. + * write-fault, gup_must_unshare()) on each PTE in the run, so this only + * does the per-folio work: the refcount grab, the FOLL_PIN accessibility + * fault-in, dirty/accessed marking, and the array fill with per-subpage + * cache flushes. */ static long follow_page_pte_commit(struct vm_area_struct *vma, - unsigned long address, struct folio *folio, struct page *page, - pte_t pte, unsigned int flags, struct page **pages) + unsigned long run_address, struct folio *folio, + struct page *run_page, pte_t run_pte, unsigned long run_len, + unsigned int flags, struct page **pages) { long ret; =20 /* try_grab_folio() does nothing unless FOLL_GET or FOLL_PIN is set. */ - ret =3D try_grab_folio(folio, 1, flags); + ret =3D try_grab_folio(folio, run_len, flags); if (unlikely(ret)) return ret; =20 @@ -850,13 +852,13 @@ static long follow_page_pte_commit(struct vm_area_str= uct *vma, if (flags & FOLL_PIN) { ret =3D arch_make_folio_accessible(folio); if (ret) { - gup_put_folio(folio, 1, flags); + gup_put_folio(folio, run_len, flags); return ret; } } if (flags & FOLL_TOUCH) { if ((flags & FOLL_WRITE) && - !pte_dirty(pte) && !folio_test_dirty(folio)) + !pte_dirty(run_pte) && !folio_test_dirty(folio)) folio_mark_dirty(folio); /* * pte_mkyoung() would be more correct here, but atomic care @@ -866,79 +868,158 @@ static long follow_page_pte_commit(struct vm_area_st= ruct *vma, folio_mark_accessed(folio); } =20 - gup_fill_pages(vma, address, page, 1, pages); + gup_fill_pages(vma, run_address, run_page, run_len, pages); =20 return 0; } =20 +/* + * Return how many PTEs from @ptep can batch with @pte's page: + * consecutive present, uniform-write PTEs of @folio, bounded by + * @walk_end. Returns at least 1. A cheap pte_same() scan, so a + * large folio's run costs one scan. + * + * gup_must_unshare()/write-fault checks are per PTE, but a writable + * run is always safe: a writable anon page is exclusive. A read-only + * run under FOLL_WRITE/FOLL_PIN needs a per-page check instead, so + * it falls back to one page at a time. + */ +static unsigned long follow_pte_batch(struct vm_area_struct *vma, + unsigned long address, unsigned long walk_end, struct folio *folio, + pte_t *ptep, pte_t pte, unsigned int flags) +{ + pte_t batch_pte =3D pte; + unsigned long max; + + if (!folio_test_large(folio)) + return 1; + if (!pte_write(pte) && (flags & (FOLL_WRITE | FOLL_PIN))) + return 1; + + max =3D (walk_end - address) >> PAGE_SHIFT; + if (max <=3D 1) + return 1; + + return folio_pte_batch_flags(folio, vma, ptep, &batch_pte, max, + FPB_RESPECT_WRITE); +} + +/* + * Walk every PTE from @address to @end (this page table and VMA). + * + * If the first PTE can't be included (not present, a write/unshare + * fault, PFN-special, ...), that reason is returned directly. + * A failure in a subsequent page results in a short read; + * __get_user_pages retrying the read will get the error. + */ static long follow_page_pte(struct vm_area_struct *vma, - unsigned long address, pmd_t *pmd, unsigned int flags, - struct page **pages) + unsigned long address, unsigned long end, pmd_t *pmd, + unsigned int flags, struct page **pages) { struct mm_struct *mm =3D vma->vm_mm; - struct folio *folio; - struct page *page; spinlock_t *ptl; - pte_t *ptep, pte; - long ret; + pte_t *ptep, *orig_ptep; + unsigned long walk_end; + unsigned long nr =3D 0; + bool need_no_page_table =3D false; + long ret =3D 0; =20 - ptep =3D pte_offset_map_lock(mm, pmd, address, &ptl); + orig_ptep =3D ptep =3D pte_offset_map_lock(mm, pmd, address, &ptl); if (!ptep) return no_page_table(vma, flags, address); - pte =3D ptep_get(ptep); - if (!pte_present(pte)) - goto no_page; - if (pte_protnone(pte) && !gup_can_follow_protnone(vma, flags)) - goto no_page; =20 - page =3D vm_normal_page(vma, address, pte); + walk_end =3D min(pmd_addr_end(address, end), vma->vm_end); =20 - /* - * We only care about anon pages in can_follow_write_pte(). - */ - if ((flags & FOLL_WRITE) && - !can_follow_write_pte(pte, page, vma, flags)) { - ret =3D 0; - goto out; - } + for (; address < walk_end; address +=3D PAGE_SIZE, ptep++) { + pte_t pte =3D ptep_get(ptep); + struct page *page; + struct folio *folio; + unsigned long batch; =20 - if (unlikely(!page)) { - if (flags & FOLL_DUMP) { - /* Avoid special (like zero) pages in core dumps */ - ret =3D -EFAULT; - goto out; + if (!pte_present(pte)) { + if (nr) + break; + if (pte_none(pte)) + need_no_page_table =3D true; + goto unlock; + } + if (pte_protnone(pte) && !gup_can_follow_protnone(vma, flags)) { + if (nr) + break; + goto unlock; } =20 - if (is_zero_pfn(pte_pfn(pte))) { - page =3D pte_page(pte); - } else { - ret =3D follow_pfn_pte(vma, address, ptep, flags); - goto out; + page =3D vm_normal_page(vma, address, pte); + + /* + * We only care about anon pages in can_follow_write_pte(). + */ + if ((flags & FOLL_WRITE) && + !can_follow_write_pte(pte, page, vma, flags)) { + if (nr) + break; + goto unlock; } - } - folio =3D page_folio(page); =20 - if (!pte_write(pte) && gup_must_unshare(vma, flags, page)) { - ret =3D -EMLINK; - goto out; - } + if (unlikely(!page)) { + if (flags & FOLL_DUMP) { + /* Avoid special (like zero) pages in core dumps */ + if (nr) + break; + ret =3D -EFAULT; + goto unlock; + } =20 - VM_WARN_ON_ONCE_PAGE((flags & FOLL_PIN) && PageAnon(page) && - !PageAnonExclusive(page), page); + if (is_zero_pfn(pte_pfn(pte))) { + page =3D pte_page(pte); + } else { + /* + * Proper page table entry exists, but no + * corresponding struct page: the caller decides + * whether that is fatal (see the -EEXIST handling + * in __get_user_pages()). + */ + if (nr) + break; + ret =3D follow_pfn_pte(vma, address, ptep, flags); + goto unlock; + } + } + folio =3D page_folio(page); =20 - ret =3D follow_page_pte_commit(vma, address, folio, page, pte, flags, - pages); - if (ret) - goto out; - ret =3D 1; -out: - pte_unmap_unlock(ptep, ptl); - return ret; -no_page: - pte_unmap_unlock(ptep, ptl); - if (!pte_none(pte)) - return 0; - return no_page_table(vma, flags, address); + if (!pte_write(pte) && gup_must_unshare(vma, flags, page)) { + if (nr) + break; + ret =3D -EMLINK; + goto unlock; + } + + VM_WARN_ON_ONCE_PAGE((flags & FOLL_PIN) && PageAnon(page) && + !PageAnonExclusive(page), page); + + batch =3D follow_pte_batch(vma, address, walk_end, folio, ptep, pte, + flags); + + ret =3D follow_page_pte_commit(vma, address, folio, page, pte, + batch, flags, + pages ? pages + nr : NULL); + if (ret) { + if (nr) + break; + goto unlock; + } + nr +=3D batch; + + /* The loop's own increment covers one PTE; skip the rest of the batch. = */ + ptep +=3D batch - 1; + address +=3D (batch - 1) * PAGE_SIZE; + } + +unlock: + pte_unmap_unlock(orig_ptep, ptl); + if (need_no_page_table) + ret =3D no_page_table(vma, flags, address); + return nr ? (long)nr : ret; } =20 static long follow_pmd_mask(struct vm_area_struct *vma, @@ -957,7 +1038,7 @@ static long follow_pmd_mask(struct vm_area_struct *vma, if (!pmd_present(pmdval)) return no_page_table(vma, flags, address); if (likely(!pmd_leaf(pmdval))) - return follow_page_pte(vma, address, pmd, flags, pages); + return follow_page_pte(vma, address, end, pmd, flags, pages); =20 if (pmd_protnone(pmdval) && !gup_can_follow_protnone(vma, flags)) return no_page_table(vma, flags, address); @@ -970,14 +1051,14 @@ static long follow_pmd_mask(struct vm_area_struct *v= ma, } if (unlikely(!pmd_leaf(pmdval))) { spin_unlock(ptl); - return follow_page_pte(vma, address, pmd, flags, pages); + return follow_page_pte(vma, address, end, pmd, flags, pages); } if (pmd_trans_huge(pmdval) && (flags & FOLL_SPLIT_PMD)) { spin_unlock(ptl); split_huge_pmd(vma, pmd, address); /* If pmd was left empty, stuff a page table in there quickly */ return pte_alloc(mm, pmd) ? -ENOMEM : - follow_page_pte(vma, address, pmd, flags, pages); + follow_page_pte(vma, address, end, pmd, flags, pages); } ret =3D follow_huge_pmd(vma, address, end, pmd, flags, pages); spin_unlock(ptl); --=20 2.53.0-Meta