include/linux/vmalloc.h | 2 +- mm/vmalloc.c | 58 ++++++++++++++++++++--------------------- 2 files changed, 29 insertions(+), 31 deletions(-)
vm_struct::nr_pages is an unsigned int, and the file keeps deriving byte
counts from it as nr_pages << PAGE_SHIFT. A shift is evaluated in the type
of its promoted left operand, so those are 32-bit arithmetic and wrap at
4 GiB of bytes, which is 2^20 pages. Every site depends on a cast being
remembered; vmap() has one, two recent commits did not. vread_iter() then
computes a size of zero for a 4 GiB VM_ALLOC area and /proc/kcore returns
it as zeros while reporting a successful read, which drgn, crash or gdb
cannot tell from real memory, and the vrealloc() grow-in-place check
declines a request that would have fit.
Widen the field so the class of bug goes away instead of one site at a
time. Everything feeding or consuming it widens too: vm_area_alloc_pages()
and its accumulators, nr_small_pages, new_nr_pages and old_nr_pages, the
index range of vm_area_free_pages(), and three page indexes that were
plain int. Five casts go. Two prints needed fixing as well, %u in
vmalloc_dump_obj() and %d for the unsigned field in vmalloc_info_show().
No bug report behind this, I found it reading the code. The 4 GiB wrap
needs only a machine with over 4 GiB of memory. Neither larger threshold
is a practical concern: 2^32 pages, where the field itself truncates, is
16 TiB and beyond what hardware can populate, and 2^31, where the plain
int indexes break, is 8 TiB and larger than anything in the tree asks for.
The int *nr cursor in the mapping path is unchanged and is separate work.
Users outside mm/vmalloc.c need no change either. Those handing the count
to a narrower parameter cannot drive it near 2^31, and
kho_preserve_vmalloc() stores it into a 32-bit ABI field that still
receives the same low bits; above 2^32 pages the truncation just moves out
of vm_struct into that store.
sizeof(struct vm_struct) on x86-64 stays 72 bytes with
CONFIG_HAVE_ARCH_HUGE_VMALLOC=n and goes from 72 to 80 with it enabled,
both inside the kmalloc-96 bucket it already comes from.
Cc: stable@vger.kernel.org
Fixes: 0bca23804632 ("mm/vmalloc: use physical page count in vread_iter() for VM_ALLOC areas")
Fixes: d57ac904ffdc ("mm/vmalloc: use physical page count for vrealloc() grow-in-place check")
Suggested-by: Andrew Morton <akpm@linux-foundation.org>
Reviewed-by: Uladzislau Rezki (Sony) <urezki@gmail.com>
Assisted-by: Claude:claude-fable-5
Signed-off-by: Artem Lytkin <iprintercanon@gmail.com>
---
This is the single switch-to-ulong patch you asked for; v3 crossed with
your mail a few hours earlier and was already that, only without the
stable tag. So v4 is v3 plus Cc: stable, rebased on today's mm-new.
I left the second Fixes: on d57ac904ffdc as well, since the same widening
is what fixes the vrealloc() grow-in-place check. Drop it if you would
rather the backport hang off one commit.
v4:
- Cc: stable (Andrew)
- corrected the kexec_handover sentence, which claimed a cap that is not
there; the ABI field just keeps taking the low 32 bits
- rebased on mm-new
v3: https://lore.kernel.org/linux-mm/20260730171142.76817-1-iprintercanon@gmail.com/
v2: https://lore.kernel.org/linux-mm/20260730090628.65814-1-iprintercanon@gmail.com/
include/linux/vmalloc.h | 2 +-
mm/vmalloc.c | 58 ++++++++++++++++++++---------------------
2 files changed, 29 insertions(+), 31 deletions(-)
diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h
index e4d8d0a9f30f9..aed121d729b01 100644
--- a/include/linux/vmalloc.h
+++ b/include/linux/vmalloc.h
@@ -62,7 +62,7 @@ struct vm_struct {
#ifdef CONFIG_HAVE_ARCH_HUGE_VMALLOC
unsigned int page_order;
#endif
- unsigned int nr_pages;
+ unsigned long nr_pages;
phys_addr_t phys_addr;
const void *caller;
unsigned long requested_size;
diff --git a/mm/vmalloc.c b/mm/vmalloc.c
index 26f32949c2f2e..196da8738cf10 100644
--- a/mm/vmalloc.c
+++ b/mm/vmalloc.c
@@ -3404,7 +3404,7 @@ struct vm_struct *remove_vm_area(const void *addr)
static inline void set_area_direct_map(const struct vm_struct *area,
int (*set_direct_map)(struct page *page))
{
- int i;
+ unsigned long i;
/* HUGE_VMALLOC passes small pages to set_direct_map */
for (i = 0; i < area->nr_pages; i++)
@@ -3420,7 +3420,7 @@ static void vm_reset_perms(struct vm_struct *area)
unsigned long start = ULONG_MAX, end = 0;
unsigned int page_order = vm_area_page_order(area);
int flush_dmap = 0;
- int i;
+ unsigned long i;
/*
* Find the start and end range of the direct mappings to make sure that
@@ -3493,10 +3493,10 @@ void vfree_atomic(const void *addr)
* Caller is responsible for unmapping (vunmap_range) and KASAN
* poisoning before calling this.
*/
-static void vm_area_free_pages(struct vm_struct *vm, unsigned int start_idx,
- unsigned int end_idx)
+static void vm_area_free_pages(struct vm_struct *vm, unsigned long start_idx,
+ unsigned long end_idx)
{
- unsigned int i;
+ unsigned long i;
if (!(vm->flags & VM_MAP_PUT_PAGES)) {
for (i = start_idx; i < end_idx; i++)
@@ -3819,12 +3819,12 @@ static inline gfp_t vmalloc_gfp_adjust(gfp_t flags, const bool large)
return flags;
}
-static inline unsigned int
+static inline unsigned long
vm_area_alloc_pages(gfp_t gfp, int nid,
- unsigned int order, unsigned int nr_pages, struct page **pages)
+ unsigned int order, unsigned long nr_pages, struct page **pages)
{
- unsigned int nr_allocated = 0;
- unsigned int nr_remaining = nr_pages;
+ unsigned long nr_allocated = 0;
+ unsigned long nr_remaining = nr_pages;
unsigned int max_attempt_order = MAX_PAGE_ORDER;
struct page *page;
int i;
@@ -3872,7 +3872,7 @@ vm_area_alloc_pages(gfp_t gfp, int nid,
if (!order) {
while (nr_allocated < nr_pages) {
unsigned int nr, nr_pages_request;
- int i;
+ unsigned long i;
/*
* A maximum allowed request is hard-coded and is 100
@@ -3880,7 +3880,7 @@ vm_area_alloc_pages(gfp_t gfp, int nid,
* long preemption off scenario in the bulk-allocator
* so the range is [1:100].
*/
- nr_pages_request = min(100U, nr_pages - nr_allocated);
+ nr_pages_request = min(100UL, nr_pages - nr_allocated);
/* memory allocation should consider mempolicy, we can't
* wrongly use nearest node when nid == NUMA_NO_NODE,
@@ -4026,12 +4026,12 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask,
unsigned long addr = (unsigned long)area->addr;
unsigned long size = get_vm_area_size(area);
unsigned long array_size;
- unsigned int nr_small_pages = size >> PAGE_SHIFT;
+ unsigned long nr_small_pages = size >> PAGE_SHIFT;
unsigned int page_order;
unsigned int flags;
int ret;
- array_size = (unsigned long)nr_small_pages * sizeof(struct page *);
+ array_size = nr_small_pages * sizeof(struct page *);
/* __GFP_NOFAIL and "noblock" flags are mutually exclusive. */
if (!gfpflags_allow_blocking(gfp_mask))
@@ -4525,7 +4525,7 @@ void *vrealloc_node_align_noprof(const void *p, size_t size, unsigned long align
}
if (size <= old_size) {
- unsigned int new_nr_pages = PAGE_ALIGN(size) >> PAGE_SHIFT;
+ unsigned long new_nr_pages = PAGE_ALIGN(size) >> PAGE_SHIFT;
/* Zero out "freed" memory, potentially for future realloc. */
if (want_init_on_free() || want_init_on_alloc(flags))
@@ -4554,7 +4554,7 @@ void *vrealloc_node_align_noprof(const void *p, size_t size, unsigned long align
!(vm->flags & (VM_FLUSH_RESET_PERMS | VM_USERMAP)) &&
gfp_has_io_fs(flags)) {
unsigned long addr = (unsigned long)kasan_reset_tag(p);
- unsigned int old_nr_pages = vm->nr_pages;
+ unsigned long old_nr_pages = vm->nr_pages;
/*
* Use the node lock to synchronize with concurrent
@@ -4567,16 +4567,13 @@ void *vrealloc_node_align_noprof(const void *p, size_t size, unsigned long align
spin_unlock(&vn->busy.lock);
/* Notify kmemleak of the reduced allocation size before unmapping. */
- kmemleak_free_part(
- (void *)addr + ((unsigned long)new_nr_pages
- << PAGE_SHIFT),
- (unsigned long)(old_nr_pages - new_nr_pages)
- << PAGE_SHIFT);
+ kmemleak_free_part((void *)addr +
+ (new_nr_pages << PAGE_SHIFT),
+ (old_nr_pages - new_nr_pages)
+ << PAGE_SHIFT);
- vunmap_range(addr + ((unsigned long)new_nr_pages
- << PAGE_SHIFT),
- addr + ((unsigned long)old_nr_pages
- << PAGE_SHIFT));
+ vunmap_range(addr + (new_nr_pages << PAGE_SHIFT),
+ addr + (old_nr_pages << PAGE_SHIFT));
vm_area_free_pages(vm, new_nr_pages, old_nr_pages);
}
@@ -5400,7 +5397,7 @@ bool vmalloc_dump_obj(void *object)
struct vmap_area *va;
struct vmap_node *vn;
unsigned long addr;
- unsigned int nr_pages;
+ unsigned long nr_pages;
addr = PAGE_ALIGN((unsigned long) object);
vn = addr_to_node(addr);
@@ -5420,7 +5417,7 @@ bool vmalloc_dump_obj(void *object)
nr_pages = vm->nr_pages;
spin_unlock(&vn->busy.lock);
- pr_cont(" %u-page vmalloc region starting at %#lx allocated at %pS\n",
+ pr_cont(" %lu-page vmalloc region starting at %#lx allocated at %pS\n",
nr_pages, addr, caller);
return true;
@@ -5438,16 +5435,17 @@ bool vmalloc_dump_obj(void *object)
static void show_numa_info(struct seq_file *m, struct vm_struct *v,
unsigned int *counters)
{
- unsigned int nr;
unsigned int step = 1U << vm_area_page_order(v);
+ unsigned long i;
+ unsigned int nr;
if (!counters)
return;
memset(counters, 0, nr_node_ids * sizeof(unsigned int));
- for (nr = 0; nr < v->nr_pages; nr += step)
- counters[page_to_nid(v->pages[nr])] += step;
+ for (i = 0; i < v->nr_pages; i += step)
+ counters[page_to_nid(v->pages[i])] += step;
for_each_node_state(nr, N_HIGH_MEMORY)
if (counters[nr])
seq_printf(m, " N%u=%u", nr, counters[nr]);
@@ -5505,7 +5503,7 @@ static int vmalloc_info_show(struct seq_file *m, void *p)
seq_printf(m, " %pS", v->caller);
if (v->nr_pages)
- seq_printf(m, " pages=%d", v->nr_pages);
+ seq_printf(m, " pages=%lu", v->nr_pages);
if (v->phys_addr)
seq_printf(m, " phys=%pa", &v->phys_addr);
base-commit: 1dbd7c34bb92dd9c0b0363b75f6ec444299af108
--
2.43.0
On Sat, 1 Aug 2026 14:49:15 +0300 Artem Lytkin <iprintercanon@gmail.com> wrote: > vm_struct::nr_pages is an unsigned int, and the file keeps deriving byte > counts from it as nr_pages << PAGE_SHIFT. A shift is evaluated in the type > of its promoted left operand, so those are 32-bit arithmetic and wrap at > 4 GiB of bytes, which is 2^20 pages. Every site depends on a cast being > remembered; vmap() has one, two recent commits did not. vread_iter() then > computes a size of zero for a 4 GiB VM_ALLOC area and /proc/kcore returns > it as zeros while reporting a successful read, which drgn, crash or gdb > cannot tell from real memory, and the vrealloc() grow-in-place check > declines a request that would have fit. > > Widen the field so the class of bug goes away instead of one site at a > time. Everything feeding or consuming it widens too: vm_area_alloc_pages() > and its accumulators, nr_small_pages, new_nr_pages and old_nr_pages, the > index range of vm_area_free_pages(), and three page indexes that were > plain int. Five casts go. Two prints needed fixing as well, %u in > vmalloc_dump_obj() and %d for the unsigned field in vmalloc_info_show(). > > No bug report behind this, I found it reading the code. The 4 GiB wrap > needs only a machine with over 4 GiB of memory. Neither larger threshold > is a practical concern: 2^32 pages, where the field itself truncates, is > 16 TiB and beyond what hardware can populate, and 2^31, where the plain > int indexes break, is 8 TiB and larger than anything in the tree asks for. > The int *nr cursor in the mapping path is unchanged and is separate work. > Users outside mm/vmalloc.c need no change either. Those handing the count > to a narrower parameter cannot drive it near 2^31, and > kho_preserve_vmalloc() stores it into a 32-bit ABI field that still > receives the same low bits; above 2^32 pages the truncation just moves out > of vm_struct into that store. > > sizeof(struct vm_struct) on x86-64 stays 72 bytes with > CONFIG_HAVE_ARCH_HUGE_VMALLOC=n and goes from 72 to 80 with it enabled, > both inside the kmalloc-96 bucket it already comes from. Thanks. Ulad, AI review suggests that vrealloc() has an issue handling __GFP_ZERO. Can you please check? https://sashiko.dev/#/patchset/20260801114915.115224-1-iprintercanon@gmail.com
On Sat, Aug 01, 2026 at 11:52:02AM -0700, Andrew Morton wrote:
> On Sat, 1 Aug 2026 14:49:15 +0300 Artem Lytkin <iprintercanon@gmail.com> wrote:
>
> > vm_struct::nr_pages is an unsigned int, and the file keeps deriving byte
> > counts from it as nr_pages << PAGE_SHIFT. A shift is evaluated in the type
> > of its promoted left operand, so those are 32-bit arithmetic and wrap at
> > 4 GiB of bytes, which is 2^20 pages. Every site depends on a cast being
> > remembered; vmap() has one, two recent commits did not. vread_iter() then
> > computes a size of zero for a 4 GiB VM_ALLOC area and /proc/kcore returns
> > it as zeros while reporting a successful read, which drgn, crash or gdb
> > cannot tell from real memory, and the vrealloc() grow-in-place check
> > declines a request that would have fit.
> >
> > Widen the field so the class of bug goes away instead of one site at a
> > time. Everything feeding or consuming it widens too: vm_area_alloc_pages()
> > and its accumulators, nr_small_pages, new_nr_pages and old_nr_pages, the
> > index range of vm_area_free_pages(), and three page indexes that were
> > plain int. Five casts go. Two prints needed fixing as well, %u in
> > vmalloc_dump_obj() and %d for the unsigned field in vmalloc_info_show().
> >
> > No bug report behind this, I found it reading the code. The 4 GiB wrap
> > needs only a machine with over 4 GiB of memory. Neither larger threshold
> > is a practical concern: 2^32 pages, where the field itself truncates, is
> > 16 TiB and beyond what hardware can populate, and 2^31, where the plain
> > int indexes break, is 8 TiB and larger than anything in the tree asks for.
> > The int *nr cursor in the mapping path is unchanged and is separate work.
> > Users outside mm/vmalloc.c need no change either. Those handing the count
> > to a narrower parameter cannot drive it near 2^31, and
> > kho_preserve_vmalloc() stores it into a 32-bit ABI field that still
> > receives the same low bits; above 2^32 pages the truncation just moves out
> > of vm_struct into that store.
> >
> > sizeof(struct vm_struct) on x86-64 stays 72 bytes with
> > CONFIG_HAVE_ARCH_HUGE_VMALLOC=n and goes from 72 to 80 with it enabled,
> > both inside the kmalloc-96 bucket it already comes from.
>
> Thanks.
>
> Ulad, AI review suggests that vrealloc() has an issue handling
> __GFP_ZERO. Can you please check?
>
> https://sashiko.dev/#/patchset/20260801114915.115224-1-iprintercanon@gmail.com
>
I have checked. I think the AI is missing at least one point.
AI argument which is:
<snip>
If a driver initially allocates memory using vmalloc() without __GFP_ZERO
(leaving spare page capacity uninitialized), and then grows the allocation
using vrealloc() with __GFP_ZERO, the caller expects the newly exposed bytes
to be zeroed.
<snip>
In the vrealloc_node_align_noprof() header documentation there is a statement:
<snip>
* If __GFP_ZERO logic is requested, callers must ensure that, starting with the
* initial memory allocation, every subsequent call to this API for the same
* memory allocation is flagged with __GFP_ZERO. Otherwise, it is possible that
* __GFP_ZERO is not fully honored by this API.
<snip>
AI argument violates the documentation, i.e. mixing __GFP_ZERO is not allowed.
From the other hand we can mix it and remove that part of documentation:
<snip>
diff --git a/mm/vmalloc.c b/mm/vmalloc.c
index 7a0cbba3d29d..28d0fed94d22 100644
--- a/mm/vmalloc.c
+++ b/mm/vmalloc.c
@@ -4294,11 +4294,6 @@ EXPORT_SYMBOL(vzalloc_node_noprof);
* __GFP_THISNODE flag should be set, otherwise the function will try to avoid
* reallocation and possibly disregard the specified @nid.
*
- * If __GFP_ZERO logic is requested, callers must ensure that, starting with the
- * initial memory allocation, every subsequent call to this API for the same
- * memory allocation is flagged with __GFP_ZERO. Otherwise, it is possible that
- * __GFP_ZERO is not fully honored by this API.
- *
* Requesting an alignment that is bigger than the alignment of the existing
* allocation will fail.
*
@@ -4415,13 +4410,12 @@ void *vrealloc_node_align_noprof(const void *p, size_t size, unsigned long align
* We already have the bytes available in the allocation; use them.
*/
if (size <= vm->nr_pages << PAGE_SHIFT) {
- /*
- * No need to zero memory here, as unused memory will have
- * already been zeroed at initial allocation time or during
- * realloc shrink time.
- */
- vm->requested_size = size;
kasan_vrealloc(p, old_size, size);
+
+ if (want_init_on_alloc(flags))
+ memset((void *)p + old_size, 0, size - old_size);
+
+ vm->requested_size = size;
return (void *)p;
}
<snip>
--
Uladzislau Rezki
On Sun, 2 Aug 2026 17:52:26 +0200 Uladzislau Rezki <urezki@gmail.com> wrote:
> On Sat, Aug 01, 2026 at 11:52:02AM -0700, Andrew Morton wrote:
> > On Sat, 1 Aug 2026 14:49:15 +0300 Artem Lytkin <iprintercanon@gmail.com> wrote:
> >
>
> ...
>
> > Ulad, AI review suggests that vrealloc() has an issue handling
> > __GFP_ZERO. Can you please check?
> >
> > https://sashiko.dev/#/patchset/20260801114915.115224-1-iprintercanon@gmail.com
> >
> I have checked. I think the AI is missing at least one point.
> AI argument which is:
>
> <snip>
> If a driver initially allocates memory using vmalloc() without __GFP_ZERO
> (leaving spare page capacity uninitialized), and then grows the allocation
> using vrealloc() with __GFP_ZERO, the caller expects the newly exposed bytes
> to be zeroed.
> <snip>
>
> In the vrealloc_node_align_noprof() header documentation there is a statement:
>
> <snip>
> * If __GFP_ZERO logic is requested, callers must ensure that, starting with the
> * initial memory allocation, every subsequent call to this API for the same
> * memory allocation is flagged with __GFP_ZERO. Otherwise, it is possible that
> * __GFP_ZERO is not fully honored by this API.
> <snip>
>
> AI argument violates the documentation, i.e. mixing __GFP_ZERO is not allowed.
Sashiko is talking about the initial allocation not using __GFP_ZERO
but vrealloc() *does* use __GFP_ZERO. The documentation you quoted
doesn't address that case?
Also, developers don't read documentation ;) What happens if some
caller *does* use vmalloc(!__GFP_ZERO) then vrealloc(__GFP_ZERO)?
Silent misbehavior would be bad - it would be good if vrealloc() were
to drop a WARN() then ignore the __GFP_ZERO.
> --- a/mm/vmalloc.c
> +++ b/mm/vmalloc.c
> @@ -4294,11 +4294,6 @@ EXPORT_SYMBOL(vzalloc_node_noprof);
> * __GFP_THISNODE flag should be set, otherwise the function will try to avoid
> * reallocation and possibly disregard the specified @nid.
> *
> - * If __GFP_ZERO logic is requested, callers must ensure that, starting with the
> - * initial memory allocation, every subsequent call to this API for the same
> - * memory allocation is flagged with __GFP_ZERO. Otherwise, it is possible that
> - * __GFP_ZERO is not fully honored by this API.
> - *
> * Requesting an alignment that is bigger than the alignment of the existing
> * allocation will fail.
> *
> @@ -4415,13 +4410,12 @@ void *vrealloc_node_align_noprof(const void *p, size_t size, unsigned long align
> * We already have the bytes available in the allocation; use them.
> */
> if (size <= vm->nr_pages << PAGE_SHIFT) {
> - /*
> - * No need to zero memory here, as unused memory will have
> - * already been zeroed at initial allocation time or during
> - * realloc shrink time.
> - */
> - vm->requested_size = size;
> kasan_vrealloc(p, old_size, size);
> +
> + if (want_init_on_alloc(flags))
> + memset((void *)p + old_size, 0, size - old_size);
> +
> + vm->requested_size = size;
> return (void *)p;
> }
OK, thanks, I'll assume you'll prepare this for real when convenient.
On Mon, Aug 03, 2026 at 05:39:35PM -0700, Andrew Morton wrote:
> On Sun, 2 Aug 2026 17:52:26 +0200 Uladzislau Rezki <urezki@gmail.com> wrote:
>
> > On Sat, Aug 01, 2026 at 11:52:02AM -0700, Andrew Morton wrote:
> > > On Sat, 1 Aug 2026 14:49:15 +0300 Artem Lytkin <iprintercanon@gmail.com> wrote:
> > >
> >
> > ...
> >
> > > Ulad, AI review suggests that vrealloc() has an issue handling
> > > __GFP_ZERO. Can you please check?
> > >
> > > https://sashiko.dev/#/patchset/20260801114915.115224-1-iprintercanon@gmail.com
> > >
> > I have checked. I think the AI is missing at least one point.
> > AI argument which is:
> >
> > <snip>
> > If a driver initially allocates memory using vmalloc() without __GFP_ZERO
> > (leaving spare page capacity uninitialized), and then grows the allocation
> > using vrealloc() with __GFP_ZERO, the caller expects the newly exposed bytes
> > to be zeroed.
> > <snip>
> >
> > In the vrealloc_node_align_noprof() header documentation there is a statement:
> >
> > <snip>
> > * If __GFP_ZERO logic is requested, callers must ensure that, starting with the
> > * initial memory allocation, every subsequent call to this API for the same
> > * memory allocation is flagged with __GFP_ZERO. Otherwise, it is possible that
> > * __GFP_ZERO is not fully honored by this API.
> > <snip>
> >
> > AI argument violates the documentation, i.e. mixing __GFP_ZERO is not allowed.
>
> Sashiko is talking about the initial allocation not using __GFP_ZERO
> but vrealloc() *does* use __GFP_ZERO. The documentation you quoted
> doesn't address that case?
>
But this is not allowed according to doc :)
<snip>
...
callers must ensure that, starting with the initial memory allocation
...
<snip>
Also, AI describes the situation like:
vmalloc()
vrealloc(grow, since need more)
i.e. from the description:
<snip>
and then grows the allocation using vrealloc() with __GFP_ZERO, the caller
expects the newly exposed bytes to be zeroed.
<snip>
and it will be zeroed in fact, because the path would be:
<snip>
need_realloc:
/* TODO: Grow the vm_area, i.e. allocate and map additional pages. */
n = __vmalloc_node_noprof(size, align, flags, nid, __builtin_return_address(0));
if (!n)
return NULL;
<snip>
and not the one which AI pointed to.
The real scenario is:
1. vmalloc(!GFP_ZERO) - alloc size 10
2. vrealloc(!GFP_ZERO) - realloc to size 5
3. vrealloc(GFP_ZERO) - realloc back to 10.
3 - will not zeroed. As noted in doc ZERO should be used starting
from the beginning [1]. But in __most__ cases it will be zeroed
anyway because we free/unmap tail pages and if:
<snip>
/*
* Free tail pages when shrink crosses a page boundary.
*
* Skip huge page allocations (page_order > 0) as partial
* freeing would require splitting.
*
* Skip VM_FLUSH_RESET_PERMS, as direct-map permissions must
* be reset before pages are returned to the allocator.
*
* Skip VM_USERMAP, as remap_vmalloc_range_partial() validates
* mapping requests against the unchanged vm->size; freeing
* tail pages would cause vmalloc_to_page() to return NULL for
* the unmapped range.
*
* Skip if either GFP_NOFS or GFP_NOIO are used.
* kmemleak_free_part() internally allocates with
* GFP_KERNEL, which could trigger a recursive deadlock
* if we are under filesystem or I/O reclaim.
*/
if (new_nr_pages < vm->nr_pages && !vm_area_page_order(vm) &&
!(vm->flags & (VM_FLUSH_RESET_PERMS | VM_USERMAP)) &&
gfp_has_io_fs(flags)) {
<snip>
is true the next grow with GFP_ZERO will be zeroed.
For huge alloc it will not be zeroed and for other conditions.
> Also, developers don't read documentation ;) What happens if some
> caller *does* use vmalloc(!__GFP_ZERO) then vrealloc(__GFP_ZERO)?
> Silent misbehavior would be bad - it would be good if vrealloc() were
> to drop a WARN() then ignore the __GFP_ZERO.
>
> > --- a/mm/vmalloc.c
> > +++ b/mm/vmalloc.c
> > @@ -4294,11 +4294,6 @@ EXPORT_SYMBOL(vzalloc_node_noprof);
> > * __GFP_THISNODE flag should be set, otherwise the function will try to avoid
> > * reallocation and possibly disregard the specified @nid.
> > *
> > - * If __GFP_ZERO logic is requested, callers must ensure that, starting with the
> > - * initial memory allocation, every subsequent call to this API for the same
> > - * memory allocation is flagged with __GFP_ZERO. Otherwise, it is possible that
> > - * __GFP_ZERO is not fully honored by this API.
> > - *
> > * Requesting an alignment that is bigger than the alignment of the existing
> > * allocation will fail.
> > *
> > @@ -4415,13 +4410,12 @@ void *vrealloc_node_align_noprof(const void *p, size_t size, unsigned long align
> > * We already have the bytes available in the allocation; use them.
> > */
> > if (size <= vm->nr_pages << PAGE_SHIFT) {
> > - /*
> > - * No need to zero memory here, as unused memory will have
> > - * already been zeroed at initial allocation time or during
> > - * realloc shrink time.
> > - */
> > - vm->requested_size = size;
> > kasan_vrealloc(p, old_size, size);
> > +
> > + if (want_init_on_alloc(flags))
> > + memset((void *)p + old_size, 0, size - old_size);
> > +
> > + vm->requested_size = size;
> > return (void *)p;
> > }
>
> OK, thanks, I'll assume you'll prepare this for real when convenient.
>
OK. I will prepare something and send out the patch after testing.
--
Uladzislau Rezki
© 2016 - 2026 Red Hat, Inc.