fs/ceph/mds_client.c | 11 +++++++++++ 1 file changed, 11 insertions(+)
handle_reply() stores a `ceph_mds_request` pointer in
`current->journal_info` while filling the inode and dentry cache from
an MDS reply.
An allocation in this section can enter direct reclaim and prune
dentries from another filesystem. If this dirties an ext4 inode, ext4
starts a JBD2 transaction. JBD2 interprets the Ceph request in
`current->journal_info` as a journal handle and dereferences the
request's `r_tid` as `h_transaction`, causing a kernel crash, e.g.:
Unable to handle kernel paging request at virtual address 00000000077b4818
[...]
Internal error: Oops: 0000000096000004 [#1] SMP
Modules linked in:
CPU: 6 UID: 0 PID: 2699135 Comm: kworker/6:3 Tainted: G W 6.18.38-i3 #1113 NONE
[...]
Workqueue: ceph-msgr ceph_con_workfn
pstate: 80400009 (Nzcv daif +PAN -UAO -TCO -DIT -SSBS BTYPE=--)
pc : jbd2__journal_start+0x2c/0x208
lr : __ext4_journal_start_sb+0x100/0x178
[...]
Call trace:
jbd2__journal_start+0x2c/0x208 (P)
__ext4_journal_start_sb+0x100/0x178
ext4_dirty_inode+0x3c/0x90
__mark_inode_dirty+0x58/0x400
iput.part.0+0x2b0/0x370
iput+0x18/0x30
dentry_unlink_inode+0xc0/0x158
__dentry_kill+0x80/0x250
shrink_dentry_list+0x90/0x130
prune_dcache_sb+0x60/0x98
super_cache_scan+0xe8/0x190
do_shrink_slab+0x174/0x388
shrink_slab+0xd8/0x4c0
shrink_node+0x31c/0x908
do_try_to_free_pages+0xd0/0x508
try_to_free_pages+0x11c/0x238
__alloc_frozen_pages_noprof+0x4d0/0xdd0
__folio_alloc_noprof+0x18/0x70
__filemap_get_folio+0x248/0x440
ceph_readdir_prepopulate+0x570/0x9e8
mds_dispatch+0x1424/0x1ba0
ceph_con_process_message+0x74/0xa0
ceph_con_v1_try_read+0x3a0/0x1510
ceph_con_workfn+0x260/0x460
Enter a scoped NOFS allocation context and leave it after clearing
`journal_info`. This prevents filesystem reclaim from recursing into
another filesystem while the field contains Ceph-private data.
Fixes: 315f24088048 ("ceph: fix security xattr deadlock")
Cc: stable@vger.kernel.org
Signed-off-by: Max Kellermann <max.kellermann@ionos.com>
---
fs/ceph/mds_client.c | 11 +++++++++++
1 file changed, 11 insertions(+)
diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
index 853bf698b356..3c692ad02c85 100644
--- a/fs/ceph/mds_client.c
+++ b/fs/ceph/mds_client.c
@@ -6,6 +6,7 @@
#include <linux/slab.h>
#include <linux/gfp.h>
#include <linux/sched.h>
+#include <linux/sched/mm.h>
#include <linux/delay.h>
#include <linux/debugfs.h>
#include <linux/seq_file.h>
@@ -4015,6 +4016,7 @@ static void handle_reply(struct ceph_mds_session *session, struct ceph_msg *msg)
struct ceph_mds_reply_head *head = msg->front.iov_base;
struct ceph_mds_reply_info_parsed *rinfo; /* parsed reply info */
struct ceph_snap_realm *realm;
+ unsigned int nofs_flags;
u64 tid;
int err, result;
int mds = session->s_mds;
@@ -4158,6 +4160,14 @@ static void handle_reply(struct ceph_mds_session *session, struct ceph_msg *msg)
/* insert trace into our cache */
mutex_lock(&req->r_fill_mutex);
+
+ /* disable fs reclaim while we are using current->journal_info
+ * for our own purposes, or else shrinkers of other
+ * filesystems might dereference this pointer as a different
+ * type
+ */
+ nofs_flags = memalloc_nofs_save();
+
current->journal_info = req;
err = ceph_fill_trace(mdsc->fsc->sb, req);
if (err == 0) {
@@ -4166,6 +4176,7 @@ static void handle_reply(struct ceph_mds_session *session, struct ceph_msg *msg)
err = ceph_readdir_prepopulate(req, req->r_session);
}
current->journal_info = NULL;
+ memalloc_nofs_restore(nofs_flags);
mutex_unlock(&req->r_fill_mutex);
up_read(&mdsc->snap_rwsem);
--
2.47.3
On Wed, 2026-07-22 at 13:49 +0200, Max Kellermann wrote:
> handle_reply() stores a `ceph_mds_request` pointer in
> `current->journal_info` while filling the inode and dentry cache from
> an MDS reply.
>
> An allocation in this section can enter direct reclaim and prune
> dentries from another filesystem. If this dirties an ext4 inode,
> ext4
> starts a JBD2 transaction. JBD2 interprets the Ceph request in
> `current->journal_info` as a journal handle and dereferences the
> request's `r_tid` as `h_transaction`, causing a kernel crash, e.g.:
>
> Unable to handle kernel paging request at virtual address
> 00000000077b4818
> [...]
> Internal error: Oops: 0000000096000004 [#1] SMP
> Modules linked in:
> CPU: 6 UID: 0 PID: 2699135 Comm: kworker/6:3 Tainted: G
> W 6.18.38-i3 #1113 NONE
> [...]
> Workqueue: ceph-msgr ceph_con_workfn
> pstate: 80400009 (Nzcv daif +PAN -UAO -TCO -DIT -SSBS BTYPE=--)
> pc : jbd2__journal_start+0x2c/0x208
> lr : __ext4_journal_start_sb+0x100/0x178
> [...]
> Call trace:
> jbd2__journal_start+0x2c/0x208 (P)
> __ext4_journal_start_sb+0x100/0x178
> ext4_dirty_inode+0x3c/0x90
> __mark_inode_dirty+0x58/0x400
> iput.part.0+0x2b0/0x370
> iput+0x18/0x30
> dentry_unlink_inode+0xc0/0x158
> __dentry_kill+0x80/0x250
> shrink_dentry_list+0x90/0x130
> prune_dcache_sb+0x60/0x98
> super_cache_scan+0xe8/0x190
> do_shrink_slab+0x174/0x388
> shrink_slab+0xd8/0x4c0
> shrink_node+0x31c/0x908
> do_try_to_free_pages+0xd0/0x508
> try_to_free_pages+0x11c/0x238
> __alloc_frozen_pages_noprof+0x4d0/0xdd0
> __folio_alloc_noprof+0x18/0x70
> __filemap_get_folio+0x248/0x440
> ceph_readdir_prepopulate+0x570/0x9e8
> mds_dispatch+0x1424/0x1ba0
> ceph_con_process_message+0x74/0xa0
> ceph_con_v1_try_read+0x3a0/0x1510
> ceph_con_workfn+0x260/0x460
>
> Enter a scoped NOFS allocation context and leave it after clearing
> `journal_info`. This prevents filesystem reclaim from recursing into
> another filesystem while the field contains Ceph-private data.
>
> Fixes: 315f24088048 ("ceph: fix security xattr deadlock")
> Cc: stable@vger.kernel.org
> Signed-off-by: Max Kellermann <max.kellermann@ionos.com>
> ---
> fs/ceph/mds_client.c | 11 +++++++++++
> 1 file changed, 11 insertions(+)
>
> diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
> index 853bf698b356..3c692ad02c85 100644
> --- a/fs/ceph/mds_client.c
> +++ b/fs/ceph/mds_client.c
> @@ -6,6 +6,7 @@
> #include <linux/slab.h>
> #include <linux/gfp.h>
> #include <linux/sched.h>
> +#include <linux/sched/mm.h>
> #include <linux/delay.h>
> #include <linux/debugfs.h>
> #include <linux/seq_file.h>
> @@ -4015,6 +4016,7 @@ static void handle_reply(struct
> ceph_mds_session *session, struct ceph_msg *msg)
> struct ceph_mds_reply_head *head = msg->front.iov_base;
> struct ceph_mds_reply_info_parsed *rinfo; /* parsed reply
> info */
> struct ceph_snap_realm *realm;
> + unsigned int nofs_flags;
> u64 tid;
> int err, result;
> int mds = session->s_mds;
> @@ -4158,6 +4160,14 @@ static void handle_reply(struct
> ceph_mds_session *session, struct ceph_msg *msg)
>
> /* insert trace into our cache */
> mutex_lock(&req->r_fill_mutex);
> +
> + /* disable fs reclaim while we are using current-
> >journal_info
> + * for our own purposes, or else shrinkers of other
> + * filesystems might dereference this pointer as a different
> + * type
> + */
> + nofs_flags = memalloc_nofs_save();
> +
> current->journal_info = req;
> err = ceph_fill_trace(mdsc->fsc->sb, req);
> if (err == 0) {
> @@ -4166,6 +4176,7 @@ static void handle_reply(struct
> ceph_mds_session *session, struct ceph_msg *msg)
> err = ceph_readdir_prepopulate(req, req-
> >r_session);
> }
> current->journal_info = NULL;
> + memalloc_nofs_restore(nofs_flags);
> mutex_unlock(&req->r_fill_mutex);
>
> up_read(&mdsc->snap_rwsem);
Makes sense.
Reviewed-by: Viacheslav Dubeyko <slava@dubeyko.com>
Thanks,
Slava.
LGTM.
Reviewed-by: Xiubo Li <xiubo.li@clyso.com>
Thanks & Regards
Xiubo
On Thu, 23 Jul 2026 at 01:20, Viacheslav Dubeyko <slava@dubeyko.com> wrote:
>
> On Wed, 2026-07-22 at 13:49 +0200, Max Kellermann wrote:
> > handle_reply() stores a `ceph_mds_request` pointer in
> > `current->journal_info` while filling the inode and dentry cache from
> > an MDS reply.
> >
> > An allocation in this section can enter direct reclaim and prune
> > dentries from another filesystem. If this dirties an ext4 inode,
> > ext4
> > starts a JBD2 transaction. JBD2 interprets the Ceph request in
> > `current->journal_info` as a journal handle and dereferences the
> > request's `r_tid` as `h_transaction`, causing a kernel crash, e.g.:
> >
> > Unable to handle kernel paging request at virtual address
> > 00000000077b4818
> > [...]
> > Internal error: Oops: 0000000096000004 [#1] SMP
> > Modules linked in:
> > CPU: 6 UID: 0 PID: 2699135 Comm: kworker/6:3 Tainted: G
> > W 6.18.38-i3 #1113 NONE
> > [...]
> > Workqueue: ceph-msgr ceph_con_workfn
> > pstate: 80400009 (Nzcv daif +PAN -UAO -TCO -DIT -SSBS BTYPE=--)
> > pc : jbd2__journal_start+0x2c/0x208
> > lr : __ext4_journal_start_sb+0x100/0x178
> > [...]
> > Call trace:
> > jbd2__journal_start+0x2c/0x208 (P)
> > __ext4_journal_start_sb+0x100/0x178
> > ext4_dirty_inode+0x3c/0x90
> > __mark_inode_dirty+0x58/0x400
> > iput.part.0+0x2b0/0x370
> > iput+0x18/0x30
> > dentry_unlink_inode+0xc0/0x158
> > __dentry_kill+0x80/0x250
> > shrink_dentry_list+0x90/0x130
> > prune_dcache_sb+0x60/0x98
> > super_cache_scan+0xe8/0x190
> > do_shrink_slab+0x174/0x388
> > shrink_slab+0xd8/0x4c0
> > shrink_node+0x31c/0x908
> > do_try_to_free_pages+0xd0/0x508
> > try_to_free_pages+0x11c/0x238
> > __alloc_frozen_pages_noprof+0x4d0/0xdd0
> > __folio_alloc_noprof+0x18/0x70
> > __filemap_get_folio+0x248/0x440
> > ceph_readdir_prepopulate+0x570/0x9e8
> > mds_dispatch+0x1424/0x1ba0
> > ceph_con_process_message+0x74/0xa0
> > ceph_con_v1_try_read+0x3a0/0x1510
> > ceph_con_workfn+0x260/0x460
> >
> > Enter a scoped NOFS allocation context and leave it after clearing
> > `journal_info`. This prevents filesystem reclaim from recursing into
> > another filesystem while the field contains Ceph-private data.
> >
> > Fixes: 315f24088048 ("ceph: fix security xattr deadlock")
> > Cc: stable@vger.kernel.org
> > Signed-off-by: Max Kellermann <max.kellermann@ionos.com>
> > ---
> > fs/ceph/mds_client.c | 11 +++++++++++
> > 1 file changed, 11 insertions(+)
> >
> > diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
> > index 853bf698b356..3c692ad02c85 100644
> > --- a/fs/ceph/mds_client.c
> > +++ b/fs/ceph/mds_client.c
> > @@ -6,6 +6,7 @@
> > #include <linux/slab.h>
> > #include <linux/gfp.h>
> > #include <linux/sched.h>
> > +#include <linux/sched/mm.h>
> > #include <linux/delay.h>
> > #include <linux/debugfs.h>
> > #include <linux/seq_file.h>
> > @@ -4015,6 +4016,7 @@ static void handle_reply(struct
> > ceph_mds_session *session, struct ceph_msg *msg)
> > struct ceph_mds_reply_head *head = msg->front.iov_base;
> > struct ceph_mds_reply_info_parsed *rinfo; /* parsed reply
> > info */
> > struct ceph_snap_realm *realm;
> > + unsigned int nofs_flags;
> > u64 tid;
> > int err, result;
> > int mds = session->s_mds;
> > @@ -4158,6 +4160,14 @@ static void handle_reply(struct
> > ceph_mds_session *session, struct ceph_msg *msg)
> >
> > /* insert trace into our cache */
> > mutex_lock(&req->r_fill_mutex);
> > +
> > + /* disable fs reclaim while we are using current-
> > >journal_info
> > + * for our own purposes, or else shrinkers of other
> > + * filesystems might dereference this pointer as a different
> > + * type
> > + */
> > + nofs_flags = memalloc_nofs_save();
> > +
> > current->journal_info = req;
> > err = ceph_fill_trace(mdsc->fsc->sb, req);
> > if (err == 0) {
> > @@ -4166,6 +4176,7 @@ static void handle_reply(struct
> > ceph_mds_session *session, struct ceph_msg *msg)
> > err = ceph_readdir_prepopulate(req, req-
> > >r_session);
> > }
> > current->journal_info = NULL;
> > + memalloc_nofs_restore(nofs_flags);
> > mutex_unlock(&req->r_fill_mutex);
> >
> > up_read(&mdsc->snap_rwsem);
>
> Makes sense.
>
> Reviewed-by: Viacheslav Dubeyko <slava@dubeyko.com>
>
> Thanks,
> Slava.
© 2016 - 2026 Red Hat, Inc.