[PATCH v2] nilfs2: fix checkpoint root lifetime on sysfs errors

Aldo Ariel Panzardo posted 1 patch 1 week, 2 days ago
There is a newer version of this series
fs/nilfs2/the_nilfs.c | 35 +++++++++++++++++++----------------
fs/nilfs2/the_nilfs.h |  2 +-
2 files changed, 20 insertions(+), 17 deletions(-)
[PATCH v2] nilfs2: fix checkpoint root lifetime on sysfs errors
Posted by Aldo Ariel Panzardo 1 week, 2 days ago
nilfs_find_or_create_root() publishes a new root in the checkpoint
tree before creating its sysfs object.  If sysfs registration fails,
the root is freed while it is still reachable from the tree.  A
concurrent nilfs_lookup_root() can then dereference freed memory.

The fix needs the lock to be held across the sysfs call, but
nilfs_sysfs_create_snapshot_group() can sleep, so the existing
spinlock is not suitable.

Convert ns_cptree_lock from a spinlock to a mutex.  All existing
callers are in process context (mount, lookup, segctor, recovery),
and nilfs_put_root() can use refcount_dec_and_mutex_lock() as the
atomic decrement-and-acquire primitive.

With the mutex, nilfs_find_or_create_root() can hold it across the
sysfs registration and only insert the root into the rbtree after
sysfs succeeds.  On failure, the root was never visible and can be
freed after waiting for the kobject release callback to complete.

Both the creation error path and the normal removal path must wait
for the embedded kobject release via wait_for_completion() before
freeing the container, because kobject_put() does not guarantee
synchronous release (CONFIG_DEBUG_KOBJECT_RELEASE defers it).

Fixes: dd70edbde262 ("nilfs2: integrate sysfs support into driver")
Cc: stable@vger.kernel.org
Signed-off-by: Aldo Ariel Panzardo <qwe.aldo@gmail.com>
---

Changes in v2:
  - Convert ns_cptree_lock from spinlock to mutex instead of adding
    a second lock, as suggested by Viacheslav Dubeyko.  Use
    refcount_dec_and_mutex_lock() in nilfs_put_root().

 fs/nilfs2/the_nilfs.c | 35 +++++++++++++++++++----------------
 fs/nilfs2/the_nilfs.h |  2 +-
 2 files changed, 20 insertions(+), 17 deletions(-)

diff --git a/fs/nilfs2/the_nilfs.c b/fs/nilfs2/the_nilfs.c
index 7b23e373a1..5bfa853471 100644
--- a/fs/nilfs2/the_nilfs.c
+++ b/fs/nilfs2/the_nilfs.c
@@ -70,7 +70,7 @@ struct the_nilfs *alloc_nilfs(struct super_block *sb)
 	spin_lock_init(&nilfs->ns_inode_lock);
 	spin_lock_init(&nilfs->ns_last_segment_lock);
 	nilfs->ns_cptree = RB_ROOT;
-	spin_lock_init(&nilfs->ns_cptree_lock);
+	mutex_init(&nilfs->ns_cptree_lock);
 	init_rwsem(&nilfs->ns_segctor_sem);
 	nilfs->ns_sb_update_freq = NILFS_SB_FREQ;
 
@@ -846,7 +846,7 @@ struct nilfs_root *nilfs_lookup_root(struct the_nilfs *nilfs, __u64 cno)
 	struct rb_node *n;
 	struct nilfs_root *root;
 
-	spin_lock(&nilfs->ns_cptree_lock);
+	mutex_lock(&nilfs->ns_cptree_lock);
 	n = nilfs->ns_cptree.rb_node;
 	while (n) {
 		root = rb_entry(n, struct nilfs_root, rb_node);
@@ -857,11 +857,11 @@ struct nilfs_root *nilfs_lookup_root(struct the_nilfs *nilfs, __u64 cno)
 			n = n->rb_right;
 		} else {
 			refcount_inc(&root->count);
-			spin_unlock(&nilfs->ns_cptree_lock);
+			mutex_unlock(&nilfs->ns_cptree_lock);
 			return root;
 		}
 	}
-	spin_unlock(&nilfs->ns_cptree_lock);
+	mutex_unlock(&nilfs->ns_cptree_lock);
 
 	return NULL;
 }
@@ -881,7 +881,7 @@ nilfs_find_or_create_root(struct the_nilfs *nilfs, __u64 cno)
 	if (!new)
 		return NULL;
 
-	spin_lock(&nilfs->ns_cptree_lock);
+	mutex_lock(&nilfs->ns_cptree_lock);
 
 	p = &nilfs->ns_cptree.rb_node;
 	parent = NULL;
@@ -896,7 +896,7 @@ nilfs_find_or_create_root(struct the_nilfs *nilfs, __u64 cno)
 			p = &(*p)->rb_right;
 		} else {
 			refcount_inc(&root->count);
-			spin_unlock(&nilfs->ns_cptree_lock);
+			mutex_unlock(&nilfs->ns_cptree_lock);
 			kfree(new);
 			return root;
 		}
@@ -909,17 +909,19 @@ nilfs_find_or_create_root(struct the_nilfs *nilfs, __u64 cno)
 	atomic64_set(&new->inodes_count, 0);
 	atomic64_set(&new->blocks_count, 0);
 
-	rb_link_node(&new->rb_node, parent, p);
-	rb_insert_color(&new->rb_node, &nilfs->ns_cptree);
-
-	spin_unlock(&nilfs->ns_cptree_lock);
-
 	err = nilfs_sysfs_create_snapshot_group(new);
 	if (err) {
+		mutex_unlock(&nilfs->ns_cptree_lock);
+		wait_for_completion(&new->snapshot_kobj_unregister);
 		kfree(new);
-		new = NULL;
+		return NULL;
 	}
 
+	rb_link_node(&new->rb_node, parent, p);
+	rb_insert_color(&new->rb_node, &nilfs->ns_cptree);
+
+	mutex_unlock(&nilfs->ns_cptree_lock);
+
 	return new;
 }
 
@@ -927,13 +929,14 @@ void nilfs_put_root(struct nilfs_root *root)
 {
 	struct the_nilfs *nilfs = root->nilfs;
 
-	if (refcount_dec_and_lock(&root->count, &nilfs->ns_cptree_lock)) {
+	if (refcount_dec_and_mutex_lock(&root->count,
+					&nilfs->ns_cptree_lock)) {
 		rb_erase(&root->rb_node, &nilfs->ns_cptree);
-		spin_unlock(&nilfs->ns_cptree_lock);
-
 		nilfs_sysfs_delete_snapshot_group(root);
-		iput(root->ifile);
+		wait_for_completion(&root->snapshot_kobj_unregister);
+		mutex_unlock(&nilfs->ns_cptree_lock);
 
+		iput(root->ifile);
 		kfree(root);
 	}
 }
diff --git a/fs/nilfs2/the_nilfs.h b/fs/nilfs2/the_nilfs.h
index 4776a70f01..074644c64a 100644
--- a/fs/nilfs2/the_nilfs.h
+++ b/fs/nilfs2/the_nilfs.h
@@ -150,7 +150,7 @@ struct the_nilfs {
 
 	/* Checkpoint tree */
 	struct rb_root		ns_cptree;
-	spinlock_t		ns_cptree_lock;
+	struct mutex		ns_cptree_lock; /* Protects ns_cptree */
 
 	/* Dirty inode list */
 	struct list_head	ns_dirty_files;
-- 
2.43.0
Re: [PATCH v2] nilfs2: fix checkpoint root lifetime on sysfs errors
Posted by Viacheslav Dubeyko 1 week, 1 day ago
On Tue, 2026-09-15 at 16:49 -0300, Aldo Ariel Panzardo wrote:
> nilfs_find_or_create_root() publishes a new root in the checkpoint
> tree before creating its sysfs object.  If sysfs registration fails,
> the root is freed while it is still reachable from the tree.  A
> concurrent nilfs_lookup_root() can then dereference freed memory.
> 
> The fix needs the lock to be held across the sysfs call, but
> nilfs_sysfs_create_snapshot_group() can sleep, so the existing
> spinlock is not suitable.
> 
> Convert ns_cptree_lock from a spinlock to a mutex.  All existing
> callers are in process context (mount, lookup, segctor, recovery),
> and nilfs_put_root() can use refcount_dec_and_mutex_lock() as the
> atomic decrement-and-acquire primitive.
> 
> With the mutex, nilfs_find_or_create_root() can hold it across the
> sysfs registration and only insert the root into the rbtree after
> sysfs succeeds.  On failure, the root was never visible and can be
> freed after waiting for the kobject release callback to complete.
> 
> Both the creation error path and the normal removal path must wait
> for the embedded kobject release via wait_for_completion() before
> freeing the container, because kobject_put() does not guarantee
> synchronous release (CONFIG_DEBUG_KOBJECT_RELEASE defers it).
> 
> Fixes: dd70edbde262 ("nilfs2: integrate sysfs support into driver")
> Cc: stable@vger.kernel.org
> Signed-off-by: Aldo Ariel Panzardo <qwe.aldo@gmail.com>
> ---
> 
> Changes in v2:
>   - Convert ns_cptree_lock from spinlock to mutex instead of adding
>     a second lock, as suggested by Viacheslav Dubeyko.  Use
>     refcount_dec_and_mutex_lock() in nilfs_put_root().
> 
>  fs/nilfs2/the_nilfs.c | 35 +++++++++++++++++++----------------
>  fs/nilfs2/the_nilfs.h |  2 +-
>  2 files changed, 20 insertions(+), 17 deletions(-)
> 
> diff --git a/fs/nilfs2/the_nilfs.c b/fs/nilfs2/the_nilfs.c
> index 7b23e373a1..5bfa853471 100644
> --- a/fs/nilfs2/the_nilfs.c
> +++ b/fs/nilfs2/the_nilfs.c
> @@ -70,7 +70,7 @@ struct the_nilfs *alloc_nilfs(struct super_block
> *sb)
>  	spin_lock_init(&nilfs->ns_inode_lock);
>  	spin_lock_init(&nilfs->ns_last_segment_lock);
>  	nilfs->ns_cptree = RB_ROOT;
> -	spin_lock_init(&nilfs->ns_cptree_lock);
> +	mutex_init(&nilfs->ns_cptree_lock);
>  	init_rwsem(&nilfs->ns_segctor_sem);
>  	nilfs->ns_sb_update_freq = NILFS_SB_FREQ;
>  
> @@ -846,7 +846,7 @@ struct nilfs_root *nilfs_lookup_root(struct
> the_nilfs *nilfs, __u64 cno)
>  	struct rb_node *n;
>  	struct nilfs_root *root;
>  
> -	spin_lock(&nilfs->ns_cptree_lock);
> +	mutex_lock(&nilfs->ns_cptree_lock);
>  	n = nilfs->ns_cptree.rb_node;
>  	while (n) {
>  		root = rb_entry(n, struct nilfs_root, rb_node);
> @@ -857,11 +857,11 @@ struct nilfs_root *nilfs_lookup_root(struct
> the_nilfs *nilfs, __u64 cno)
>  			n = n->rb_right;
>  		} else {
>  			refcount_inc(&root->count);
> -			spin_unlock(&nilfs->ns_cptree_lock);
> +			mutex_unlock(&nilfs->ns_cptree_lock);
>  			return root;
>  		}
>  	}
> -	spin_unlock(&nilfs->ns_cptree_lock);
> +	mutex_unlock(&nilfs->ns_cptree_lock);
>  
>  	return NULL;
>  }
> @@ -881,7 +881,7 @@ nilfs_find_or_create_root(struct the_nilfs
> *nilfs, __u64 cno)
>  	if (!new)
>  		return NULL;
>  
> -	spin_lock(&nilfs->ns_cptree_lock);
> +	mutex_lock(&nilfs->ns_cptree_lock);
>  
>  	p = &nilfs->ns_cptree.rb_node;
>  	parent = NULL;
> @@ -896,7 +896,7 @@ nilfs_find_or_create_root(struct the_nilfs
> *nilfs, __u64 cno)
>  			p = &(*p)->rb_right;
>  		} else {
>  			refcount_inc(&root->count);
> -			spin_unlock(&nilfs->ns_cptree_lock);
> +			mutex_unlock(&nilfs->ns_cptree_lock);
>  			kfree(new);
>  			return root;
>  		}
> @@ -909,17 +909,19 @@ nilfs_find_or_create_root(struct the_nilfs
> *nilfs, __u64 cno)
>  	atomic64_set(&new->inodes_count, 0);
>  	atomic64_set(&new->blocks_count, 0);
>  
> -	rb_link_node(&new->rb_node, parent, p);
> -	rb_insert_color(&new->rb_node, &nilfs->ns_cptree);
> -
> -	spin_unlock(&nilfs->ns_cptree_lock);
> -
>  	err = nilfs_sysfs_create_snapshot_group(new);
>  	if (err) {
> +		mutex_unlock(&nilfs->ns_cptree_lock);
> +		wait_for_completion(&new->snapshot_kobj_unregister);
>  		kfree(new);
> -		new = NULL;
> +		return NULL;
>  	}
>  
> +	rb_link_node(&new->rb_node, parent, p);
> +	rb_insert_color(&new->rb_node, &nilfs->ns_cptree);
> +
> +	mutex_unlock(&nilfs->ns_cptree_lock);
> +
>  	return new;
>  }
>  
> @@ -927,13 +929,14 @@ void nilfs_put_root(struct nilfs_root *root)
>  {
>  	struct the_nilfs *nilfs = root->nilfs;
>  
> -	if (refcount_dec_and_lock(&root->count, &nilfs-
> >ns_cptree_lock)) {
> +	if (refcount_dec_and_mutex_lock(&root->count,
> +					&nilfs->ns_cptree_lock)) {
>  		rb_erase(&root->rb_node, &nilfs->ns_cptree);
> -		spin_unlock(&nilfs->ns_cptree_lock);
> -
>  		nilfs_sysfs_delete_snapshot_group(root);
> -		iput(root->ifile);
> +		wait_for_completion(&root-
> >snapshot_kobj_unregister);

Is it right behavior that we are waiting completion under the lock?
Could we have potential deadlock here? I am feeling slightly bad about
waiting something under the lock.

> +		mutex_unlock(&nilfs->ns_cptree_lock);
>  
> +		iput(root->ifile);
>  		kfree(root);
>  	}
>  }
> diff --git a/fs/nilfs2/the_nilfs.h b/fs/nilfs2/the_nilfs.h
> index 4776a70f01..074644c64a 100644
> --- a/fs/nilfs2/the_nilfs.h
> +++ b/fs/nilfs2/the_nilfs.h
> @@ -150,7 +150,7 @@ struct the_nilfs {
>  
>  	/* Checkpoint tree */
>  	struct rb_root		ns_cptree;
> -	spinlock_t		ns_cptree_lock;
> +	struct mutex		ns_cptree_lock; /* Protects
> ns_cptree */

I assume that you prefer mutex because of exclusive lock nature. Could
we improve something for the case of rw_semaphore?

Thanks,
Slava.

>  
>  	/* Dirty inode list */
>  	struct list_head	ns_dirty_files;