mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Jeff Layton <jlayton@kernel.org>
To: NeilBrown <neilb@suse.de>,
	Alexander Viro <viro@zeniv.linux.org.uk>,
	 Christian Brauner <brauner@kernel.org>, Jan Kara <jack@suse.cz>,
	Linus Torvalds <torvalds@linux-foundation.org>,
	Dave Chinner <david@fromorbit.com>
Cc: linux-fsdevel@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: Re: [PATCH 03/19] VFS: use d_alloc_parallel() in lookup_one_qstr_excl() and rename it.
Date: Thu, 06 Feb 2025 09:30:47 -0500	[thread overview]
Message-ID: <017ca787f3a167302281b65e60d301d9f1c0f5de.camel@kernel.org> (raw)
In-Reply-To: <20250206054504.2950516-4-neilb@suse.de>

On Thu, 2025-02-06 at 16:42 +1100, NeilBrown wrote:
> lookup_one_qstr_excl() is used for lookups prior to directory
> modifications, whether create, unlink, rename, or whatever.
> 
> To prepare for allowing modification to happen in parallel, change
> lookup_one_qstr_excl() to use d_alloc_parallel().
> 
> To reflect this, name is changed to lookup_one_qtr() - as the directory
> may be locked shared.
> 
> If any for the "intent" LOOKUP flags are passed, the caller must ensure
> d_lookup_done() is called at an appropriate time.  If none are passed
> then we can be sure ->lookup() will do a real lookup and d_lookup_done()
> is called internally.
> 
> Signed-off-by: NeilBrown <neilb@suse.de>
> ---
>  fs/namei.c            | 47 +++++++++++++++++++++++++------------------
>  fs/smb/server/vfs.c   |  7 ++++---
>  include/linux/namei.h |  9 ++++++---
>  3 files changed, 37 insertions(+), 26 deletions(-)
> 
> diff --git a/fs/namei.c b/fs/namei.c
> index 5cdbd2eb4056..d684102d873d 100644
> --- a/fs/namei.c
> +++ b/fs/namei.c
> @@ -1665,15 +1665,13 @@ static struct dentry *lookup_dcache(const struct qstr *name,
>  }
>  
>  /*
> - * Parent directory has inode locked exclusive.  This is one
> - * and only case when ->lookup() gets called on non in-lookup
> - * dentries - as the matter of fact, this only gets called
> - * when directory is guaranteed to have no in-lookup children
> - * at all.
> + * Parent directory has inode locked: exclusive or shared.
> + * If @flags contains any LOOKUP_INTENT_FLAGS then d_lookup_done()
> + * must be called after the intended operation is performed - or aborted.
>   */
> -struct dentry *lookup_one_qstr_excl(const struct qstr *name,
> -				    struct dentry *base,
> -				    unsigned int flags)
> +struct dentry *lookup_one_qstr(const struct qstr *name,
> +			       struct dentry *base,
> +			       unsigned int flags)
>  {
>  	struct dentry *dentry = lookup_dcache(name, base, flags);
>  	struct dentry *old;
> @@ -1686,18 +1684,25 @@ struct dentry *lookup_one_qstr_excl(const struct qstr *name,
>  	if (unlikely(IS_DEADDIR(dir)))
>  		return ERR_PTR(-ENOENT);
>  
> -	dentry = d_alloc(base, name);
> -	if (unlikely(!dentry))
> +	dentry = d_alloc_parallel(base, name);
> +	if (unlikely(IS_ERR_OR_NULL(dentry)))
>  		return ERR_PTR(-ENOMEM);
> +	if (!d_in_lookup(dentry))
> +		/* Raced with another thread which did the lookup */
> +		return dentry;
>  
>  	old = dir->i_op->lookup(dir, dentry, flags);
>  	if (unlikely(old)) {
> +		d_lookup_done(dentry);
>  		dput(dentry);
>  		dentry = old;
>  	}
> +	if ((flags & LOOKUP_INTENT_FLAGS) == 0)
> +		/* ->lookup must have given final answer */
> +		d_lookup_done(dentry);

This is kind of an ugly thing for the callers to get right. I think it
would be cleaner to just push the d_lookup_done() into all of the
callers that don't pass any intent flags, and do away with this.

>  	return dentry;
>  }
> -EXPORT_SYMBOL(lookup_one_qstr_excl);
> +EXPORT_SYMBOL(lookup_one_qstr);
>  
>  /**
>   * lookup_fast - do fast lockless (but racy) lookup of a dentry
> @@ -2739,7 +2744,7 @@ static struct dentry *__kern_path_locked(int dfd, struct filename *name, struct
>  		return ERR_PTR(-EINVAL);
>  	}
>  	inode_lock_nested(path->dentry->d_inode, I_MUTEX_PARENT);
> -	d = lookup_one_qstr_excl(&last, path->dentry, 0);
> +	d = lookup_one_qstr(&last, path->dentry, 0);
>  	if (IS_ERR(d)) {
>  		inode_unlock(path->dentry->d_inode);
>  		path_put(path);
> @@ -4078,8 +4083,8 @@ static struct dentry *filename_create(int dfd, struct filename *name,
>  	if (last.name[last.len] && !want_dir)
>  		create_flags = 0;
>  	inode_lock_nested(path->dentry->d_inode, I_MUTEX_PARENT);
> -	dentry = lookup_one_qstr_excl(&last, path->dentry,
> -				      reval_flag | create_flags);
> +	dentry = lookup_one_qstr(&last, path->dentry,
> +				 reval_flag | create_flags);
>  	if (IS_ERR(dentry))
>  		goto unlock;
>  
> @@ -4103,6 +4108,7 @@ static struct dentry *filename_create(int dfd, struct filename *name,
>  	}
>  	return dentry;
>  fail:
> +	d_lookup_done(dentry);
>  	dput(dentry);
>  	dentry = ERR_PTR(error);
>  unlock:
> @@ -4508,7 +4514,7 @@ int do_rmdir(int dfd, struct filename *name)
>  		goto exit2;
>  
>  	inode_lock_nested(path.dentry->d_inode, I_MUTEX_PARENT);
> -	dentry = lookup_one_qstr_excl(&last, path.dentry, lookup_flags);
> +	dentry = lookup_one_qstr(&last, path.dentry, lookup_flags);
>  	error = PTR_ERR(dentry);
>  	if (IS_ERR(dentry))
>  		goto exit3;
> @@ -4641,7 +4647,7 @@ int do_unlinkat(int dfd, struct filename *name)
>  		goto exit2;
>  retry_deleg:
>  	inode_lock_nested(path.dentry->d_inode, I_MUTEX_PARENT);
> -	dentry = lookup_one_qstr_excl(&last, path.dentry, lookup_flags);
> +	dentry = lookup_one_qstr(&last, path.dentry, lookup_flags);
>  	error = PTR_ERR(dentry);
>  	if (!IS_ERR(dentry)) {
>  
> @@ -5231,8 +5237,8 @@ int do_renameat2(int olddfd, struct filename *from, int newdfd,
>  		goto exit_lock_rename;
>  	}
>  
> -	old_dentry = lookup_one_qstr_excl(&old_last, old_path.dentry,
> -					  lookup_flags);
> +	old_dentry = lookup_one_qstr(&old_last, old_path.dentry,
> +				     lookup_flags);
>  	error = PTR_ERR(old_dentry);
>  	if (IS_ERR(old_dentry))
>  		goto exit3;
> @@ -5240,8 +5246,8 @@ int do_renameat2(int olddfd, struct filename *from, int newdfd,
>  	error = -ENOENT;
>  	if (d_is_negative(old_dentry))
>  		goto exit4;
> -	new_dentry = lookup_one_qstr_excl(&new_last, new_path.dentry,
> -					  lookup_flags | target_flags);
> +	new_dentry = lookup_one_qstr(&new_last, new_path.dentry,
> +				     lookup_flags | target_flags);
>  	error = PTR_ERR(new_dentry);
>  	if (IS_ERR(new_dentry))
>  		goto exit4;
> @@ -5292,6 +5298,7 @@ int do_renameat2(int olddfd, struct filename *from, int newdfd,
>  	rd.flags	   = flags;
>  	error = vfs_rename(&rd);
>  exit5:
> +	d_lookup_done(new_dentry);
>  	dput(new_dentry);
>  exit4:
>  	dput(old_dentry);
> diff --git a/fs/smb/server/vfs.c b/fs/smb/server/vfs.c
> index 4e580bb7baf8..89b3823f6405 100644
> --- a/fs/smb/server/vfs.c
> +++ b/fs/smb/server/vfs.c
> @@ -109,7 +109,7 @@ static int ksmbd_vfs_path_lookup_locked(struct ksmbd_share_config *share_conf,
>  	}
>  
>  	inode_lock_nested(parent_path->dentry->d_inode, I_MUTEX_PARENT);
> -	d = lookup_one_qstr_excl(&last, parent_path->dentry, 0);
> +	d = lookup_one_qstr(&last, parent_path->dentry, 0);
>  	if (IS_ERR(d))
>  		goto err_out;
>  
> @@ -726,8 +726,8 @@ int ksmbd_vfs_rename(struct ksmbd_work *work, const struct path *old_path,
>  		ksmbd_fd_put(work, parent_fp);
>  	}
>  
> -	new_dentry = lookup_one_qstr_excl(&new_last, new_path.dentry,
> -					  lookup_flags | LOOKUP_RENAME_TARGET);
> +	new_dentry = lookup_one_qstr(&new_last, new_path.dentry,
> +				     lookup_flags | LOOKUP_RENAME_TARGET);
>  	if (IS_ERR(new_dentry)) {
>  		err = PTR_ERR(new_dentry);
>  		goto out3;
> @@ -771,6 +771,7 @@ int ksmbd_vfs_rename(struct ksmbd_work *work, const struct path *old_path,
>  		ksmbd_debug(VFS, "vfs_rename failed err %d\n", err);
>  
>  out4:
> +	d_lookup_done(new_dentry);
>  	dput(new_dentry);
>  out3:
>  	dput(old_parent);
> diff --git a/include/linux/namei.h b/include/linux/namei.h
> index 8ec8fed3bce8..06bb3ea65beb 100644
> --- a/include/linux/namei.h
> +++ b/include/linux/namei.h
> @@ -34,6 +34,9 @@ enum {LAST_NORM, LAST_ROOT, LAST_DOT, LAST_DOTDOT};
>  #define LOOKUP_EXCL		0x0400	/* ... in exclusive creation */
>  #define LOOKUP_RENAME_TARGET	0x0800	/* ... in destination of rename() */
>  
> +#define LOOKUP_INTENT_FLAGS	(LOOKUP_OPEN | LOOKUP_CREATE | LOOKUP_EXCL |	\
> +				 LOOKUP_RENAME_TARGET)
> +
>  /* internal use only */
>  #define LOOKUP_PARENT		0x0010
>  
> @@ -52,9 +55,9 @@ extern int path_pts(struct path *path);
>  
>  extern int user_path_at(int, const char __user *, unsigned, struct path *);
>  
> -struct dentry *lookup_one_qstr_excl(const struct qstr *name,
> -				    struct dentry *base,
> -				    unsigned int flags);
> +struct dentry *lookup_one_qstr(const struct qstr *name,
> +			       struct dentry *base,
> +			       unsigned int flags);
>  extern int kern_path(const char *, unsigned, struct path *);
>  
>  extern struct dentry *kern_path_create(int, const char *, struct path *, unsigned int);

-- 
Jeff Layton <jlayton@kernel.org>

  reply	other threads:[~2025-02-06 14:30 UTC|newest]

Thread overview: 83+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2025-02-06  5:42 [PATCH 00/19 v7?] RFC: Allow concurrent and async changes in a directory NeilBrown
2025-02-06  5:42 ` [PATCH 01/19] VFS: introduce vfs_mkdir_return() NeilBrown
2025-02-06 12:24   ` Christian Brauner
2025-02-06 23:52     ` NeilBrown
2025-02-06 13:52   ` Jeff Layton
2025-02-06 23:57     ` NeilBrown
2025-02-07 19:45   ` Al Viro
2025-02-10  4:36     ` NeilBrown
2025-02-06  5:42 ` [PATCH 02/19] VFS: use global wait-queue table for d_alloc_parallel() NeilBrown
2025-02-07 19:32   ` Al Viro
2025-02-10  4:58     ` NeilBrown
2025-02-10  5:15       ` Al Viro
2025-02-11 23:35         ` NeilBrown
2025-02-12  0:25           ` Al Viro
2025-02-12  1:46             ` NeilBrown
2025-02-06  5:42 ` [PATCH 03/19] VFS: use d_alloc_parallel() in lookup_one_qstr_excl() and rename it NeilBrown
2025-02-06 14:30   ` Jeff Layton [this message]
2025-02-07  0:04     ` NeilBrown
2025-02-07  0:23       ` Jeff Layton
2025-02-07 20:01   ` Al Viro
2025-02-06  5:42 ` [PATCH 04/19] VFS: change kern_path_locked() and user_path_locked_at() to never return negative dentry NeilBrown
2025-02-06 12:31   ` Christian Brauner
2025-02-06 13:09     ` Christian Brauner
2025-02-07  0:08       ` NeilBrown
2025-02-06  5:42 ` [PATCH 05/19] VFS: add common error checks to lookup_one_qstr() NeilBrown
2025-02-06 12:33   ` Christian Brauner
2025-02-07 20:14   ` Al Viro
2025-02-09 20:23   ` Al Viro
2025-02-06  5:42 ` [PATCH 06/19] VFS: repack DENTRY_ flags NeilBrown
2025-02-06 12:34   ` (subset) " Christian Brauner
2025-02-06  5:42 ` [PATCH 07/19] VFS: repack LOOKUP_ bit flags NeilBrown
2025-02-06 12:44   ` Christian Brauner
2025-02-07  0:24     ` NeilBrown
2025-02-06 12:54   ` (subset) " Christian Brauner
2025-02-06  5:42 ` [PATCH 08/19] VFS: introduce lookup_and_lock() and friends NeilBrown
2025-02-06 13:49   ` Christian Brauner
2025-02-07  1:28     ` NeilBrown
2025-02-07 20:22   ` Al Viro
2025-02-08 23:18     ` Al Viro
2025-02-12  5:22       ` NeilBrown
2025-02-12 15:51         ` Al Viro
2025-02-12 20:11           ` Al Viro
2025-02-12  4:49     ` NeilBrown
2025-02-06  5:42 ` [PATCH 09/19] VFS: add _async versions of the various directory modifying inode_operations NeilBrown
2025-02-06 13:15   ` Christian Brauner
2025-02-07  1:46     ` NeilBrown
2025-02-07 22:41   ` Al Viro
2025-02-09  1:09     ` Al Viro
2025-02-09  4:57       ` Al Viro
2025-02-06  5:42 ` [PATCH 10/19] VFS: introduce inode flags to report locking needs for directory ops NeilBrown
2025-02-06 13:22   ` Christian Brauner
2025-02-07  2:01     ` NeilBrown
2025-02-06  5:42 ` [PATCH 11/19] VFS: Add ability to exclusively lock a dentry and use for create/remove operations NeilBrown
2025-02-08  1:38   ` Al Viro
2025-02-09  6:40   ` Al Viro
2025-02-06  5:42 ` [PATCH 12/19] VFS: enhance d_splice_alias to accommodate shared-lock updates NeilBrown
2025-02-06  5:42 ` [PATCH 13/19] VFS: lock dentry for ->revalidate to avoid races with rename etc NeilBrown
2025-02-07 20:28   ` Al Viro
2025-02-07 20:35     ` Al Viro
2025-02-08  1:30   ` Al Viro
2025-02-08  1:35     ` Al Viro
2025-02-12 21:22     ` Al Viro
2025-02-06  5:42 ` [PATCH 14/19] VFS: Ensure no async updates happening in directory being removed NeilBrown
2025-02-06 14:06   ` Christian Brauner
2025-02-07  2:17     ` NeilBrown
2025-02-07 21:06   ` Al Viro
2025-02-08 22:06     ` Al Viro
2025-02-08 22:30       ` Linus Torvalds
2025-02-08 22:34         ` Linus Torvalds
2025-02-08 23:25         ` Al Viro
2025-02-06  5:42 ` [PATCH 15/19] VFS: Change lookup_and_lock() to use shared lock when possible NeilBrown
2025-02-06  5:42 ` [PATCH 16/19] VFS: add lookup_and_lock_rename() NeilBrown
2025-02-07 21:21   ` Al Viro
2025-02-06  5:42 ` [PATCH 17/19] nfsd: use lookup_and_lock_one() and lookup_and_lock_rename_one() NeilBrown
2025-02-06  5:42 ` [PATCH 18/19] nfs: change mkdir inode_operation to mkdir_async NeilBrown
2025-02-06  5:42 ` [PATCH 19/19] nfs: switch to _async for all directory ops NeilBrown
2025-02-13  3:51   ` Al Viro
2025-02-13  4:09     ` Al Viro
2025-02-13 18:01       ` Al Viro
2025-02-06 14:36 ` [PATCH 00/19 v7?] RFC: Allow concurrent and async changes in a directory Christian Brauner
2025-02-06 15:36 ` John Stoffel
2025-02-07  2:18   ` NeilBrown
2025-02-09 23:33 ` Al Viro

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=017ca787f3a167302281b65e60d301d9f1c0f5de.camel@kernel.org \
    --to=jlayton@kernel.org \
    --cc=brauner@kernel.org \
    --cc=david@fromorbit.com \
    --cc=jack@suse.cz \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=neilb@suse.de \
    --cc=torvalds@linux-foundation.org \
    --cc=viro@zeniv.linux.org.uk \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®