mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Baolin Wang <baolin.wang@linux.alibaba.com>
To: Chi Zhiling <chizhiling@163.com>,
	linux-fsdevel@vger.kernel.org, linux-mm@kvack.org,
	linux-kernel@vger.kernel.org
Cc: "Matthew Wilcox (Oracle)" <willy@infradead.org>,
	Jan Kara <jack@suse.cz>,
	Andrew Morton <akpm@linux-foundation.org>,
	Hugh Dickins <hughd@google.com>,
	Chi Zhiling <chizhiling@kylinos.cn>
Subject: Re: [PATCH v2 5/5] mm/shmem: optimize file read with folio batching
Date: Tue, 2 Jun 2026 13:54:09 +0800	[thread overview]
Message-ID: <6af198e3-89aa-4d31-b596-3a1b623513fa@linux.alibaba.com> (raw)
In-Reply-To: <20260601055704.167436-6-chizhiling@163.com>



On 6/1/26 1:57 PM, Chi Zhiling wrote:
> From: Chi Zhiling <chizhiling@kylinos.cn>
> 
> Optimize shmem file read by using filemap_get_folios_contig() to batch
> fetch contiguous folios from the page cache, reducing the overhead of
> repeated shmem_get_folio() calls.
> 
> This patch checks the uptodate flag without holding the folio lock, so
> it may observe a non-uptodate state on a locked folio that is still
> being initialized. This is safe because only zero-filled data can be
> copied to the user buffer in that scenario.
> 
> A non-uptodate folio in the swap cache cannot be added to the shmem page
> cache. This creates a semantic conflict, as shmem zeroes the folio out,
> but the swap cache would fill it by reading from the swap backing store.
> 
> Signed-off-by: Chi Zhiling <chizhiling@kylinos.cn>
> ---
>   mm/shmem.c | 57 ++++++++++++++++++++++++++++++++++++++++--------------
>   1 file changed, 42 insertions(+), 15 deletions(-)
> 
> diff --git a/mm/shmem.c b/mm/shmem.c
> index cac355685e49..61937582f08c 100644
> --- a/mm/shmem.c
> +++ b/mm/shmem.c
> @@ -891,6 +891,14 @@ int shmem_add_to_page_cache(struct folio *folio,
>   	VM_BUG_ON_FOLIO(!folio_test_locked(folio), folio);
>   	VM_BUG_ON_FOLIO(!folio_test_swapbacked(folio), folio);
>   
> +	/*
> +	 * Don't add a non-uptodate folio that is in swap cache to page
> +	 * cache, since shmem will zero it instead of reading from swap
> +	 * backing.
> +	 */
> +	VM_BUG_ON_FOLIO(folio_test_swapcache(folio) &&
> +			!folio_test_uptodate(folio), folio);

It's impossible for a folio to be in both the swap cache and the shmem 
page cache. We can drop this.

>   	folio_ref_add(folio, nr);
>   	folio->mapping = mapping;
>   	folio->index = index;
> @@ -3382,11 +3390,13 @@ static ssize_t shmem_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   	struct file *file = iocb->ki_filp;
>   	struct inode *inode = file_inode(file);
>   	struct address_space *mapping = inode->i_mapping;
> -	pgoff_t index;
> +	struct folio_batch fbatch;
>   	unsigned long offset;
>   	int error = 0;
>   	ssize_t retval = 0;
>   
> +	folio_batch_init(&fbatch);
> +
>   	for (;;) {
>   		struct folio *folio = NULL;
>   		unsigned long nr, ret;
> @@ -3395,15 +3405,33 @@ static ssize_t shmem_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   
>   		if (unlikely(iocb->ki_pos >= i_size))
>   			break;
> +fetch:
> +		folio = folio_batch_next(&fbatch);
> +		if (!folio) {
> +			pgoff_t start = iocb->ki_pos >> PAGE_SHIFT;
> +			pgoff_t end = (iocb->ki_pos + to->count - 1) >> PAGE_SHIFT;

You should consider the inode size when calculating the 'end'. You can 
reuse the 'end_offset':

end_offset = min_t(loff_t, i_size, iocb->ki_pos + to->count);

then pass 'end_offset - 1' to filemap_get_folios_contig().

> +
> +			if (folio_batch_count(&fbatch)) {
> +				for (int i = 0; i < folio_batch_count(&fbatch); i++)
> +					folio_put(fbatch.folios[i]);
> +				folio_batch_reinit(&fbatch);
> +			}
>   
> -		index = iocb->ki_pos >> PAGE_SHIFT;
> -		error = shmem_get_folio(inode, index, 0, &folio, SGP_READ);
> -		if (folio)
> -			folio_unlock(folio);
> -		if (error) {
> -			if (error == -EINVAL)
> -				error = 0;
> -			break;
> +			filemap_get_folios_contig(inode->i_mapping, &start, end, &fbatch);
> +			if (folio_batch_count(&fbatch))
> +				goto fetch;
> +
> +			error = shmem_get_folio(inode, start, 0, &folio, SGP_READ);
> +			if (unlikely(error)) {
> +				if (error == -EINVAL)
> +					error = 0;
> +				break;
> +			}
> +			if (folio) {
> +				folio_unlock(folio);
> +				folio_batch_add(&fbatch, folio);
> +				fbatch.i++;
> +			}
>   		}
>   
>   		/*
> @@ -3411,17 +3439,15 @@ static ssize_t shmem_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   		 * are called without i_rwsem protection against truncate
>   		 */
>   		i_size = i_size_read(inode);
> -		if (unlikely(iocb->ki_pos >= i_size)) {
> -			if (folio)
> -				folio_put(folio);
> +		if (unlikely(iocb->ki_pos >= i_size))
>   			break;
> -		}
> +
>   		fsize = folio ? folio_size(folio) : PAGE_SIZE;
>   		offset = iocb->ki_pos & (fsize - 1);
>   		end_offset = min_t(loff_t, i_size, iocb->ki_pos + to->count);
>   		nr = min_t(loff_t, end_offset - iocb->ki_pos, fsize - offset);
>   
> -		if (folio) {
> +		if (folio && folio_test_uptodate(folio)) {
>   			/*
>   			 * If users can be writing to this page using arbitrary
>   			 * virtual addresses, take care about potential aliasing
> @@ -3443,7 +3469,6 @@ static ssize_t shmem_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   				ret = copy_folio_to_iter(folio, offset, nr, to);
>   			else
>   				ret = copy_pages_to_iter(folio, offset, nr, to, &error);
> -			folio_put(folio);
>   		} else if (user_backed_iter(to)) {
>   			/*
>   			 * Copy to user tends to be so well optimized, but
> @@ -3474,6 +3499,8 @@ static ssize_t shmem_file_read_iter(struct kiocb *iocb, struct iov_iter *to)
>   		cond_resched();
>   	}
>   
> +	for (int i = 0; i < folio_batch_count(&fbatch); i++)
> +		folio_put(fbatch.folios[i]);
>   	file_accessed(file);
>   	return retval ? retval : error;
>   }


  parent reply	other threads:[~2026-06-02  5:54 UTC|newest]

Thread overview: 24+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-06-01  5:56 [PATCH v2 0/5] mm/shmem: optimize read with reduced xarray lookups and " Chi Zhiling
2026-06-01  5:57 ` [PATCH v2 1/5] mm/filemap: reduce unnecessary xarray lookups when read cached pages Chi Zhiling
2026-06-01 12:51   ` Matthew Wilcox
2026-06-01 14:39     ` Chi Zhiling
2026-06-01  5:57 ` [PATCH v2 2/5] mm/filemap: reduce xarray lookups in filemap_get_folios_contig() Chi Zhiling
2026-06-01  5:57 ` [PATCH v2 3/5] mm/shmem: introduce copy_zero_to_iter() for large zeroing Chi Zhiling
2026-06-01 13:22   ` Matthew Wilcox
2026-06-01 14:57     ` Chi Zhiling
2026-06-01 15:02     ` Mateusz Guzik
2026-06-01 15:11       ` Mateusz Guzik
2026-06-01 15:13       ` Matthew Wilcox
2026-06-01 15:40         ` Mateusz Guzik
2026-06-01 22:03       ` David Laight
2026-06-01  5:57 ` [PATCH v2 4/5] mm/shmem: remove page-copy fallback in shmem read path Chi Zhiling
2026-06-02  5:39   ` Baolin Wang
2026-06-02  6:17     ` Chi Zhiling
2026-06-01  5:57 ` [PATCH v2 5/5] mm/shmem: optimize file read with folio batching Chi Zhiling
2026-06-01 13:59   ` Matthew Wilcox
2026-06-01 15:00     ` Chi Zhiling
2026-06-02  5:54   ` Baolin Wang [this message]
2026-06-02  6:58     ` Chi Zhiling
2026-06-01  7:16 ` [PATCH v2 0/5] mm/shmem: optimize read with reduced xarray lookups and " Chi Zhiling
2026-06-02  0:43 ` Andrew Morton
2026-06-02  2:44   ` Chi Zhiling

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=6af198e3-89aa-4d31-b596-3a1b623513fa@linux.alibaba.com \
    --to=baolin.wang@linux.alibaba.com \
    --cc=akpm@linux-foundation.org \
    --cc=chizhiling@163.com \
    --cc=chizhiling@kylinos.cn \
    --cc=hughd@google.com \
    --cc=jack@suse.cz \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=willy@infradead.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

Powered by JetHome