Re: [f2fs-dev] [PATCH v2 08/14] f2fs: optimize small block size large folio read
From: Daeho Jeong
Date: Wed Sep 16 2026 - 00:33:40 EST
On Mon, Sep 14, 2026 at 9:21 PM Nanzhe Zhao via Linux-f2fs-devel
<linux-f2fs-devel@xxxxxxxxxxxxxxxxxxxxx> wrote:
>
> The original f2fs_read_data_large_folio() implementation has limited
> benefit with a 4KB block size, mainly because updating
> read_pages_pending greatly increases the number of spinlock
> operations.
>
> Use len_blks to batch read_pages_pending and iostat updates for
> contiguous mapped blocks. If the contiguous mapping covers the whole
> folio, skip f2fs_folio_state allocation for that folio.
>
> Signed-off-by: Nanzhe Zhao <zhaonanzhe@xxxxxxxxxx>
> ---
> fs/f2fs/data.c | 65 ++++++++++++++++++++++++++++++++++----------------
> 1 file changed, 45 insertions(+), 20 deletions(-)
>
> diff --git a/fs/f2fs/data.c b/fs/f2fs/data.c
> index 287d83debf95..09ad5cf0d9aa 100644
> --- a/fs/f2fs/data.c
> +++ b/fs/f2fs/data.c
> @@ -173,8 +173,9 @@ static void f2fs_finish_read_bio(struct bio *bio, bool in_task)
> continue;
> }
>
> - if (folio_test_large(folio)) {
> - struct f2fs_folio_state *ffs = folio->private;
> + if (f2fs_folio_has_ffs(folio)) {
> + struct f2fs_folio_state *ffs =
> + (struct f2fs_folio_state *)folio->private;
>
> spin_lock_irqsave(&ffs->state_lock, flags);
> ffs->read_pages_pending -= nr_pages;
> @@ -2844,8 +2845,7 @@ static int f2fs_read_data_large_folio(struct inode *inode,
> pgoff_t index, offset, next_pgofs = 0;
> unsigned max_nr_pages = rac ? readahead_count(rac) :
> folio_nr_pages(folio);
> - unsigned int nrpages;
> - struct f2fs_folio_state *ffs;
> + unsigned int nrpages, len_blks;
> int ret = 0;
> bool folio_in_bio = false;
>
> @@ -2868,11 +2868,17 @@ static int f2fs_read_data_large_folio(struct inode *inode,
> folio_in_bio = false;
> index = folio->index;
> offset = 0;
> - ffs = NULL;
> nrpages = folio_nr_pages(folio);
>
> - for (; nrpages; nrpages--, max_nr_pages--, index++, offset++) {
> + for (; nrpages;
> + nrpages -= len_blks, max_nr_pages -= len_blks,
> + index += len_blks, offset += len_blks) {
> sector_t block_nr;
> + bool whole_folio_in_bio;
> + unsigned int i;
> +
> + len_blks = 1;
> +
> /*
> * Map blocks using the previous result first.
> */
> @@ -2901,13 +2907,31 @@ static int f2fs_read_data_large_folio(struct inode *inode,
> got_it:
> if ((map.m_flags & F2FS_MAP_MAPPED)) {
> block_nr = map.m_pblk + index - map.m_lblk;
> - if (!f2fs_is_valid_blkaddr(F2FS_I_SB(inode), block_nr,
> +
> + len_blks = min_t(unsigned int, nrpages, max_nr_pages);
> + len_blks = min_t(unsigned int, len_blks,
> + (unsigned int)(map.m_lblk + map.m_len - index));
> +
> + for (i = 0; i < len_blks; i++) {
> + if (!f2fs_is_valid_blkaddr(F2FS_I_SB(inode),
> + block_nr + i,
> DATA_GENERIC_ENHANCE_READ)) {
> - ret = -EFSCORRUPTED;
> - goto err_out;
> + ret = -EFSCORRUPTED;
> + goto err_out;
> + }
> }
> +
> + /*
> + * If an entire folio is added to one bio,
> + * folio_end_read() can complete the folio read status
> + * without relying on f2fs_folio_state.
> + */
> + whole_folio_in_bio = offset == 0 &&
> + len_blks == folio_nr_pages(folio);
> +
> } else {
> size_t page_offset = offset << PAGE_SHIFT;
> +
> folio_zero_range(folio, page_offset, PAGE_SIZE);
> if (vi && !fsverity_verify_blocks(vi, folio, PAGE_SIZE, page_offset)) {
> ret = -EIO;
> @@ -2917,15 +2941,13 @@ static int f2fs_read_data_large_folio(struct inode *inode,
> }
>
> /* We must increment read_pages_pending before possible BIOs submitting
> - * to prevent from premature folio_end_read() call on folio
> + * to prevent from premature folio_end_read() call on folio.
> */
> - if (folio_test_large(folio)) {
> - ffs = f2fs_ffs_find_or_alloc(folio);
> + if (folio_test_large(folio) && !whole_folio_in_bio) {
Hi Nanzhe,
I found a hang issue caused by the whole_folio_in_bio optimization
during my local 16KB page-4KB block testing:
Plz, check the whole_folio_in_bio optimization is safe for the below conditions.
1. read_blocks_pending underflow: If a folio already has ffs attached
(from previous writes/holes/reads), whole_folio_in_bio skips
incrementing read_blocks_pending.
When the BIO completes, f2fs_finish_read_bio() still decrements it,
underflowing the counter.
2. BIO split risk: If a large folio BIO gets split by the block/crypto
layer, f2fs_finish_read_bio() would call folio_end_read() prematurely
on the first completing piece.
Thanks,
> + f2fs_ffs_find_or_alloc(folio);
>
> /* set the bitmap to wait */
> - spin_lock_irq(&ffs->state_lock);
> - ffs->read_pages_pending++;
> - spin_unlock_irq(&ffs->state_lock);
> + f2fs_update_read_folio_pending(folio, len_blks);
> }
>
> /*
> @@ -2949,17 +2971,20 @@ static int f2fs_read_data_large_folio(struct inode *inode,
> * If the page is under writeback, we need to wait for
> * its completion to see the correct decrypted data.
> */
> - f2fs_wait_on_block_writeback(inode, block_nr);
> + for (i = 0; i < len_blks; i++)
> + f2fs_wait_on_block_writeback(inode, block_nr + i);
>
> - if (!bio_add_folio(bio, folio, F2FS_BLKSIZE(F2FS_I_SB(inode)),
> + if (!bio_add_folio(bio, folio,
> + len_blks * F2FS_BLKSIZE(F2FS_I_SB(inode)),
> offset << PAGE_SHIFT))
> goto submit_and_realloc;
>
> folio_in_bio = true;
> - inc_page_count(F2FS_I_SB(inode), F2FS_RD_DATA);
> + for (i = 0; i < len_blks; i++)
> + inc_page_count(F2FS_I_SB(inode), F2FS_RD_DATA);
> f2fs_update_iostat(F2FS_I_SB(inode), NULL, FS_DATA_READ_IO,
> - F2FS_BLKSIZE(F2FS_I_SB(inode)));
> - last_block_in_bio = block_nr;
> + len_blks * F2FS_BLKSIZE(F2FS_I_SB(inode)));
> + last_block_in_bio = block_nr + len_blks - 1;
> }
> trace_f2fs_read_folio(folio, DATA);
> err_out:
> --
> 2.43.0
>
>
>
> _______________________________________________
> Linux-f2fs-devel mailing list
> Linux-f2fs-devel@xxxxxxxxxxxxxxxxxxxxx
> https://lists.sourceforge.net/lists/listinfo/linux-f2fs-devel