* [PATCH 1/5] f2fs: cache: introduce metadata_cache sysfs node
@ 2026-10-07 11:49 Chao Yu
2026-10-07 11:49 ` [PATCH 2/5] f2fs: cache: shrink meta and node caches in f2fs_balance_fs_bg Chao Yu
` (3 more replies)
0 siblings, 4 replies; 5+ messages in thread
From: Chao Yu @ 2026-10-07 11:49 UTC (permalink / raw)
To: jaegeuk; +Cc: linux-f2fs-devel, linux-kernel, Chao Yu
From: Chao Yu <chao@kernel.org>
This patch exposes a new sysfs node 'metadata_cache' under
/sys/fs/f2fs/features/ to indicate runtime support for the in-kernel
metadata cache infrastructure.
Userspace can check this feature entry to determine support before
configuring metadata cache knobs (e.g., cache_wb_interval) or parsing
per-cache memory statistics in debugfs.
Signed-off-by: Chao Yu <chao@kernel.org>
---
Documentation/ABI/testing/sysfs-fs-f2fs | 2 +-
fs/f2fs/sysfs.c | 2 ++
2 files changed, 3 insertions(+), 1 deletion(-)
diff --git a/Documentation/ABI/testing/sysfs-fs-f2fs b/Documentation/ABI/testing/sysfs-fs-f2fs
index c4746c416ac2..5b196806b3bc 100644
--- a/Documentation/ABI/testing/sysfs-fs-f2fs
+++ b/Documentation/ABI/testing/sysfs-fs-f2fs
@@ -271,7 +271,7 @@ Description: Shows all enabled kernel features.
inode_crtime, lost_found, verity, sb_checksum,
casefold, readonly, compression, test_dummy_encryption_v2,
atomic_write, pin_file, encrypted_casefold, linear_lookup,
- fserror.
+ fserror, metadata_cache.
What: /sys/fs/f2fs/<disk>/inject_rate
Date: May 2016
diff --git a/fs/f2fs/sysfs.c b/fs/f2fs/sysfs.c
index 9749da70089a..afce74744d4d 100644
--- a/fs/f2fs/sysfs.c
+++ b/fs/f2fs/sysfs.c
@@ -1421,6 +1421,7 @@ F2FS_FEATURE_RO_ATTR(linear_lookup);
#endif
F2FS_FEATURE_RO_ATTR(packed_ssa);
F2FS_FEATURE_RO_ATTR(fserror);
+F2FS_FEATURE_RO_ATTR(metadata_cache);
#define ATTR_LIST(name) (&f2fs_attr_##name.attr)
static struct attribute *f2fs_attrs[] = {
@@ -1592,6 +1593,7 @@ static struct attribute *f2fs_feat_attrs[] = {
#endif
BASE_ATTR_LIST(packed_ssa),
BASE_ATTR_LIST(fserror),
+ BASE_ATTR_LIST(metadata_cache),
NULL,
};
ATTRIBUTE_GROUPS(f2fs_feat);
--
2.49.0
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH 2/5] f2fs: cache: shrink meta and node caches in f2fs_balance_fs_bg
2026-10-07 11:49 [PATCH 1/5] f2fs: cache: introduce metadata_cache sysfs node Chao Yu
@ 2026-10-07 11:49 ` Chao Yu
2026-10-07 11:49 ` [PATCH 3/5] f2fs: cache: wake up f2fs_writeback when exceeding threshold Chao Yu
` (2 subsequent siblings)
3 siblings, 0 replies; 5+ messages in thread
From: Chao Yu @ 2026-10-07 11:49 UTC (permalink / raw)
To: jaegeuk; +Cc: linux-f2fs-devel, linux-kernel, Chao Yu
From: Chao Yu <chao@kernel.org>
This patch introduces META_BLOCK and NODE_BLOCK memory types to monitor
meta and node cache usage against available free memory, and integrates
them into f2fs_balance_fs_bg().
In f2fs_balance_fs_bg(), shrink 128 clean cached blocks of each types when
memory is insufficient (cache memory usage is larger than 50% of low memory).
Signed-off-by: Chao Yu <chao@kernel.org>
---
fs/f2fs/cache.c | 12 ++++++++++++
fs/f2fs/cache.h | 6 ++++++
fs/f2fs/node.c | 10 ++++++++++
fs/f2fs/node.h | 2 ++
fs/f2fs/segment.c | 8 ++++++++
5 files changed, 38 insertions(+)
diff --git a/fs/f2fs/cache.c b/fs/f2fs/cache.c
index 04b5408cbc58..f385fe928b5b 100644
--- a/fs/f2fs/cache.c
+++ b/fs/f2fs/cache.c
@@ -669,6 +669,18 @@ unsigned long f2fs_shrink_cache(struct f2fs_sb_info *sbi,
return freed;
}
+unsigned long f2fs_shrink_meta_cache(struct f2fs_sb_info *sbi,
+ unsigned long nr_to_scan)
+{
+ return f2fs_do_shrink_cache(META_CACHE(sbi), nr_to_scan);
+}
+
+unsigned long f2fs_shrink_node_cache(struct f2fs_sb_info *sbi,
+ unsigned long nr_to_scan)
+{
+ return f2fs_do_shrink_cache(NODE_CACHE(sbi), nr_to_scan);
+}
+
static int f2fs_cache_writeback_kthread(void *data)
{
struct f2fs_sb_info *sbi = data;
diff --git a/fs/f2fs/cache.h b/fs/f2fs/cache.h
index 6c4db910d767..3df3a562c46f 100644
--- a/fs/f2fs/cache.h
+++ b/fs/f2fs/cache.h
@@ -84,6 +84,8 @@ enum f2fs_cache_request_flag {
#define F2FS_ONSTACK_CACHES (32)
+#define METADATA_CACHE_SHRINK_NUMBER (128)
+
#define F2FS_CACHE_FLAG_TEST_FUNC(name, flagname) \
static inline bool f2fs_cache_test_##name( \
const struct f2fs_cached_block *entry) \
@@ -223,6 +225,10 @@ void f2fs_drop_cache_range(struct f2fs_cached_block_list *cache,
unsigned long f2fs_shrink_cache(struct f2fs_sb_info *sbi,
unsigned long nr_to_scan);
+unsigned long f2fs_shrink_meta_cache(struct f2fs_sb_info *sbi,
+ unsigned long nr_to_scan);
+unsigned long f2fs_shrink_node_cache(struct f2fs_sb_info *sbi,
+ unsigned long nr_to_scan);
#define DEF_DIRTY_CACHE_TIMEOUT 5000
#define MIN_DIRTY_CACHE_TIMEOUT 100
diff --git a/fs/f2fs/node.c b/fs/f2fs/node.c
index be14dfa31e53..7b2ff2c3dd03 100644
--- a/fs/f2fs/node.c
+++ b/fs/f2fs/node.c
@@ -123,6 +123,16 @@ bool f2fs_available_free_memory(struct f2fs_sb_info *sbi, int type)
#else
res = false;
#endif
+ } else if (type == META_BLOCK) {
+ mem_size = (META_CACHE(sbi)->num_entries *
+ (sizeof(struct f2fs_cached_block) +
+ F2FS_BLKSIZE(sbi))) >> PAGE_SHIFT;
+ res = mem_size < ((avail_ram * nm_i->ram_thresh / 100) >> 1);
+ } else if (type == NODE_BLOCK) {
+ mem_size = (NODE_CACHE(sbi)->num_entries *
+ (sizeof(struct f2fs_cached_block) +
+ F2FS_BLKSIZE(sbi))) >> PAGE_SHIFT;
+ res = mem_size < ((avail_ram * nm_i->ram_thresh / 100) >> 1);
} else {
if (!bdi_wb_dirty_exceeded(sbi->sb->s_bdi))
return true;
diff --git a/fs/f2fs/node.h b/fs/f2fs/node.h
index 2704a5c6a54d..ff1f04864391 100644
--- a/fs/f2fs/node.h
+++ b/fs/f2fs/node.h
@@ -158,6 +158,8 @@ enum mem_type {
AGE_EXTENT_CACHE, /* indicates age extent cache */
DISCARD_CACHE, /* indicates memory of cached discard cmds */
COMPRESS_BLOCK, /* indicates memory of cached compressed blocks */
+ META_BLOCK, /* indicates memory of cached meta blocks */
+ NODE_BLOCK, /* indicates memory of cached node blocks */
BASE_CHECK, /* check kernel status */
};
diff --git a/fs/f2fs/segment.c b/fs/f2fs/segment.c
index 794e0d99fc7c..dab66ab3311d 100644
--- a/fs/f2fs/segment.c
+++ b/fs/f2fs/segment.c
@@ -510,6 +510,14 @@ void f2fs_balance_fs_bg(struct f2fs_sb_info *sbi, bool from_bg)
f2fs_shrink_age_extent_tree(sbi,
AGE_EXTENT_CACHE_SHRINK_NUMBER);
+ /* try to shrink meta cache when there is no enough memory */
+ if (!f2fs_available_free_memory(sbi, META_BLOCK))
+ f2fs_shrink_meta_cache(sbi, METADATA_CACHE_SHRINK_NUMBER);
+
+ /* try to shrink node cache when there is no enough memory */
+ if (!f2fs_available_free_memory(sbi, NODE_BLOCK))
+ f2fs_shrink_node_cache(sbi, METADATA_CACHE_SHRINK_NUMBER);
+
/* check the # of cached NAT entries */
if (!f2fs_available_free_memory(sbi, NAT_ENTRIES))
f2fs_try_to_free_nats(sbi, NAT_ENTRY_PER_BLOCK(sbi));
--
2.49.0
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH 3/5] f2fs: cache: wake up f2fs_writeback when exceeding threshold
2026-10-07 11:49 [PATCH 1/5] f2fs: cache: introduce metadata_cache sysfs node Chao Yu
2026-10-07 11:49 ` [PATCH 2/5] f2fs: cache: shrink meta and node caches in f2fs_balance_fs_bg Chao Yu
@ 2026-10-07 11:49 ` Chao Yu
2026-10-07 11:49 ` [PATCH 4/5] f2fs: cache: support asynchronous write_end_io Chao Yu
2026-10-07 11:49 ` [PATCH 5/5] f2fs: introduce max_atc_write_bio_entry_cnt Chao Yu
3 siblings, 0 replies; 5+ messages in thread
From: Chao Yu @ 2026-10-07 11:49 UTC (permalink / raw)
To: jaegeuk; +Cc: linux-f2fs-devel, linux-kernel, Chao Yu
From: Chao Yu <chao@kernel.org>
This patch introduces cache_wb_dirty_threshold and cache_wb_total_threshold
sysfs nodes under /sys/fs/f2fs/<disk>/, and allows waking up the
f2fs_writeback kthread when both thresholds are exceeded.
By default, cache_wb_total_threshold is zero, so if dirty cache number exceeds
cache_wb_dirty_threshold, it will trigger writebacking.
Signed-off-by: Chao Yu <chao@kernel.org>
---
Documentation/ABI/testing/sysfs-fs-f2fs | 14 +++++++
fs/f2fs/cache.c | 50 ++++++++++++++++++++++++-
fs/f2fs/cache.h | 5 +++
fs/f2fs/sysfs.c | 8 +++-
4 files changed, 75 insertions(+), 2 deletions(-)
diff --git a/Documentation/ABI/testing/sysfs-fs-f2fs b/Documentation/ABI/testing/sysfs-fs-f2fs
index 5b196806b3bc..f50739f90ae9 100644
--- a/Documentation/ABI/testing/sysfs-fs-f2fs
+++ b/Documentation/ABI/testing/sysfs-fs-f2fs
@@ -1027,3 +1027,17 @@ Contact: "Chao Yu" <chao@kernel.org>
Description: This is a writable entry to control writeback interval of
f2fs_writeback-x:y, the range is [100, 30000], by default the value
is 5000, unit is ms.
+
+What: /sys/fs/f2fs/<disk>/cache_wb_dirty_threshold
+Date: October 2026
+Contact: "Chao Yu" <chao@kernel.org>
+Description: This is a writable entry for metadata cache, it is used to control
+ dirty cache threshold of to wake up f2fs_writeback-x:y for writeback,
+ by default the value is 8192, unit is blocks.
+
+What: /sys/fs/f2fs/<disk>/cache_wb_total_threshold
+Date: October 2026
+Contact: "Chao Yu" <chao@kernel.org>
+Description: This is a writable entry for metadata cache, it is used to control
+ total cache threshold to wake up f2fs_writeback-x:y for writeback,
+ by default the value is 0, unit is blocks.
diff --git a/fs/f2fs/cache.c b/fs/f2fs/cache.c
index f385fe928b5b..fcccfd3428df 100644
--- a/fs/f2fs/cache.c
+++ b/fs/f2fs/cache.c
@@ -81,6 +81,7 @@ bool f2fs_mark_cache_dirty(struct f2fs_cached_block *entry)
f2fs_cache_update_tag(entry, F2FS_CACHE_TAG_NONE,
F2FS_CACHE_TAG_DIRTY);
inc_cache_count(cache->sbi, type);
+ f2fs_wake_up_cache_wb(cache->sbi);
return true;
}
@@ -681,6 +682,39 @@ unsigned long f2fs_shrink_node_cache(struct f2fs_sb_info *sbi,
return f2fs_do_shrink_cache(NODE_CACHE(sbi), nr_to_scan);
}
+static inline unsigned long f2fs_total_cached_entries(struct f2fs_sb_info *sbi)
+{
+ unsigned long total = META_CACHE(sbi)->num_entries +
+ NODE_CACHE(sbi)->num_entries;
+#ifdef CONFIG_F2FS_FS_COMPRESSION
+ if (test_opt(sbi, COMPRESS_CACHE))
+ total += COMPRESS_CACHE(sbi)->num_entries;
+#endif
+ return total;
+}
+
+static inline bool f2fs_should_wake_up_cache_wb(struct f2fs_sb_info *sbi)
+{
+ struct f2fs_cache_kthread *cache_thread = &sbi->cache_thread;
+ s64 nr_dirty;
+
+ if (!cache_thread->cache_wb_task)
+ return false;
+
+ if (cache_thread->cache_wb_total_threshold &&
+ f2fs_total_cached_entries(sbi) < cache_thread->cache_wb_total_threshold)
+ return false;
+
+ nr_dirty = get_nr_caches(sbi, F2FS_DIRTY_META) +
+ get_nr_caches(sbi, F2FS_DIRTY_NODES);
+
+ if (cache_thread->cache_wb_dirty_threshold &&
+ nr_dirty < cache_thread->cache_wb_dirty_threshold)
+ return false;
+
+ return true;
+}
+
static int f2fs_cache_writeback_kthread(void *data)
{
struct f2fs_sb_info *sbi = data;
@@ -693,7 +727,8 @@ static int f2fs_cache_writeback_kthread(void *data)
unsigned int interval = cache_thread->cache_wb_interval;
wait_event_freezable_timeout(*wq,
- kthread_should_stop(),
+ kthread_should_stop() ||
+ f2fs_should_wake_up_cache_wb(sbi),
msecs_to_jiffies(interval));
if (kthread_should_stop())
@@ -719,6 +754,17 @@ static int f2fs_cache_writeback_kthread(void *data)
return 0;
}
+void f2fs_wake_up_cache_wb(struct f2fs_sb_info *sbi)
+{
+ struct f2fs_cache_kthread *cache_thread = &sbi->cache_thread;
+
+ if (!f2fs_should_wake_up_cache_wb(sbi))
+ return;
+
+ if (wq_has_sleeper(&cache_thread->cache_wb_wq))
+ wake_up(&cache_thread->cache_wb_wq);
+}
+
int f2fs_start_cache_wb_thread(struct f2fs_sb_info *sbi)
{
struct f2fs_cache_kthread *cache_thread = &sbi->cache_thread;
@@ -731,6 +777,8 @@ int f2fs_start_cache_wb_thread(struct f2fs_sb_info *sbi)
init_waitqueue_head(&cache_thread->cache_wb_wq);
cache_thread->cache_wb_interval = DEF_DIRTY_CACHE_TIMEOUT;
+ cache_thread->cache_wb_dirty_threshold = DEF_CACHE_WB_DIRTY_THRESH;
+ cache_thread->cache_wb_total_threshold = 0;
snprintf(name, sizeof(name), "f2fs_writeback-%u:%u",
MAJOR(dev), MINOR(dev));
diff --git a/fs/f2fs/cache.h b/fs/f2fs/cache.h
index 3df3a562c46f..7a94fb8139d3 100644
--- a/fs/f2fs/cache.h
+++ b/fs/f2fs/cache.h
@@ -234,13 +234,18 @@ unsigned long f2fs_shrink_node_cache(struct f2fs_sb_info *sbi,
#define MIN_DIRTY_CACHE_TIMEOUT 100
#define MAX_DIRTY_CACHE_TIMEOUT 30000
+#define DEF_CACHE_WB_DIRTY_THRESH (8192)
+
struct f2fs_cache_kthread {
struct task_struct *cache_wb_task;
wait_queue_head_t cache_wb_wq;
unsigned int cache_wb_interval;
+ unsigned int cache_wb_dirty_threshold;
+ unsigned int cache_wb_total_threshold;
};
int f2fs_start_cache_wb_thread(struct f2fs_sb_info *sbi);
void f2fs_stop_cache_wb_thread(struct f2fs_sb_info *sbi);
+void f2fs_wake_up_cache_wb(struct f2fs_sb_info *sbi);
#endif /* _LINUX_F2FS_CACHE_H */
diff --git a/fs/f2fs/sysfs.c b/fs/f2fs/sysfs.c
index afce74744d4d..95f8dc0c4218 100644
--- a/fs/f2fs/sysfs.c
+++ b/fs/f2fs/sysfs.c
@@ -1019,7 +1019,9 @@ static ssize_t f2fs_sbi_store(struct f2fs_attr *a,
a->struct_type == GC_THREAD);
bool thread_entry = !strcmp(a->attr.name, "ckpt_thread_ioprio") ||
!strcmp(a->attr.name, "critical_task_priority") ||
- !strcmp(a->attr.name, "cache_wb_interval");
+ !strcmp(a->attr.name, "cache_wb_interval") ||
+ !strcmp(a->attr.name, "cache_wb_dirty_threshold") ||
+ !strcmp(a->attr.name, "cache_wb_total_threshold");
if (gc_entry || thread_entry) {
if (!down_read_trylock(&sbi->sb->s_umount))
@@ -1363,6 +1365,8 @@ ATGC_INFO_RW_ATTR(atgc_age_threshold, age_threshold);
/* WB_THREAD ATTR */
WB_THREAD_RW_ATTR(cache_wb_interval, cache_wb_interval);
+WB_THREAD_RW_ATTR(cache_wb_dirty_threshold, cache_wb_dirty_threshold);
+WB_THREAD_RW_ATTR(cache_wb_total_threshold, cache_wb_total_threshold);
F2FS_GENERAL_RO_ATTR(dirty_segments);
F2FS_GENERAL_RO_ATTR(free_segments);
@@ -1552,6 +1556,8 @@ static struct attribute *f2fs_attrs[] = {
ATTR_LIST(adjust_lock_priority),
ATTR_LIST(critical_task_priority),
ATTR_LIST(cache_wb_interval),
+ ATTR_LIST(cache_wb_dirty_threshold),
+ ATTR_LIST(cache_wb_total_threshold),
NULL,
};
ATTRIBUTE_GROUPS(f2fs);
--
2.49.0
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH 4/5] f2fs: cache: support asynchronous write_end_io
2026-10-07 11:49 [PATCH 1/5] f2fs: cache: introduce metadata_cache sysfs node Chao Yu
2026-10-07 11:49 ` [PATCH 2/5] f2fs: cache: shrink meta and node caches in f2fs_balance_fs_bg Chao Yu
2026-10-07 11:49 ` [PATCH 3/5] f2fs: cache: wake up f2fs_writeback when exceeding threshold Chao Yu
@ 2026-10-07 11:49 ` Chao Yu
2026-10-07 11:49 ` [PATCH 5/5] f2fs: introduce max_atc_write_bio_entry_cnt Chao Yu
3 siblings, 0 replies; 5+ messages in thread
From: Chao Yu @ 2026-10-07 11:49 UTC (permalink / raw)
To: jaegeuk; +Cc: linux-f2fs-devel, linux-kernel, Chao Yu
From: Chao Yu <chao@kernel.org>
Currently, f2fs_cache_write_end_io() synchronously traverses all
cached blocks in the write bio to perform node footer sanity checks,
warm node cleanup, and writeback completion. If called from atomic
context (e.g. interrupt context) with a large batch of metadata
blocks, this traversal can induce significant IRQ latency.
Furthermore, f2fs_write_end_io() already implements asynchronous
offloading via workqueue when in atomic context and exceeding the
max_atc_write_bio_size threshold.
To avoid code duplication and mitigate atomic context latency for
metadata cache writes:
1. Unify the write completion handler by pointing all write bios to
f2fs_write_end_io() with bio->bi_private = sbi.
2. In f2fs_write_end_bio(), check f2fs_is_cache_bio(bio) to dispatch
to f2fs_cache_write_end_bio().
3. Simplify f2fs_zone_write_end_io() to call f2fs_write_end_io() directly.
Signed-off-by: Chao Yu <chao@kernel.org>
---
fs/f2fs/data.c | 92 +++++++++++++++++++++++---------------------------
1 file changed, 42 insertions(+), 50 deletions(-)
diff --git a/fs/f2fs/data.c b/fs/f2fs/data.c
index 6d4ba5e77906..0a6aa24caa07 100644
--- a/fs/f2fs/data.c
+++ b/fs/f2fs/data.c
@@ -298,6 +298,40 @@ static void f2fs_read_end_io(struct bio *bio)
f2fs_verify_and_finish_bio(bio, intask);
}
+static void f2fs_cache_write_end_bio(struct bio *bio)
+{
+ struct f2fs_cached_block *entry = F2FS_BIO(bio)->entry;
+ struct f2fs_sb_info *sbi = entry->cache->sbi;
+ struct f2fs_cached_block *next;
+
+ if (bio->bi_status != BLK_STS_OK)
+ f2fs_stop_checkpoint(sbi, true,
+ STOP_CP_REASON_WRITE_FAIL);
+
+ while (entry) {
+ next = entry->next_entry;
+ entry->next_entry = NULL;
+
+ if (f2fs_is_node_cache(entry)) {
+ f2fs_sanity_check_node_footer(sbi, entry,
+ entry->index, NODE_TYPE_REGULAR, true);
+ f2fs_bug_on(sbi, entry->index != nid_of_node(sbi, entry));
+ }
+ if (f2fs_in_warm_node_list(sbi, entry))
+ f2fs_del_fsync_node_entry(sbi, entry);
+
+ dec_cache_count(sbi, F2FS_WB_CP_DATA);
+
+ if (!get_nr_caches(sbi, F2FS_WB_CP_DATA) &&
+ wq_has_sleeper(&sbi->cp_wait))
+ wake_up(&sbi->cp_wait);
+
+ f2fs_end_cache_writeback(entry);
+ entry = next;
+ }
+ bio_put(bio);
+}
+
static void f2fs_write_end_bio(struct bio *bio)
{
struct f2fs_sb_info *sbi = bio->bi_private;
@@ -306,6 +340,11 @@ static void f2fs_write_end_bio(struct bio *bio)
if (time_to_inject(sbi, FAULT_WRITE_IO))
bio->bi_status = BLK_STS_IOERR;
+ if (f2fs_is_cache_bio(bio)) {
+ f2fs_cache_write_end_bio(bio);
+ return;
+ }
+
bio_for_each_folio_all(fi, bio) {
struct folio *folio = fi.folio;
enum count_type type;
@@ -404,45 +443,6 @@ static void f2fs_cache_read_end_io(struct bio *bio)
bio_put(bio);
}
-static void f2fs_cache_write_end_io(struct bio *bio)
-{
- struct f2fs_cached_block *entry = F2FS_BIO(bio)->entry;
- struct f2fs_sb_info *sbi = entry->cache->sbi;
- struct f2fs_cached_block *next;
-
- iostat_update_and_unbind_ctx(bio);
-
- if (time_to_inject(sbi, FAULT_WRITE_IO))
- bio->bi_status = BLK_STS_IOERR;
-
- if (bio->bi_status != BLK_STS_OK)
- f2fs_stop_checkpoint(sbi, true,
- STOP_CP_REASON_WRITE_FAIL);
-
- while (entry) {
- next = entry->next_entry;
- entry->next_entry = NULL;
-
- if (f2fs_is_node_cache(entry)) {
- f2fs_sanity_check_node_footer(sbi, entry,
- entry->index, NODE_TYPE_REGULAR, true);
- f2fs_bug_on(sbi, entry->index != nid_of_node(sbi, entry));
- }
- if (f2fs_in_warm_node_list(sbi, entry))
- f2fs_del_fsync_node_entry(sbi, entry);
-
- dec_cache_count(sbi, F2FS_WB_CP_DATA);
-
- if (!get_nr_caches(sbi, F2FS_WB_CP_DATA) &&
- wq_has_sleeper(&sbi->cp_wait))
- wake_up(&sbi->cp_wait);
-
- f2fs_end_cache_writeback(entry);
- entry = next;
- }
- bio_put(bio);
-}
-
#ifdef CONFIG_BLK_DEV_ZONED
static void f2fs_zone_write_end_io(struct bio *bio)
{
@@ -450,10 +450,7 @@ static void f2fs_zone_write_end_io(struct bio *bio)
bio->bi_private = io->bi_private;
complete(&io->zone_wait);
- if (f2fs_is_cache_bio(bio))
- f2fs_cache_write_end_io(bio);
- else
- f2fs_write_end_io(bio);
+ f2fs_write_end_io(bio);
}
#endif
@@ -548,13 +545,8 @@ static struct bio *__bio_alloc(struct f2fs_io_info *fio, int npages)
else
bio->bi_end_io = f2fs_read_end_io;
} else {
- if (fio->is_cache) {
- bio->bi_end_io = f2fs_cache_write_end_io;
- } else {
- bio->bi_end_io = f2fs_write_end_io;
- bio->bi_private = sbi;
- }
-
+ bio->bi_end_io = f2fs_write_end_io;
+ bio->bi_private = sbi;
bio->bi_write_hint = f2fs_io_type_to_rw_hint(sbi,
fio->type, fio->temp);
bio->bi_write_stream = f2fs_io_type_to_write_stream(bdev, fio->type,
--
2.49.0
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH 5/5] f2fs: introduce max_atc_write_bio_entry_cnt
2026-10-07 11:49 [PATCH 1/5] f2fs: cache: introduce metadata_cache sysfs node Chao Yu
` (2 preceding siblings ...)
2026-10-07 11:49 ` [PATCH 4/5] f2fs: cache: support asynchronous write_end_io Chao Yu
@ 2026-10-07 11:49 ` Chao Yu
3 siblings, 0 replies; 5+ messages in thread
From: Chao Yu @ 2026-10-07 11:49 UTC (permalink / raw)
To: jaegeuk; +Cc: linux-f2fs-devel, linux-kernel, Chao Yu
From: Chao Yu <chao@kernel.org>
Commit 3de6b8094115 ("f2fs: Run f2fs_write_end_io() asynchronously")
introduced max_atc_write_bio_size to offload write bio completion to a
workqueue when f2fs_write_end_io() is called in atomic context.
However, using the bio byte size (bi_iter.bi_size) as the threshold has
limitations once large folios or multi-page bvecs are involved:
1. In f2fs_write_end_bio(), execution time is dominated by the
completion loop (bio_for_each_folio_all() for pagecache folios,
or while(entry) for cached metadata blocks), where per-entry
locking, status clearing, sanity checks, and writeback end are
performed. The CPU latency in atomic context scales with the number
of entries in the bio, not its byte size.
2. For large folios, a large bio (e.g. 1MB comprised of single large
folios) only iterates a few times through the loop. Throttling solely
by max_atc_write_bio_size unnecessarily offloads such low-latency
bios to the workqueue, adding context-switch overhead.
3. Conversely, relying on bio->bi_vcnt is insufficient because the block
layer (bvec_try_merge_page) merges physically consecutive pages/folios
into the same bio_vec segment. Thus, a bio with bi_vcnt == 1 could
still contain multiple entries and iterate many times in end_io.
To address this, track the exact number of entries (folios or cached
blocks) queued to the write bio via entry_cnt in struct f2fs_bio, and
introduce a sysfs attribute max_atc_write_bio_entry_cnt. This enables
precise control over atomic context latency without penalizing large
folios.
Signed-off-by: Chao Yu <chao@kernel.org>
---
Documentation/ABI/testing/sysfs-fs-f2fs | 11 +++++++++++
fs/f2fs/data.c | 7 ++++++-
fs/f2fs/f2fs.h | 3 +++
fs/f2fs/super.c | 1 +
fs/f2fs/sysfs.c | 2 ++
5 files changed, 23 insertions(+), 1 deletion(-)
diff --git a/Documentation/ABI/testing/sysfs-fs-f2fs b/Documentation/ABI/testing/sysfs-fs-f2fs
index f50739f90ae9..5ff693dd4bff 100644
--- a/Documentation/ABI/testing/sysfs-fs-f2fs
+++ b/Documentation/ABI/testing/sysfs-fs-f2fs
@@ -1014,6 +1014,17 @@ Description: Every time a write operation completes f2fs_write_end_io() is
(atc) context. The default value for this attribute is UINT_MAX
which means that this functionality is disabled by default.
+What: /sys/fs/f2fs/<disk>/max_atc_write_bio_entry_cnt
+Date: October 2026
+Contact: Chao Yu <chao@kernel.org>
+Description: Every time a write operation completes f2fs_write_end_io() is
+ called. This function may be called from an atomic context,
+ e.g. from inside an interrupt handler. This attribute controls
+ the maximum number of entries (folios or cached blocks) in a
+ write bio that is completed in atomic (atc) context. The default
+ value for this attribute is UINT_MAX which means that this
+ functionality is disabled by default.
+
What: /sys/fs/f2fs/<disk>/pinned_area_max_secno
Date: August 2026
Contact: "Daeho Jeong" <daehojeong@google.com>
diff --git a/fs/f2fs/data.c b/fs/f2fs/data.c
index 0a6aa24caa07..395794d00194 100644
--- a/fs/f2fs/data.c
+++ b/fs/f2fs/data.c
@@ -398,7 +398,8 @@ static void f2fs_write_end_io(struct bio *bio)
sbi = bio->bi_private;
- if (in_atomic() && bio->bi_iter.bi_size > sbi->max_atc_write_bio_size) {
+ if (in_atomic() && (bio->bi_iter.bi_size > sbi->max_atc_write_bio_size ||
+ F2FS_BIO(bio)->entry_cnt > sbi->max_atc_write_bio_entry_cnt)) {
struct work_struct *w;
w = &container_of(bio, struct f2fs_bio, bio)->work;
@@ -551,6 +552,7 @@ static struct bio *__bio_alloc(struct f2fs_io_info *fio, int npages)
fio->type, fio->temp);
bio->bi_write_stream = f2fs_io_type_to_write_stream(bdev, fio->type,
fio->temp);
+ F2FS_BIO(bio)->entry_cnt = 0;
}
iostat_alloc_and_bind_ctx(sbi, bio, NULL);
@@ -1218,6 +1220,8 @@ void f2fs_submit_page_write(struct f2fs_io_info *fio)
goto alloc_new;
}
+ F2FS_BIO(io->bio)->entry_cnt++;
+
if (fio->io_wbc)
wbc_account_cgroup_owner(fio->io_wbc, fio->folio,
folio_size(fio->folio));
@@ -1324,6 +1328,7 @@ void f2fs_submit_cache_write(struct f2fs_io_info *fio)
}
f2fs_bio_add_cache(fio, io->bio);
+ F2FS_BIO(io->bio)->entry_cnt++;
io->last_block_in_bio = fio->new_blkaddr;
diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h
index 089a62c054ea..0acbc4d299af 100644
--- a/fs/f2fs/f2fs.h
+++ b/fs/f2fs/f2fs.h
@@ -1778,6 +1778,7 @@ struct f2fs_gc_kthread {
struct f2fs_bio {
struct work_struct work;
struct f2fs_cached_block *entry;
+ unsigned int entry_cnt;
struct bio bio;
};
@@ -1807,6 +1808,8 @@ struct f2fs_sb_info {
/* for bio operations */
/* Largest write bio size completed in atomic context (atc). */
u32 max_atc_write_bio_size;
+ /* Largest write bio entry count completed in atomic context (atc). */
+ u32 max_atc_write_bio_entry_cnt;
struct f2fs_bio_info *write_io[NR_PAGE_TYPE]; /* for write bios */
/* keep migration IO order for LFS mode */
struct f2fs_rwsem io_order_lock;
diff --git a/fs/f2fs/super.c b/fs/f2fs/super.c
index d683240040f0..9e55febe18f4 100644
--- a/fs/f2fs/super.c
+++ b/fs/f2fs/super.c
@@ -5211,6 +5211,7 @@ static int f2fs_fill_super(struct super_block *sb, struct fs_context *fc)
goto free_sb_buf;
}
sbi->max_atc_write_bio_size = UINT_MAX;
+ sbi->max_atc_write_bio_entry_cnt = UINT_MAX;
INIT_WORK(&sbi->s_error_work, f2fs_record_error_work);
memcpy(sbi->errors, raw_super->s_errors, MAX_F2FS_ERRORS);
diff --git a/fs/f2fs/sysfs.c b/fs/f2fs/sysfs.c
index 95f8dc0c4218..32dfd564094b 100644
--- a/fs/f2fs/sysfs.c
+++ b/fs/f2fs/sysfs.c
@@ -1285,6 +1285,7 @@ F2FS_SBI_RW_ATTR(umount_discard_timeout, interval_time[UMOUNT_DISCARD_TIMEOUT]);
F2FS_SBI_RW_ATTR(gc_pin_file_thresh, gc_pin_file_threshold);
F2FS_SBI_RW_ATTR(gc_reclaimed_segments, gc_reclaimed_segs);
F2FS_SBI_RW_ATTR(max_atc_write_bio_size, max_atc_write_bio_size);
+F2FS_SBI_RW_ATTR(max_atc_write_bio_entry_cnt, max_atc_write_bio_entry_cnt);
F2FS_SBI_GENERAL_RW_ATTR(max_victim_search);
F2FS_SBI_GENERAL_RW_ATTR(migration_granularity);
F2FS_SBI_GENERAL_RW_ATTR(migration_window_granularity);
@@ -1536,6 +1537,7 @@ static struct attribute *f2fs_attrs[] = {
ATTR_LIST(gc_segment_mode),
ATTR_LIST(gc_reclaimed_segments),
ATTR_LIST(max_atc_write_bio_size),
+ ATTR_LIST(max_atc_write_bio_entry_cnt),
ATTR_LIST(max_fragment_chunk),
ATTR_LIST(max_fragment_hole),
ATTR_LIST(current_atomic_write),
--
2.49.0
^ permalink raw reply [flat|nested] 5+ messages in thread
end of thread, other threads:[~2026-10-07 11:50 UTC | newest]
Thread overview: 5+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-07 11:49 [PATCH 1/5] f2fs: cache: introduce metadata_cache sysfs node Chao Yu
2026-10-07 11:49 ` [PATCH 2/5] f2fs: cache: shrink meta and node caches in f2fs_balance_fs_bg Chao Yu
2026-10-07 11:49 ` [PATCH 3/5] f2fs: cache: wake up f2fs_writeback when exceeding threshold Chao Yu
2026-10-07 11:49 ` [PATCH 4/5] f2fs: cache: support asynchronous write_end_io Chao Yu
2026-10-07 11:49 ` [PATCH 5/5] f2fs: introduce max_atc_write_bio_entry_cnt Chao Yu
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®